{"meta":{"query_hash":"09153c5ce259","filters":{"venue":"Assessment & Evaluation in Higher Education"},"cohort_total":44,"direct_labels_cover":1,"predictions_cover":44,"exported":44,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/09153c5ce259","api":"https://metacan.xera.ac/api/v1/cohort?venue=Assessment+%26+Evaluation+in+Higher+Education"},"results":[{"id":"W1841123149","doi":"10.1080/02602938.2015.1044421","title":"Whose feedback? A multilevel analysis of student completion of end-of-term teaching evaluations","year":2015,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Respondent; Set (abstract data type); Psychology; Quality (philosophy); Quality assurance; Term (time); Higher education; Medical education; Multilevel model; Course evaluation; Mathematics education; Computer science; Medicine; Political science","score_opus":0.4085236037266901,"score_gpt":0.6050658361994158,"score_spread":0.19654223247272568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1841123149","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97589374,0.00019351921,0.0014961567,0.0029880302,0.0011512167,0.0011935065,0.000014843736,0.000024334066,0.017044643],"genre_scores_gemma":[0.9876195,0.000019169747,0.011264879,0.000043332482,0.00010803782,0.00023433179,0.00027977093,0.000012422744,0.0004185548],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9906278,0.0043634544,0.001189572,0.00033802065,0.0032902234,0.00019093086],"domain_scores_gemma":[0.99432796,0.0010113912,0.001739525,0.0004280951,0.0023927784,0.000100239195],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.020404484,0.00015029029,0.00044517432,0.001184331,0.00016638577,0.000048516773,0.00036287852,0.00011301509,0.0014754601],"category_scores_gemma":[0.0013446012,0.00016641102,0.00013968913,0.0012466523,0.0001835577,0.00085715187,0.000053618565,0.00021624134,0.000005868434],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029669938,0.0028336751,0.80011433,0.00004083569,0.00046360792,4.5265654e-8,0.050819125,0.018816503,0.0016478924,0.089051545,0.00037652682,0.035806242],"study_design_scores_gemma":[0.00083436124,0.00009101739,0.9699257,0.00009525783,0.0015793851,6.684686e-8,0.010898855,0.013363497,0.000066086424,0.002276757,0.0007276679,0.0001413169],"about_ca_topic_score_codex":0.0047745625,"about_ca_topic_score_gemma":0.0010902109,"teacher_disagreement_score":0.1698114,"about_ca_system_score_codex":0.0011636117,"about_ca_system_score_gemma":0.0035329217,"threshold_uncertainty_score":0.99943733},"labels":[],"label_agreement":null},{"id":"W1904991648","doi":"10.1080/02602938.2015.1089977","title":"New assessment process in an introductory undergraduate physics laboratory: an exploration on collaborative learning","year":2015,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Innovative Teaching Methods","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Grading (engineering); Session (web analytics); Peer assessment; Process (computing); Peer evaluation; Mathematics education; Incentive; Class (philosophy); Task (project management); Higher education; Medical education; Psychology; Computer science; Engineering","score_opus":0.24140766209384704,"score_gpt":0.5511415627937231,"score_spread":0.30973390069987605,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1904991648","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95375264,0.000040376,0.007727033,0.012190418,0.003373353,0.002074983,0.0000028317788,0.00016011832,0.02067825],"genre_scores_gemma":[0.9776131,0.000008850507,0.019147005,0.00027167093,0.0014144045,0.0006292635,0.00032598595,0.000036020643,0.00055367965],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9880824,0.008701549,0.0005193431,0.0006352573,0.0017244704,0.00033697335],"domain_scores_gemma":[0.9964725,0.00018315011,0.0004804018,0.00033661776,0.0023315095,0.0001958436],"candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.011573677,0.00023910904,0.0002560586,0.00042173945,0.0003082854,0.00030712594,0.00026931867,0.00014560728,0.00016375326],"category_scores_gemma":[0.0005361982,0.00027038905,0.000017668252,0.0027851088,0.000080536665,0.004420137,0.000019084237,0.00061435427,0.000014118046],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":true,"study_design_scores_codex":[0.0000728856,0.0020556313,0.30976412,0.000020025518,0.000019952196,6.184325e-7,0.07975242,0.05667152,0.00096807664,0.50760734,0.001308866,0.04175853],"study_design_scores_gemma":[0.0026665996,0.0010862239,0.6180264,0.00016920247,0.00006212309,9.6310615e-8,0.09625308,0.008683151,0.00037683788,0.26292923,0.008869864,0.0008772511],"about_ca_topic_score_codex":0.00064951164,"about_ca_topic_score_gemma":0.0010085929,"teacher_disagreement_score":0.30826223,"about_ca_system_score_codex":0.004259724,"about_ca_system_score_gemma":0.015309599,"threshold_uncertainty_score":0.99997485},"labels":[],"label_agreement":null},{"id":"W1963676041","doi":"10.1080/02602930600760884","title":"Reflections on using journals in higher education: a focus group discussion with faculty","year":2006,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Reflective Practices in Education","field":"Social Sciences","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lakehead University","funders":"Lakehead University","keywords":"Journaling file system; Reflective writing; Focus group; Higher education; Psychology; Journal writing; Medical education; Writing process; Pedagogy; Experiential learning; Professional writing; Narrative; Mathematics education; Teaching method; Sociology; Medicine; Computer science; Political science","score_opus":0.26059440791254546,"score_gpt":0.5788218747837022,"score_spread":0.3182274668711567,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1963676041","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4217473,0.0004013857,0.0002610183,0.061642356,0.008151811,0.0025248392,0.0000039531346,0.00008770179,0.50517964],"genre_scores_gemma":[0.9781502,0.000026434454,0.00897337,0.00036324735,0.0019312601,0.0011388914,0.0001534242,0.000035561654,0.009227595],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99482125,0.0018647705,0.00068188633,0.0006104522,0.001607414,0.00041423074],"domain_scores_gemma":[0.9978522,0.00020491124,0.0006881005,0.00037158426,0.0007719143,0.00011127459],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003145733,0.00025512354,0.00023183653,0.00090409996,0.000639004,0.00039377224,0.0002510124,0.00020149081,0.0037032345],"category_scores_gemma":[0.00011894605,0.00020873934,0.000056552723,0.0023903148,0.000112891605,0.0019461163,0.000022654644,0.00048869825,0.000030883413],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015472544,0.010628728,0.51181185,0.00003833056,0.00004901589,0.0000010627348,0.009958451,0.008222542,0.0024376353,0.38106963,0.0054359525,0.070192054],"study_design_scores_gemma":[0.00061356934,0.000096168194,0.8716769,0.00029101482,0.000079155696,0.0000019882561,0.005174669,0.0001334738,0.000028068902,0.06001298,0.06157646,0.00031550508],"about_ca_topic_score_codex":0.007600913,"about_ca_topic_score_gemma":0.0060100076,"teacher_disagreement_score":0.5564029,"about_ca_system_score_codex":0.006410132,"about_ca_system_score_gemma":0.005192076,"threshold_uncertainty_score":0.9990076},"labels":[],"label_agreement":null},{"id":"W1964857994","doi":"10.1080/02602938.2013.845647","title":"Student personality differences are related to their responses on instructor evaluation forms","year":2013,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Cape Breton University","funders":"","keywords":"Agreeableness; Conscientiousness; Neuroticism; Psychology; Extraversion and introversion; Big Five personality traits; Personality; Hierarchical structure of the Big Five; Openness to experience; Regression analysis; Social psychology; Statistics; Mathematics","score_opus":0.22323133286836125,"score_gpt":0.5409676873858742,"score_spread":0.3177363545175129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1964857994","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91047156,0.00028235212,0.0000054902475,0.03537015,0.0041860975,0.0037451999,0.000011036495,0.00007443838,0.045853693],"genre_scores_gemma":[0.9811028,0.000020000043,0.00037180338,0.0021487833,0.00017070606,0.010351756,0.00027867194,0.000030072828,0.005525397],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.994223,0.0025747118,0.000911971,0.0006527987,0.0012930792,0.00034441752],"domain_scores_gemma":[0.99621075,0.0005333487,0.00052688195,0.0011202169,0.0014135833,0.00019523183],"candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0031640618,0.00028055772,0.000283789,0.0006119697,0.00027331262,0.00016240863,0.0004624028,0.00019856653,0.04726952],"category_scores_gemma":[0.00018198114,0.00025158157,0.00007534096,0.00077898975,0.00005447914,0.00036269598,0.000058745904,0.00040382685,0.00087326765],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000055861292,0.0020476612,0.8340484,0.000023885419,0.000058863752,5.7288357e-8,0.02030395,0.00005608205,0.00010824301,0.050348844,0.007933676,0.085014485],"study_design_scores_gemma":[0.00086095196,0.00016243319,0.9543015,0.00011280588,0.000037451893,0.0000011890363,0.02879446,0.00030476972,0.0000073346137,0.012221097,0.002937984,0.00025802746],"about_ca_topic_score_codex":0.00067002396,"about_ca_topic_score_gemma":0.00017959317,"teacher_disagreement_score":0.1202531,"about_ca_system_score_codex":0.001984492,"about_ca_system_score_gemma":0.0012842995,"threshold_uncertainty_score":0.9999936},"labels":[],"label_agreement":null},{"id":"W1977788398","doi":"10.1080/02602930903337612","title":"Student satisfaction with Canadian music programmes: the application of the American Customer Satisfaction Model in higher education","year":2010,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Customer Service Quality and Loyalty","field":"Business, Management and Accounting","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Lakehead University","funders":"","keywords":"Loyalty; Customer satisfaction; Psychology; Higher education; Word of mouth; Marketing; Quality (philosophy); Minor (academic); Social psychology; Advertising; Business; Political science","score_opus":0.04703466291038546,"score_gpt":0.3601511729886637,"score_spread":0.3131165100782783,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1977788398","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9773761,0.00002193845,0.000035974223,0.012111145,0.0017226874,0.0022626966,0.0000021373285,0.000028503053,0.0064388323],"genre_scores_gemma":[0.9937403,0.00000331853,0.0004034633,0.0027239237,0.0005863154,0.0020370146,0.000113909766,0.000029931509,0.00036179283],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9975044,0.00016854772,0.0006228954,0.00043033322,0.000993664,0.00028014835],"domain_scores_gemma":[0.9976273,0.00005964473,0.00090111245,0.00070422445,0.0006766936,0.000031009582],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018135122,0.00023095134,0.00023004289,0.0005255939,0.00023537198,0.00021416727,0.00031109402,0.00009609319,0.00062680297],"category_scores_gemma":[0.000020985726,0.00016442063,0.000056929162,0.0019527597,0.00011298519,0.0009765695,0.000042488904,0.00045772683,0.000039491923],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000016399268,0.00038964697,0.895443,0.000058118552,0.00001572543,2.4640364e-8,0.00030352036,0.0033155945,0.00044205587,0.03739421,0.00056610996,0.0620556],"study_design_scores_gemma":[0.000356167,0.0000108922195,0.9804933,0.000047879097,0.00011792787,4.352682e-7,0.0012254344,0.009039901,0.0000067994365,0.003094138,0.0054075336,0.000199591],"about_ca_topic_score_codex":0.32334045,"about_ca_topic_score_gemma":0.64693606,"teacher_disagreement_score":0.3235956,"about_ca_system_score_codex":0.0007393732,"about_ca_system_score_gemma":0.00170282,"threshold_uncertainty_score":0.6863053},"labels":[],"label_agreement":null},{"id":"W1981924652","doi":"10.1080/02602938.2011.630977","title":"Assessing the psychometric properties of Kember and Leung’s Reflection Questionnaire","year":2011,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Reflective Practices in Education","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Western University","funders":"","keywords":"Psychology; Reflective thinking; Confirmatory factor analysis; Test (biology); Construct validity; Critical thinking; Reflection (computer programming); Construct (python library); Nurse education; Reliability (semiconductor); Reflective practice; Medical education; Psychometrics; Pedagogy; Medicine; Structural equation modeling; Clinical psychology","score_opus":0.3077723684892498,"score_gpt":0.5370573176723951,"score_spread":0.22928494918314524,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1981924652","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.831743,0.001245291,0.00014890768,0.0034412155,0.004223568,0.00087438076,1.3156269e-7,0.000029306419,0.15829422],"genre_scores_gemma":[0.9953549,0.00019355437,0.002455517,0.000104767125,0.00027728354,0.0004929447,0.000004168425,0.000011815917,0.0011050695],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9967261,0.0017205435,0.0003804612,0.00027015008,0.0007390037,0.00016374594],"domain_scores_gemma":[0.99839777,0.00009824097,0.00044860263,0.00023222715,0.0007809096,0.000042263797],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047967155,0.00010527977,0.000113250026,0.00039427954,0.00038303665,0.0001571332,0.00015331806,0.00009627676,0.0004605447],"category_scores_gemma":[0.0005938518,0.00008629344,0.000025703732,0.0013553884,0.00020994552,0.001882819,0.000019173995,0.00018875007,0.0000059224167],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014966218,0.00080270914,0.77585125,0.000042626725,0.000026601596,2.4967678e-8,0.030871095,0.000015203823,0.0014023831,0.056762844,0.00033242424,0.13387787],"study_design_scores_gemma":[0.00015431973,0.00003147993,0.9691465,0.000106874264,0.00006503098,6.830927e-7,0.008891859,0.00012906812,0.00025590727,0.018014085,0.0031025666,0.00010164344],"about_ca_topic_score_codex":0.0035289656,"about_ca_topic_score_gemma":0.00040046423,"teacher_disagreement_score":0.19329524,"about_ca_system_score_codex":0.00083901297,"about_ca_system_score_gemma":0.0016737898,"threshold_uncertainty_score":0.5334764},"labels":[],"label_agreement":null},{"id":"W1996403924","doi":"10.1080/02602930701772788","title":"An investigation into electronic‐source plagiarism in a first‐year essay assignment","year":2008,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Academic integrity and plagiarism","field":"Social Sciences","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Inyuvesi Yakwazulu-Natali","keywords":"Acknowledgement; Ignorance; The Internet; Quarter (Canadian coin); Perception; Psychology; Electronic publishing; Sociology; Pedagogy; Mathematics education; Public relations; Political science; Computer science; World Wide Web; History; Law","score_opus":0.049372582319956566,"score_gpt":0.3927665071328492,"score_spread":0.34339392481289266,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996403924","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.971504,0.00016776529,0.00088661903,0.014612028,0.0009985224,0.0010566999,8.386366e-7,0.000056658075,0.010716898],"genre_scores_gemma":[0.9937228,0.0001769725,0.0019903213,0.0006270838,0.000500657,0.0005546119,0.00017512019,0.000016593665,0.0022358517],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9956521,0.0017382799,0.00048484627,0.00043279867,0.001289792,0.00040218362],"domain_scores_gemma":[0.9989063,0.00021835907,0.00022331078,0.00024610414,0.00027799106,0.0001279337],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0064811865,0.00016110564,0.00016621705,0.0003462539,0.00046109012,0.00007014809,0.00029228049,0.0006216856,0.0011897688],"category_scores_gemma":[0.00018906988,0.00018386378,0.00003403608,0.00073452084,0.00015136092,0.00081306475,0.00001935646,0.0010871487,0.000045573823],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":true,"study_design_scores_codex":[0.000023495715,0.0006324825,0.7560186,0.000015238068,0.000013453744,5.481709e-7,0.09498713,0.0035015247,0.00033075857,0.13416928,0.00407434,0.0062331567],"study_design_scores_gemma":[0.0008758753,0.00011647082,0.8673861,0.00009988211,0.00003116811,0.0000012021563,0.01249727,0.0047853366,0.000087628025,0.07870213,0.035038855,0.00037806985],"about_ca_topic_score_codex":0.011687872,"about_ca_topic_score_gemma":0.005671833,"teacher_disagreement_score":0.11136752,"about_ca_system_score_codex":0.0038360432,"about_ca_system_score_gemma":0.005797246,"threshold_uncertainty_score":0.9999881},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low"},{"model":"gpt","categories":["research_integrity"],"domain":null,"study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W2029621922","doi":"10.1080/02602930802563094","title":"Web‐based student feedback: comparing teaching‐award and research‐award recipients","year":2009,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"MacEwan University","funders":"","keywords":"Helpfulness; CLARITY; Psychology; Credibility; Competence (human resources); Medical education; Personality; Mathematics education; Pedagogy; Social psychology; Medicine; Political science","score_opus":0.3379119990917349,"score_gpt":0.600605115458447,"score_spread":0.26269311636671205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2029621922","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8246219,0.00021515663,0.000064797634,0.08230612,0.00228368,0.0013542309,8.5169376e-7,0.000099356286,0.0890539],"genre_scores_gemma":[0.99098307,0.000050019462,0.0044376925,0.00067620544,0.0004950684,0.00028345027,0.00005784664,0.00001598069,0.0030006797],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9878934,0.006970679,0.0005792266,0.00058809534,0.0035086789,0.00045993528],"domain_scores_gemma":[0.99732846,0.00092444837,0.00035848087,0.0003850694,0.00082398555,0.00017956343],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.030781327,0.00017927666,0.00022688891,0.00064730673,0.0011195978,0.00056150433,0.00040271124,0.00014033276,0.0006716619],"category_scores_gemma":[0.0012208705,0.00019823897,0.000043436223,0.0006200997,0.00014190312,0.0012252111,0.00005428103,0.00077008805,0.0000606707],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004236468,0.002414393,0.78577703,0.000018303868,0.000028420343,5.190512e-7,0.015360124,0.0011002123,0.00047720628,0.12437671,0.011701827,0.0587029],"study_design_scores_gemma":[0.0010190558,0.000116561016,0.9214819,0.00013586394,0.000050125254,2.1420848e-7,0.0038805665,0.003702705,0.000006087618,0.0071520437,0.06225311,0.00020174822],"about_ca_topic_score_codex":0.0014800242,"about_ca_topic_score_gemma":0.0005430199,"teacher_disagreement_score":0.16636114,"about_ca_system_score_codex":0.0024743685,"about_ca_system_score_gemma":0.0027073114,"threshold_uncertainty_score":0.99801457},"labels":[],"label_agreement":null},{"id":"W2043118965","doi":"10.1080/02602930802082228","title":"What do students consider useful about student ratings?","year":2008,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Psychology; Ranking (information retrieval); Reliability (semiconductor); Applied psychology; Variance (accounting); Higher education; Mathematics education; Computer science","score_opus":0.27636819001779894,"score_gpt":0.581646382299723,"score_spread":0.30527819228192404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043118965","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9443405,0.0012314204,0.000021459029,0.01964658,0.0076768836,0.001673771,7.5263654e-7,0.00009259496,0.025315989],"genre_scores_gemma":[0.98513126,0.0011275072,0.0015630426,0.0012374187,0.0006356651,0.0007729402,0.000050916868,0.000023585531,0.009457642],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9911293,0.0032607806,0.0007092284,0.0005427483,0.004008668,0.000349286],"domain_scores_gemma":[0.99708986,0.0007422422,0.0006248961,0.00041477702,0.000990411,0.00013781669],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.009992114,0.00020148241,0.00022499978,0.0003152441,0.0008794519,0.00083415606,0.00049012643,0.00014486667,0.00695002],"category_scores_gemma":[0.0006752294,0.00022190987,0.00006383234,0.00057285174,0.00017670085,0.0036626114,0.000075690536,0.00034039663,0.00021576927],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000007724738,0.0013154832,0.93126684,0.000008316024,0.00004052494,0.0000010187691,0.034585368,0.00040418585,0.00004423017,0.020246912,0.005934813,0.0061445977],"study_design_scores_gemma":[0.000779366,0.000040541057,0.9309947,0.00013097544,0.00007495368,0.0000013461544,0.019417726,0.00006560029,0.000008892374,0.0025702142,0.045680344,0.00023533043],"about_ca_topic_score_codex":0.0009452125,"about_ca_topic_score_gemma":0.00046879458,"teacher_disagreement_score":0.040790733,"about_ca_system_score_codex":0.0017319183,"about_ca_system_score_gemma":0.0034714644,"threshold_uncertainty_score":0.99395776},"labels":[],"label_agreement":null},{"id":"W2055314676","doi":"10.1080/02602930601122555","title":"Assessment purposes and procedures in ESL/EFL classrooms","year":2007,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta; Queen's University","funders":"Beijing Foreign Studies University","keywords":"Psychology; Mathematics education; Pedagogy; Teaching method; Linguistics","score_opus":0.07152919239186825,"score_gpt":0.4767516717717072,"score_spread":0.4052224793798389,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2055314676","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8048983,0.0004378584,0.0000853541,0.004960719,0.0018986425,0.0017855255,0.0000010021623,0.000055424967,0.18587713],"genre_scores_gemma":[0.9912133,0.00022335214,0.0028443325,0.00036141585,0.0004887061,0.0005173077,0.000048357066,0.000019284269,0.004283909],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9964021,0.00050533336,0.0006263299,0.0005043023,0.0014640965,0.00049784366],"domain_scores_gemma":[0.99874437,0.00034475356,0.00025407106,0.00021322389,0.00031364302,0.00012996806],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0063800155,0.00020665421,0.00023295623,0.0005510771,0.00026642383,0.00023436971,0.0002334461,0.00017589876,0.0018266847],"category_scores_gemma":[0.00011915034,0.00022123841,0.000040409108,0.0009773329,0.00013235351,0.0007661438,0.00005259586,0.00027414254,0.000014492205],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000106116195,0.00067643286,0.9030169,0.000025293697,0.000012681162,8.2166866e-7,0.0040269494,0.000033707747,0.0002119078,0.072178595,0.00071056484,0.019095497],"study_design_scores_gemma":[0.00083870126,0.000052488576,0.96646416,0.00008957901,0.000036435198,3.7331196e-7,0.013483385,0.00018430372,0.000007552866,0.00654562,0.012054151,0.00024322125],"about_ca_topic_score_codex":0.0011585525,"about_ca_topic_score_gemma":0.0063330876,"teacher_disagreement_score":0.18631499,"about_ca_system_score_codex":0.0016661279,"about_ca_system_score_gemma":0.002593879,"threshold_uncertainty_score":0.9990858},"labels":[],"label_agreement":null},{"id":"W2060845473","doi":"10.1080/02602930701772754","title":"Telling the second half of the story: linking academic development to student experience of learning","year":2008,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Psychology; Value (mathematics); Higher education; Pedagogy; Mathematics education; Computer science; Political science","score_opus":0.258701240075958,"score_gpt":0.5317956474131663,"score_spread":0.2730944073372083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2060845473","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98244053,0.00021700353,0.00012225832,0.008902817,0.0013232551,0.0008868166,2.0334153e-7,0.000014374307,0.0060927467],"genre_scores_gemma":[0.9935571,0.000029997165,0.0017170844,0.00019148877,0.00014479239,0.00029603732,0.0000044977232,0.000008695887,0.0040503293],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9947731,0.002401926,0.00052229315,0.0002255195,0.0019119972,0.00016516446],"domain_scores_gemma":[0.9978186,0.00072240987,0.0007017745,0.00022712373,0.00049060676,0.000039517196],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009443984,0.0000942645,0.00012682128,0.0001371393,0.0008116438,0.000027562652,0.00051009766,0.00007507235,0.0007744054],"category_scores_gemma":[0.0006065645,0.000072764284,0.000035519566,0.0005793218,0.00013554031,0.00040853204,0.00007353123,0.00044842428,0.000009506915],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000057453476,0.00020183019,0.6434925,0.00001786293,0.000018714793,3.504632e-8,0.32868147,0.0089352345,0.002175016,0.00846707,0.00017380826,0.007830717],"study_design_scores_gemma":[0.00014331515,0.000019552908,0.8957646,0.00012460903,0.000021388916,2.9666063e-7,0.029688567,0.00016996398,0.0008863327,0.00020753958,0.07287934,0.00009452236],"about_ca_topic_score_codex":0.00030902363,"about_ca_topic_score_gemma":0.00029655106,"teacher_disagreement_score":0.2989929,"about_ca_system_score_codex":0.0007343085,"about_ca_system_score_gemma":0.0028100992,"threshold_uncertainty_score":0.8479196},"labels":[],"label_agreement":null},{"id":"W2071016044","doi":"10.1080/02602930500260688","title":"Ratings of university teacher instruction: how much do student and course characteristics really matter?","year":2005,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":119,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Psychology; Mathematics education; Class (philosophy); Higher education; Student teacher; Medical education; Teacher education; Medicine; Computer science","score_opus":0.0842783477103564,"score_gpt":0.4658851886852968,"score_spread":0.3816068409749404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2071016044","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91867566,0.000074991316,0.000027739034,0.056018088,0.00076185755,0.0005432308,0.00000223553,0.00002477302,0.023871433],"genre_scores_gemma":[0.98952025,0.00008000387,0.0038223164,0.00015619866,0.0003747596,0.000027678867,0.000034914636,0.000009154647,0.0059747407],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99661463,0.0016694214,0.0002902458,0.0002699414,0.0010127398,0.0001430338],"domain_scores_gemma":[0.9984021,0.0002125419,0.00057397666,0.00019432111,0.0005502325,0.000066779736],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0051904228,0.000106735046,0.0001476967,0.00017004955,0.00026723708,0.00014179856,0.00017016826,0.00009600392,0.0023000122],"category_scores_gemma":[0.00016574956,0.00012554537,0.000025275285,0.000281424,0.00011955376,0.0011734776,0.00003520302,0.00018945264,0.000014216506],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000020274905,0.00077556056,0.88786495,0.000021691329,0.000032764878,1.3647954e-7,0.015866242,0.000042133786,0.00019456477,0.035349604,0.002755075,0.057077006],"study_design_scores_gemma":[0.0005164538,0.000028541614,0.95310426,0.00004244875,0.00010322987,5.2022864e-7,0.01525056,0.00023706336,0.0000065463096,0.00042349542,0.030167786,0.00011908463],"about_ca_topic_score_codex":0.00049589196,"about_ca_topic_score_gemma":0.00027420843,"teacher_disagreement_score":0.07084458,"about_ca_system_score_codex":0.0008246859,"about_ca_system_score_gemma":0.001359811,"threshold_uncertainty_score":0.99861205},"labels":[],"label_agreement":null},{"id":"W2073387991","doi":"10.1080/02602938.2014.911244","title":"Record of assessment moderation practice (RAMP): survey software as a mechanism of continuous quality improvement","year":2014,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Alberta","keywords":"Moderation; Quality (philosophy); Identification (biology); Quality management; Process (computing); Unit (ring theory); Psychology; Computer science; Process management; Applied psychology; Operations management; Engineering; Mathematics education; Social psychology","score_opus":0.10712062213460302,"score_gpt":0.49623219056414075,"score_spread":0.38911156842953776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2073387991","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.90695465,0.00006535163,0.015657438,0.0036237554,0.0039407494,0.0032015748,0.000015052856,0.00006650467,0.06647494],"genre_scores_gemma":[0.9796294,0.000072847775,0.017741136,0.00027871892,0.00021321097,0.00051930896,0.00025808156,0.000020248342,0.0012670291],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9916078,0.004005195,0.001222615,0.0004851109,0.0023705326,0.00030875296],"domain_scores_gemma":[0.99407315,0.0015106013,0.0014783954,0.00043912823,0.0024059003,0.00009283395],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.016967416,0.00021167126,0.00044098712,0.0002566142,0.0002129065,0.00011543775,0.00031251082,0.00017993685,0.001369642],"category_scores_gemma":[0.0014162841,0.00023173084,0.000091279704,0.00062663684,0.00009478239,0.0008686189,0.00006137231,0.00020451559,0.000009775086],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012357136,0.0036144962,0.36284626,0.00010830528,0.00013744397,1.037603e-7,0.0062421435,0.00023991746,0.004234096,0.5362723,0.0008422037,0.08533919],"study_design_scores_gemma":[0.0014260306,0.00046646994,0.95400995,0.00009197699,0.000159235,1.2004703e-7,0.007570669,0.0012131751,0.00030002726,0.032026973,0.00241343,0.00032195263],"about_ca_topic_score_codex":0.017671937,"about_ca_topic_score_gemma":0.0022299218,"teacher_disagreement_score":0.5911637,"about_ca_system_score_codex":0.0010407818,"about_ca_system_score_gemma":0.0028589894,"threshold_uncertainty_score":0.99954325},"labels":[],"label_agreement":null},{"id":"W2087036401","doi":"10.1080/02602930902862842","title":"Bases of competence: an instrument for self and institutional assessment","year":2009,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Higher Education and Employability","field":"Social Sciences","cited_by":48,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"","keywords":"Competence (human resources); Psychology; Self-assessment; Medical education; Pedagogy; Mathematics education; Social psychology; Medicine","score_opus":0.1171535369150104,"score_gpt":0.4856852979474642,"score_spread":0.3685317610324538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2087036401","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97047913,0.00007742354,0.00044618413,0.012848999,0.001449156,0.0015617117,0.000008003119,0.000047126432,0.013082273],"genre_scores_gemma":[0.96059954,0.00003683104,0.037795316,0.0004895092,0.00024580312,0.00043070805,0.00015534551,0.000005190679,0.00024176248],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9977342,0.00054188573,0.0004447951,0.0003290341,0.0007527337,0.00019735387],"domain_scores_gemma":[0.99868053,0.00016173873,0.0001997496,0.00021330325,0.0006203313,0.00012434462],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028652928,0.00011766693,0.00016747223,0.00018436163,0.00031796953,0.00007814064,0.00014346806,0.000086164735,0.00080865505],"category_scores_gemma":[0.000068664245,0.00012575072,0.000036160636,0.00035408806,0.00012437101,0.0006072007,0.00000983201,0.00009020011,0.0000016317399],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000014258559,0.0028849638,0.3789487,0.000036703575,0.000013470499,3.3076407e-8,0.0052523287,0.00019836702,0.0003797681,0.5584859,0.00051594886,0.05326954],"study_design_scores_gemma":[0.00054517016,0.00016631233,0.9545297,0.000034732035,0.000038570106,2.007583e-7,0.0013725013,0.000694976,0.000022303604,0.028153451,0.014308362,0.00013373935],"about_ca_topic_score_codex":0.0005250796,"about_ca_topic_score_gemma":0.0001995103,"teacher_disagreement_score":0.575581,"about_ca_system_score_codex":0.0009732373,"about_ca_system_score_gemma":0.004667391,"threshold_uncertainty_score":0.88542056},"labels":[],"label_agreement":null},{"id":"W2119158672","doi":"10.1080/02602938.2012.693906","title":"Interpreting differences between the United States and New Zealand university students’ engagement scores as measured by the NSSE and AUSSE","year":2012,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Higher Education Research Studies","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"University of Auckland; National Center For Environmental Assessment; University of Victoria","keywords":"Student engagement; Constructive; Psychology; Scale (ratio); Medical education; Mathematics education; Medicine; Geography; Computer science; Process (computing)","score_opus":0.13220037945104948,"score_gpt":0.4711410347303959,"score_spread":0.33894065527934647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119158672","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96033156,0.0006710184,0.000018692454,0.036512528,0.00042315252,0.00088971027,0.0000031537375,0.000019370329,0.0011308122],"genre_scores_gemma":[0.9931257,0.00074135425,0.00007162103,0.00022668445,0.00022642102,0.000079727826,0.00005209562,0.000006359132,0.0054700472],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99564815,0.002597928,0.00019556521,0.00020705473,0.0010734813,0.00027782298],"domain_scores_gemma":[0.9979967,0.001296379,0.00014647903,0.00016112732,0.00025450985,0.00014478895],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054706084,0.000118464064,0.00011529799,0.00013906574,0.00089313486,0.00027129555,0.00028641545,0.000046287285,0.00036460147],"category_scores_gemma":[0.00022396634,0.00008125641,0.000015482752,0.0005310907,0.00026727514,0.0003843048,0.000116847965,0.00021228407,0.0000055835367],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000052289442,0.00010523672,0.9329675,0.0000045883835,0.000040124174,1.778766e-8,0.049828492,0.0000032458381,0.000009990704,0.002246334,0.011731066,0.0030581697],"study_design_scores_gemma":[0.0002435337,0.000026078456,0.89323145,0.00003279482,0.000069522466,6.8117025e-8,0.04662518,0.00001837009,0.0000042826173,0.0011214964,0.058541,0.000086230015],"about_ca_topic_score_codex":0.016748216,"about_ca_topic_score_gemma":0.0007446633,"teacher_disagreement_score":0.046809934,"about_ca_system_score_codex":0.00043643496,"about_ca_system_score_gemma":0.00062722084,"threshold_uncertainty_score":0.9897993},"labels":[],"label_agreement":null},{"id":"W2401839864","doi":"10.1080/02602938.2016.1188057","title":"Impact assessment of a department-wide science education initiative using students’ perceptions of teaching and learning experiences","year":2016,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Helpfulness; Context (archaeology); Psychology; Perception; Medical education; Preference; Science education; Class (philosophy); Teaching method; Mathematics education; Medicine; Computer science","score_opus":0.1817612835993253,"score_gpt":0.5919669342211382,"score_spread":0.4102056506218129,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2401839864","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98691094,0.00008372218,0.00072596193,0.0016772209,0.00078093697,0.00088313426,0.0000017693358,0.000021800328,0.00891454],"genre_scores_gemma":[0.9858087,0.000050268158,0.013368783,0.000055156448,0.000107690445,0.00029887466,0.000013572482,0.000011347721,0.0002856431],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9925905,0.0037746811,0.00064980664,0.00041989595,0.0023064783,0.00025867394],"domain_scores_gemma":[0.99628115,0.0012866752,0.0010080552,0.00022734705,0.0010860285,0.00011071734],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.015622433,0.00015256768,0.00022182395,0.0006681596,0.0008688761,0.00014707421,0.00033081198,0.00007069694,0.0010643402],"category_scores_gemma":[0.002515879,0.00012961654,0.000050288978,0.00064389664,0.0005395298,0.0025855089,0.00008129002,0.00026529565,0.0000016383573],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000004590525,0.0007967736,0.9327905,0.000011863477,0.000019480749,1.9151637e-8,0.035787303,0.00023142091,0.0040104277,0.009863156,0.00002986282,0.0164546],"study_design_scores_gemma":[0.00034783047,0.00012481656,0.91684645,0.00022978894,0.00007711832,3.2387175e-7,0.08032297,0.00058331393,0.000047611495,0.0011031916,0.00018010064,0.00013646137],"about_ca_topic_score_codex":0.00205402,"about_ca_topic_score_gemma":0.00010146872,"teacher_disagreement_score":0.044535667,"about_ca_system_score_codex":0.0021746375,"about_ca_system_score_gemma":0.011878883,"threshold_uncertainty_score":0.99984884},"labels":[],"label_agreement":null},{"id":"W2705397363","doi":"10.1080/02602938.2017.1343799","title":"Comparing student, instructor, classroom and institutional data to evaluate a seven-year department-wide science education initiative","year":2017,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Helpfulness; Enthusiasm; Graduation (instrument); Psychology; Medical education; Perception; Class (philosophy); Class size; Higher education; Mathematics education; Medicine; Social psychology; Computer science; Political science","score_opus":0.41775580264030054,"score_gpt":0.5790085889658491,"score_spread":0.16125278632554851,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2705397363","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9052697,0.000101661884,0.00009036717,0.020066893,0.0041998276,0.0017479121,0.000009687499,0.000044821758,0.06846917],"genre_scores_gemma":[0.98653555,0.00005248761,0.010890562,0.0007261343,0.00048424056,0.00044810647,0.00022443305,0.000014790102,0.000623686],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99420106,0.0012250938,0.00056767516,0.0008724444,0.0027648527,0.0003688574],"domain_scores_gemma":[0.99621725,0.00035950541,0.00077281665,0.001142102,0.0012600102,0.0002483256],"candidate_categories":["sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.013621912,0.00020035353,0.00021629133,0.00052509183,0.0029764418,0.0014435092,0.0014461968,0.00008692776,0.00049782585],"category_scores_gemma":[0.004108587,0.00022678032,0.000023483812,0.0004938287,0.0005588583,0.006167211,0.0005564046,0.000290283,0.00006660615],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013372956,0.00058828446,0.8821089,0.000012992394,0.000024204606,1.4163305e-7,0.0065638227,0.0002291555,0.000100934165,0.08731863,0.0012396134,0.021799933],"study_design_scores_gemma":[0.00067590986,0.00003888743,0.9670017,0.00013368703,0.00009881621,5.6130756e-7,0.006686409,0.0014796519,0.000011065286,0.004942663,0.018687267,0.000243383],"about_ca_topic_score_codex":0.0021716973,"about_ca_topic_score_gemma":0.002786822,"teacher_disagreement_score":0.08489279,"about_ca_system_score_codex":0.002460761,"about_ca_system_score_gemma":0.019638821,"threshold_uncertainty_score":0.9995931},"labels":[],"label_agreement":null},{"id":"W2762181356","doi":"10.1080/02602938.2017.1380161","title":"Team dynamics feedback for post-secondary student learning teams","year":2017,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Team Dynamics and Performance","field":"Psychology","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa; University of Calgary","funders":"","keywords":"Psychology; Team composition; Team effectiveness; Context (archaeology); Medical education; Health care; Teamwork; Suite; Perception; Applied psychology; Knowledge management; Computer science; Medicine; Social psychology","score_opus":0.04679159502534595,"score_gpt":0.4609525818626039,"score_spread":0.41416098683725794,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2762181356","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8732196,0.00016056439,0.0002958827,0.0035984472,0.0075513935,0.0013869867,0.000018664896,0.000039085826,0.11372939],"genre_scores_gemma":[0.97410023,0.00001662282,0.0010797732,0.00026216696,0.00045818058,0.00152206,0.0007290677,0.000034862253,0.021797061],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99824345,0.0001896096,0.00042634737,0.00043264925,0.00041448057,0.00029347182],"domain_scores_gemma":[0.9982351,0.000110498106,0.00049445557,0.0006641327,0.0004340958,0.00006168448],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0016916706,0.00018058208,0.00018865196,0.0001731845,0.00047657892,0.00024655953,0.0003973843,0.00014514699,0.0030772933],"category_scores_gemma":[0.000048310434,0.0001923392,0.00006834862,0.00007995929,0.000048540594,0.00042893,0.00005707988,0.0003364417,0.000097628414],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047071313,0.0008967894,0.734304,0.000040286242,0.000059908038,1.7517996e-7,0.0017778642,0.00043930204,0.00008813894,0.049952436,0.0024907074,0.20990333],"study_design_scores_gemma":[0.0013778774,0.0001991643,0.96425754,0.00003893341,0.000056594086,0.0000014321806,0.0023720239,0.015154027,0.0000018694086,0.0019903698,0.0143304905,0.00021968287],"about_ca_topic_score_codex":0.00042191125,"about_ca_topic_score_gemma":0.0002820446,"teacher_disagreement_score":0.22995354,"about_ca_system_score_codex":0.00089317234,"about_ca_system_score_gemma":0.000801875,"threshold_uncertainty_score":0.997834},"labels":[],"label_agreement":null},{"id":"W2771690381","doi":"10.1080/02602938.2017.1412397","title":"Taking stock and effecting change: curriculum evaluation through a review of course syllabi","year":2017,"lang":"en","type":"review","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Windsor","funders":"University of Windsor","keywords":"Syllabus; Curriculum; Experiential learning; Unit (ring theory); Higher education; Medical education; Discipline; Psychology; Course evaluation; Reading (process); Pedagogy; Mathematics education; Political science; Medicine","score_opus":0.6459460384304073,"score_gpt":0.6783012568282019,"score_spread":0.03235521839779454,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2771690381","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00011403568,0.96083236,0.000009290876,0.003946115,0.0037118779,0.009648295,0.00000647175,0.000044197048,0.021687355],"genre_scores_gemma":[0.004900793,0.9852898,0.0013021283,0.00019975164,0.0011003241,0.0063785836,0.00041974263,0.000050859122,0.00035799656],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98196447,0.01182092,0.0015156133,0.00082793424,0.0034960406,0.0003750483],"domain_scores_gemma":[0.9891223,0.0013381267,0.0065827062,0.0008290108,0.0020322173,0.00009566507],"candidate_categories":["metaresearch","metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.04082786,0.00044265075,0.0013560094,0.00034738542,0.0007205781,0.00027261948,0.0005725865,0.00038989802,0.0024790035],"category_scores_gemma":[0.006590876,0.0004387896,0.0002275614,0.00069471233,0.00018022311,0.002100857,0.0001098374,0.00065235054,0.000024560846],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[8.1495097e-7,0.0003187975,0.0016380161,0.021818882,0.0000751132,1.3088881e-7,0.0015232051,0.0000017305669,9.7698205e-8,0.0048941155,0.0007753576,0.9689537],"study_design_scores_gemma":[0.00037064322,0.000057144385,0.013038277,0.14815895,0.0049006436,0.0000026011055,0.00054891553,0.00020550613,6.263303e-8,0.00094717595,0.8312931,0.0004769491],"about_ca_topic_score_codex":0.0040743034,"about_ca_topic_score_gemma":0.00036054067,"teacher_disagreement_score":0.9684768,"about_ca_system_score_codex":0.0021742003,"about_ca_system_score_gemma":0.008808434,"threshold_uncertainty_score":0.9998064},"labels":[],"label_agreement":null},{"id":"W2962316688","doi":"10.1080/02602938.2019.1637819","title":"Voices at the gate: Faculty members’ and students’ differing perspectives on the purposes of the PhD comprehensive examination","year":2019,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Doctoral Education Challenges and Solutions","field":"Health Professions","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Gatekeeping; Psychology; Medical education; Final examination; Empirical examination; Identity (music); Function (biology); Higher education; Affect (linguistics); Pedagogy; Perspective (graphical); Mathematics education; Medicine; Political science","score_opus":0.3458290584160393,"score_gpt":0.5752187854825841,"score_spread":0.2293897270665448,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2962316688","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9632757,0.0005185173,0.0000017889032,0.02380107,0.0021914165,0.0026057707,0.000010999633,0.000013531888,0.007581252],"genre_scores_gemma":[0.98897064,0.00015065417,0.000025316313,0.0005268243,0.00015755423,0.0014285374,0.00004868345,0.000013308244,0.008678489],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9965775,0.0016966602,0.00040651814,0.0002903567,0.00083890464,0.00019004532],"domain_scores_gemma":[0.9973765,0.0010809196,0.00044329782,0.0004509573,0.00061485823,0.00003344724],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0015329315,0.00014522791,0.00015328688,0.00008020776,0.00073524733,0.00003683168,0.0002762503,0.00008423984,0.0033750022],"category_scores_gemma":[0.00010533317,0.000077842014,0.000047117523,0.0003060587,0.000106056235,0.00018106062,0.0001445167,0.00034562737,0.00004268277],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029918516,0.001018162,0.79963684,0.00017583877,0.000104966166,2.2406988e-8,0.10059138,0.00089251244,0.000990821,0.08744289,0.006549369,0.0025672764],"study_design_scores_gemma":[0.00039071924,0.000036985923,0.88915837,0.00015313992,0.000043414202,1.594708e-7,0.10205243,0.00041467932,0.000016698936,0.0005716335,0.0070845895,0.00007717483],"about_ca_topic_score_codex":0.0003261983,"about_ca_topic_score_gemma":0.00023069783,"teacher_disagreement_score":0.08952153,"about_ca_system_score_codex":0.00086488115,"about_ca_system_score_gemma":0.00045764164,"threshold_uncertainty_score":0.99753606},"labels":[],"label_agreement":null},{"id":"W2983876144","doi":"10.1080/02602938.2019.1689381","title":"The effects of perceived professor competence, warmth and gender on students’ likelihood to register for a course","year":2019,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Carleton University","funders":"","keywords":"Vignette; Psychology; Gender bias; Competence (human resources); Social psychology; Developmental psychology","score_opus":0.10712911187854181,"score_gpt":0.527350665480816,"score_spread":0.4202215536022742,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2983876144","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96417016,0.0005051488,0.000022524559,0.016945725,0.0051452965,0.005177502,0.0000033130898,0.00001725547,0.008013102],"genre_scores_gemma":[0.9829445,0.000048091373,0.00072100223,0.0016406517,0.000115176204,0.004749417,0.000059639235,0.000019843686,0.009701698],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99781394,0.0006967881,0.0004464026,0.00032984762,0.00049853115,0.0002145193],"domain_scores_gemma":[0.9971941,0.0011103577,0.00027707074,0.0007649112,0.0005735753,0.00007999627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015346789,0.00013188503,0.00017413404,0.00014999026,0.00014782201,0.000051554794,0.00031390233,0.00008329336,0.00055446854],"category_scores_gemma":[0.00007047956,0.00011026619,0.00003829256,0.00023960082,0.00003728657,0.000074579155,0.000042074767,0.00016415454,0.00005734999],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021548934,0.0032441749,0.7723192,0.0004238043,0.00011462989,4.5670493e-8,0.027597168,0.000013434943,0.00078842015,0.13742276,0.020213915,0.037646975],"study_design_scores_gemma":[0.0012605501,0.00031433319,0.97672963,0.00012630694,0.000051964405,3.7319367e-7,0.005187521,0.00005473108,0.0000075459643,0.0034119787,0.012741805,0.00011325695],"about_ca_topic_score_codex":0.00004789287,"about_ca_topic_score_gemma":0.00003900224,"teacher_disagreement_score":0.20441045,"about_ca_system_score_codex":0.00028015513,"about_ca_system_score_gemma":0.0006382452,"threshold_uncertainty_score":0.6071042},"labels":[],"label_agreement":null},{"id":"W3008488197","doi":"10.1080/02602938.2020.1727412","title":"Team dynamics feedback for post-secondary student learning teams: introducing the “Bare CARE” assessment and report","year":2020,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Team Dynamics and Performance","field":"Psychology","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Formative assessment; Teamwork; Psychology; Experiential learning; Medical education; Health care; Sample (material); Applied psychology; Pedagogy; Medicine","score_opus":0.03439414168239335,"score_gpt":0.42985498815841316,"score_spread":0.3954608464760198,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3008488197","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9560162,0.0007550064,0.000979,0.018823566,0.0033964724,0.0022184053,0.000019743631,0.000057290807,0.017734345],"genre_scores_gemma":[0.9925658,0.000028181958,0.0018942974,0.0010407898,0.00068455056,0.0014383144,0.0012605214,0.000039150418,0.0010483869],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99757475,0.0003778272,0.0006153611,0.00063572504,0.0005215376,0.00027482267],"domain_scores_gemma":[0.99844825,0.00020414966,0.00030738956,0.00040163795,0.0005548076,0.0000837628],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016983248,0.00021569962,0.00022753351,0.00011349504,0.00028939618,0.0001648668,0.00021558262,0.00011418648,0.00073422975],"category_scores_gemma":[0.000056658828,0.00019002803,0.00005821407,0.0002710219,0.000045348577,0.0002477685,0.00007651246,0.0005185651,0.0000113774395],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000052304913,0.00039436112,0.8335823,0.00015280879,0.00011680196,0.0000013437875,0.019541658,0.0030253832,0.00019836854,0.015452867,0.003224574,0.124257274],"study_design_scores_gemma":[0.0009500376,0.00030302067,0.9268766,0.00003501498,0.000110997775,0.000009239382,0.03468575,0.02393527,0.000001990701,0.00024445774,0.012627572,0.00022006483],"about_ca_topic_score_codex":0.00021157444,"about_ca_topic_score_gemma":0.00010328393,"teacher_disagreement_score":0.124037206,"about_ca_system_score_codex":0.0010839687,"about_ca_system_score_gemma":0.001269985,"threshold_uncertainty_score":0.80393004},"labels":[],"label_agreement":null},{"id":"W3049736335","doi":"10.1080/02602938.2020.1805410","title":"Are students gender-neutral in their assessment of online teaching staff?","year":2020,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Workload; Psychology; Construct (python library); Quality (philosophy); Promotion (chess); Medical education; Course evaluation; Higher education; Computer science; Medicine","score_opus":0.37231232978525863,"score_gpt":0.5823678399594441,"score_spread":0.21005551017418544,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3049736335","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94946116,0.00011081501,0.00012450702,0.032742515,0.0010452751,0.0012118806,0.000009066344,0.000049668062,0.01524511],"genre_scores_gemma":[0.9918222,0.000041303887,0.0063196076,0.0008142819,0.0003940183,0.00021236442,0.00011987425,0.000020900103,0.00025546996],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99207073,0.0042936145,0.0008296034,0.00047773978,0.0020358548,0.0002924342],"domain_scores_gemma":[0.997614,0.00052642176,0.0010733004,0.00026283468,0.00039905508,0.00012437187],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.008926777,0.00018686663,0.00029992647,0.00030747344,0.00021007836,0.00014424372,0.0005204664,0.00012457781,0.0010208341],"category_scores_gemma":[0.00079368544,0.00019928292,0.000065174994,0.0005867462,0.0000666512,0.0010717008,0.000078379766,0.00060004275,0.0000075326207],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000013020799,0.0028297594,0.93659884,0.000050675535,0.00003566547,4.4819578e-7,0.018324431,0.0047358517,0.00028374142,0.027293218,0.0007008898,0.009133476],"study_design_scores_gemma":[0.00080666353,0.00006849294,0.96281064,0.000096419426,0.000053220927,9.463963e-8,0.024182338,0.003776573,0.00001087942,0.002465267,0.0055463337,0.00018308326],"about_ca_topic_score_codex":0.0014150862,"about_ca_topic_score_gemma":0.00151661,"teacher_disagreement_score":0.042361017,"about_ca_system_score_codex":0.0014035945,"about_ca_system_score_gemma":0.0025903787,"threshold_uncertainty_score":0.99989235},"labels":[],"label_agreement":null},{"id":"W3092685673","doi":"10.1080/02602938.2020.1826900","title":"When academic integrity rules should not apply: a survey of academic staff","year":2020,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Academic integrity and plagiarism","field":"Social Sciences","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Social Innovation","funders":"","keywords":"Academic integrity; Ambiguity; Public relations; Compliance (psychology); Ideology; Psychology; Multinational corporation; Welfare; Social psychology; Action (physics); Political science; Law","score_opus":0.29559258214728357,"score_gpt":0.4864898433562915,"score_spread":0.1908972612090079,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3092685673","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8558592,0.00076631503,0.0010592408,0.110519946,0.002420065,0.0019730981,0.000090160014,0.00010802229,0.027203944],"genre_scores_gemma":[0.994217,0.00040411108,0.0012498852,0.0021973944,0.00060975563,0.00022997597,0.00036147606,0.000017172197,0.0007132797],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9934456,0.0029974144,0.000864685,0.00046849393,0.001873041,0.0003507667],"domain_scores_gemma":[0.9973081,0.00095634337,0.0005130613,0.00018970451,0.00084695726,0.00018585086],"candidate_categories":["research_integrity","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.012161912,0.00019385279,0.00032366873,0.00018644353,0.00018135204,0.0000408545,0.0006084289,0.0012234581,0.0032425155],"category_scores_gemma":[0.002041222,0.00020403169,0.000060680246,0.00071087695,0.00019875348,0.00070930173,0.00006603642,0.0038560666,0.00007366654],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014328209,0.0004138091,0.5919715,0.00011577313,0.00006234446,2.500081e-7,0.058522023,0.00019405945,0.001817366,0.22943984,0.049855955,0.067463785],"study_design_scores_gemma":[0.00056862837,0.00007697776,0.92059153,0.00015441146,0.000094818475,2.1478809e-7,0.007914933,0.0029621164,0.00026702153,0.034339346,0.032661725,0.00036825446],"about_ca_topic_score_codex":0.013222737,"about_ca_topic_score_gemma":0.0005624033,"teacher_disagreement_score":0.32862005,"about_ca_system_score_codex":0.0007525152,"about_ca_system_score_gemma":0.0047849324,"threshold_uncertainty_score":0.99844205},"labels":[],"label_agreement":null},{"id":"W3095117930","doi":"10.1080/02602938.2020.1836123","title":"University students’ negative emotions in a computer-based examination: the roles of trait test-emotion, prior test-taking methods and gender","year":2020,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Education, Achievement, and Giftedness","field":"Psychology","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; Concordia University; McGill University Health Centre; University of Alberta","funders":"","keywords":"Test anxiety; Psychology; Test (biology); Trait; Anxiety; Social psychology; Computer science","score_opus":0.15587259605893805,"score_gpt":0.4787915717209989,"score_spread":0.32291897566206085,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3095117930","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9731333,0.000089223824,0.013915724,0.00865351,0.00078304694,0.0014178989,0.00001350968,0.000028900009,0.0019648657],"genre_scores_gemma":[0.98470986,0.000008916511,0.014186755,0.00044485994,0.00017062889,0.00013617247,0.00009614452,0.000014602512,0.00023203435],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9964742,0.0020086512,0.000439491,0.00044327823,0.00046617395,0.00016821409],"domain_scores_gemma":[0.99714094,0.001578279,0.0005035223,0.00026629877,0.00044727864,0.000063648244],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002578935,0.00016596953,0.00019817047,0.00029297138,0.00014882034,0.00005132639,0.00025904804,0.00009309115,0.0010734783],"category_scores_gemma":[0.00025247099,0.00016341978,0.000038799833,0.0010144291,0.000100357756,0.00025146655,0.000037668357,0.00020442088,0.000003671428],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012245779,0.0022684708,0.9149113,0.000057435635,0.00004016311,2.7332297e-7,0.031694517,0.00069161825,0.0002597578,0.005762797,0.00030033002,0.044001088],"study_design_scores_gemma":[0.0014879745,0.0001505436,0.97825223,0.000050393508,0.000109716035,4.5931745e-7,0.013363397,0.005070062,0.000026381278,0.0011127369,0.00023330671,0.00014280515],"about_ca_topic_score_codex":0.000121294725,"about_ca_topic_score_gemma":0.000111178924,"teacher_disagreement_score":0.063340925,"about_ca_system_score_codex":0.0003540909,"about_ca_system_score_gemma":0.0006176085,"threshold_uncertainty_score":0.99983966},"labels":[],"label_agreement":null},{"id":"W3162164515","doi":"10.1080/02602938.2021.1921105","title":"Fitted: the impact of academics’ attire on students’ evaluations and intentions","year":2021,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Psychology; Higher education; Pedagogy; Mathematics education; Medical education; Medicine; Political science","score_opus":0.2935359884430912,"score_gpt":0.625609991914194,"score_spread":0.3320740034711028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3162164515","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9343823,0.0016722073,0.000035808862,0.026445234,0.002001499,0.0010975202,0.000016127866,0.000025819225,0.03432345],"genre_scores_gemma":[0.9935219,0.00018327228,0.0003160356,0.00067401695,0.00014952497,0.0012703885,0.00028460406,0.000017615574,0.0035825993],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99645925,0.0015254744,0.0006546358,0.00034852844,0.0008249284,0.00018716328],"domain_scores_gemma":[0.99691874,0.0005905492,0.00035759152,0.00088087964,0.0011743181,0.00007793168],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0020656846,0.00015269885,0.00018183517,0.00026764613,0.00021805585,0.00006446451,0.00031223986,0.00013601505,0.010253518],"category_scores_gemma":[0.00018373491,0.00012918007,0.000091457456,0.0008061858,0.000088446504,0.00013847466,0.000073052695,0.0004475932,0.000045323402],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000214928,0.0025811968,0.82653177,0.00002202207,0.00013712171,1.2991552e-7,0.007087109,0.00035288805,0.00038731043,0.10767077,0.015961848,0.039246336],"study_design_scores_gemma":[0.00062493124,0.00008143154,0.9842854,0.00008976972,0.00007427849,0.0000040375694,0.0066670636,0.00032218348,0.000011369237,0.0058562877,0.0018726873,0.0001105404],"about_ca_topic_score_codex":0.0003188931,"about_ca_topic_score_gemma":0.00007032758,"teacher_disagreement_score":0.15775365,"about_ca_system_score_codex":0.00068351947,"about_ca_system_score_gemma":0.0021179207,"threshold_uncertainty_score":0.99065125},"labels":[],"label_agreement":null},{"id":"W3172383618","doi":"10.1080/02602938.2021.1912286","title":"Student satisfaction with use of an online peer feedback system","year":2021,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Team Dynamics and Performance","field":"Psychology","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Teamwork; Peer feedback; Psychology; Team composition; Higher education; Social loafing; Curriculum; Knowledge management; Medical education; Computer science; Mathematics education; Pedagogy; Social psychology; Political science","score_opus":0.12416158612870165,"score_gpt":0.45557304381090247,"score_spread":0.3314114576822008,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3172383618","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9932513,0.00006786111,0.00013716512,0.00075951737,0.0023897104,0.00045563348,0.000016782333,0.00002525601,0.0028967606],"genre_scores_gemma":[0.99165016,0.000005895205,0.0038483145,0.000094047835,0.00016895388,0.00022954232,0.0007868996,0.000018661818,0.0031974993],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9980484,0.00039828496,0.0003886875,0.00031057285,0.00071859313,0.00013545562],"domain_scores_gemma":[0.99829006,0.000058293772,0.00025403575,0.00043341174,0.00091923977,0.000044956512],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00063034217,0.00011813818,0.00015776635,0.00013765837,0.00004321983,0.000056013763,0.00006361989,0.000080769285,0.0017371938],"category_scores_gemma":[0.0000075670728,0.00011272808,0.000023690749,0.00036458197,0.000018449755,0.00039247843,0.000014050452,0.00013719115,0.000015630323],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000038084443,0.0014719927,0.9449684,0.000047741003,0.00005627951,0.0000011794058,0.0013793107,0.002788877,0.00033541233,0.021853883,0.00036348045,0.02669537],"study_design_scores_gemma":[0.00075886265,0.00013503926,0.9897222,0.000084941974,0.00007715378,0.000007929747,0.003780826,0.004292025,0.00001401403,0.000073900665,0.000930791,0.0001223243],"about_ca_topic_score_codex":0.00093185017,"about_ca_topic_score_gemma":0.001114163,"teacher_disagreement_score":0.044753805,"about_ca_system_score_codex":0.0005097797,"about_ca_system_score_gemma":0.0006110391,"threshold_uncertainty_score":0.99917537},"labels":[],"label_agreement":null},{"id":"W3184059431","doi":"10.1080/02602938.2021.1956428","title":"Patterns of special consideration requests at a UK university: reasons given and associations with demographic factors","year":2021,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Medical Education and Admissions","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Psychology; Quarter (Canadian coin); Mental health; Control (management); Medical education; Medicine; Psychiatry","score_opus":0.06568450014595346,"score_gpt":0.3824971704876142,"score_spread":0.3168126703416608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3184059431","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9767166,0.00008329869,0.00006615331,0.016058777,0.00048227466,0.00048527544,0.000018374707,0.0000151425465,0.006074107],"genre_scores_gemma":[0.99281627,0.0000864973,0.0012101814,0.0003962788,0.00019910495,0.000022349739,0.0010165737,0.000009773934,0.004242952],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99847674,0.00029724062,0.00027448047,0.00026836115,0.00056250504,0.00012068284],"domain_scores_gemma":[0.9984622,0.00020846457,0.00021850801,0.00019903711,0.0006572014,0.00025461832],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00037233304,0.00010938506,0.00020499101,0.00023699018,0.00012111967,0.000019962381,0.00003113123,0.00009578741,0.00838862],"category_scores_gemma":[0.00043454158,0.00010315406,0.000037917245,0.00046301825,0.00004787459,0.00013049788,0.000020485924,0.00014742759,0.0000023022913],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000015307285,0.00050430617,0.98718977,0.000034298584,0.000046314395,0.0000019515728,0.00077759917,0.000007659979,0.0006823967,0.007157896,0.0026104036,0.00097209145],"study_design_scores_gemma":[0.0011246754,0.00007593591,0.99056286,0.00022382823,0.00023908449,0.00000784725,0.0021780299,0.000079350895,0.00020674264,0.00044792064,0.0047552446,0.000098454424],"about_ca_topic_score_codex":0.00009186821,"about_ca_topic_score_gemma":0.00032129473,"teacher_disagreement_score":0.016099691,"about_ca_system_score_codex":0.0005586074,"about_ca_system_score_gemma":0.0043773903,"threshold_uncertainty_score":0.9925178},"labels":[],"label_agreement":null},{"id":"W4200173019","doi":"10.1080/02602938.2021.2009439","title":"Does a classroom-based curriculum offer authentic assessments? A strategy to uncover their prevalence and incorporate opportunities for authenticity","year":2021,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Authentic assessment; Syllabus; Curriculum; Capstone; Judgement; Psychology; Context (archaeology); Medical education; Summative assessment; Authentic learning; Mathematics education; Curriculum development; Pedagogy; Formative assessment; Medicine; Computer science","score_opus":0.10888072822505045,"score_gpt":0.42622515487460516,"score_spread":0.3173444266495547,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4200173019","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96925163,0.000118157106,0.004160129,0.019915191,0.0026659467,0.0026108979,0.00002419286,0.000051179213,0.0012027017],"genre_scores_gemma":[0.9724663,0.000027231015,0.01405176,0.0033071225,0.00030880774,0.0029529498,0.0008814207,0.000035167537,0.005969252],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9970869,0.0003806864,0.0007297847,0.0006234718,0.00086783315,0.00031136593],"domain_scores_gemma":[0.99686754,0.00016546548,0.00031420766,0.00050774345,0.0020219132,0.00012312607],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0020347715,0.00026360058,0.00032556805,0.00042655252,0.00014861634,0.00014911187,0.00011448545,0.00015635835,0.0014422666],"category_scores_gemma":[0.00038092947,0.00020261586,0.00006521535,0.000830799,0.00007974487,0.000371067,0.000044308457,0.00025195134,0.0000064301553],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00017291897,0.008273856,0.72490215,0.0031200433,0.000240628,0.0000059705917,0.0014099573,0.0005223424,0.004180801,0.03368194,0.018084163,0.20540522],"study_design_scores_gemma":[0.0023808551,0.0004965324,0.93417996,0.0009718157,0.0004982899,0.000006074164,0.0045253243,0.03330765,0.00068359834,0.014439263,0.008097338,0.00041329072],"about_ca_topic_score_codex":0.000035561745,"about_ca_topic_score_gemma":0.000025453788,"teacher_disagreement_score":0.20927781,"about_ca_system_score_codex":0.0010128473,"about_ca_system_score_gemma":0.005496587,"threshold_uncertainty_score":0.99947053},"labels":[],"label_agreement":null},{"id":"W4366159769","doi":"10.1080/02602938.2023.2199181","title":"Students’ online evaluation of teaching and system continuance usage intention: new directions from a multidisciplinary perspective","year":2023,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Online and Blended Learning","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"National Science and Technology Council","keywords":"Helpfulness; CLARITY; Set (abstract data type); Continuance; Psychology; Perspective (graphical); Higher education; Medical education; Knowledge management; Computer science; Social psychology; Political science","score_opus":0.10627568801785903,"score_gpt":0.5058444972128289,"score_spread":0.39956880919496984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4366159769","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98178643,0.0004915098,0.000084656735,0.0051743765,0.0016713144,0.0010576624,0.000009977528,0.00009812213,0.009625957],"genre_scores_gemma":[0.9929323,0.00007008986,0.0031697832,0.000014041921,0.0005677943,0.00021186678,0.00024475425,0.000012409943,0.002776951],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99511296,0.0023237644,0.00041464242,0.0003735618,0.0016007534,0.00017434168],"domain_scores_gemma":[0.9981785,0.00035111725,0.00032053556,0.00017424587,0.0009040913,0.00007151281],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007327515,0.00011517716,0.00018173728,0.00033535063,0.00040412752,0.00009344107,0.00014601304,0.00008845746,0.00036206562],"category_scores_gemma":[0.00049659493,0.00012512674,0.00004428991,0.0007105838,0.000057534467,0.0005252253,0.00005711201,0.00026638588,0.000011253403],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028234186,0.0012825662,0.6579407,0.000042044558,0.00012211376,4.6814029e-7,0.041627858,0.0014459783,0.0009930476,0.042757954,0.0010616274,0.25269744],"study_design_scores_gemma":[0.0008007569,0.00003320967,0.8515886,0.00026465353,0.00019173228,1.652772e-7,0.12944466,0.0112797,0.0000054986967,0.004470675,0.0018057048,0.000114656534],"about_ca_topic_score_codex":0.009359992,"about_ca_topic_score_gemma":0.0030558656,"teacher_disagreement_score":0.2525828,"about_ca_system_score_codex":0.0014818289,"about_ca_system_score_gemma":0.002272896,"threshold_uncertainty_score":0.9972368},"labels":[],"label_agreement":null},{"id":"W4386923563","doi":"10.1080/02602938.2023.2259632","title":"Are assessment accommodations cheating? A critical policy analysis","year":2023,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Disability Education and Employment","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Ontario Tech University","funders":"","keywords":"Cheating; Accommodation; Psychology; Context (archaeology); Realm; Inclusion (mineral); Public relations; Reasonable accommodation; Social psychology; Political science; Law","score_opus":0.21021702937233872,"score_gpt":0.577718254142636,"score_spread":0.3675012247702973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386923563","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53765833,0.00003603098,0.00024246424,0.30472705,0.0030472982,0.0014673773,0.000017371962,0.00031320713,0.15249084],"genre_scores_gemma":[0.9891775,0.000024532495,0.0011878965,0.0011902758,0.0007082488,0.0018438253,0.00044314004,0.0000159529,0.0054085758],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9957323,0.0012921421,0.0005742974,0.00048255824,0.00148246,0.0004362217],"domain_scores_gemma":[0.997732,0.00061425066,0.0002559493,0.00046376043,0.000733443,0.00020060582],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0045062145,0.00016129552,0.0002461641,0.0012379924,0.0006042375,0.00032031286,0.00027736748,0.00012222816,0.008070788],"category_scores_gemma":[0.0010982052,0.00018430025,0.00013048261,0.0062384563,0.00016291934,0.00058207544,0.00004924185,0.00019570469,0.00018070308],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000020259968,0.0011626468,0.6979384,0.000018174513,0.000073637544,1.3908333e-7,0.0042023975,0.0012100565,0.000017693823,0.27903146,0.009830084,0.006513316],"study_design_scores_gemma":[0.0001895284,0.000011943828,0.93793017,0.000025494444,0.00022178602,5.2175505e-8,0.015040663,0.0020937126,0.0000014725346,0.021988869,0.022317905,0.00017839513],"about_ca_topic_score_codex":0.0041079787,"about_ca_topic_score_gemma":0.0030086439,"teacher_disagreement_score":0.4515192,"about_ca_system_score_codex":0.0029655679,"about_ca_system_score_gemma":0.0049214414,"threshold_uncertainty_score":0.992836},"labels":[],"label_agreement":null},{"id":"W4387267871","doi":"10.1080/02602938.2023.2263668","title":"Flexible assessment: some benefits and costs for students and instructors","year":2023,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Grading (engineering); Workload; Flexibility (engineering); Rigour; Formative assessment; Psychology; Medical education; Mathematics education; Pedagogy; Computer science; Engineering; Management; Medicine","score_opus":0.27421430645693995,"score_gpt":0.5825374862958459,"score_spread":0.30832317983890595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387267871","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9708639,0.00041780452,0.000019386182,0.019485565,0.002148075,0.0017671181,0.000006971615,0.00011726842,0.0051739467],"genre_scores_gemma":[0.9927557,0.00056108116,0.0026061411,0.0003258073,0.0003404812,0.0010254316,0.0001572424,0.000019513127,0.0022085994],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99668545,0.00093708537,0.0003586373,0.00041036677,0.001345868,0.00026261483],"domain_scores_gemma":[0.9982723,0.00081626326,0.0002815712,0.00017223692,0.00034862928,0.000109011766],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008769674,0.00013538872,0.00015518222,0.00040644797,0.00057069515,0.00038692687,0.00018411092,0.0001086423,0.0002505668],"category_scores_gemma":[0.00044289924,0.00015247075,0.000023714765,0.0005284335,0.00008570851,0.001645508,0.00006721316,0.00015019147,0.000011759354],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000008837269,0.00019812616,0.5815931,0.000026654681,0.000026326485,5.477712e-8,0.00294849,0.00015916073,0.000072555136,0.2917919,0.0015913165,0.12158349],"study_design_scores_gemma":[0.0008624863,0.00005516007,0.96697915,0.000075405165,0.000072762065,1.9569524e-7,0.0036771153,0.000927736,0.000007915244,0.015398885,0.011776651,0.0001665297],"about_ca_topic_score_codex":0.0007788041,"about_ca_topic_score_gemma":0.00025757044,"teacher_disagreement_score":0.38538605,"about_ca_system_score_codex":0.00081816263,"about_ca_system_score_gemma":0.001197883,"threshold_uncertainty_score":0.6217576},"labels":[],"label_agreement":null},{"id":"W4391679980","doi":"10.1080/02602938.2024.2312918","title":"Decolonizing academic integrity: knowledge caretaking as ethical practice","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"African cultural and philosophical studies","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"National Natural Science Foundation of China","keywords":"Academic integrity; Psychology; Pedagogy; Higher education; Engineering ethics; Medical education; Social psychology; Medicine; Political science","score_opus":0.27405782341470886,"score_gpt":0.5728610453092766,"score_spread":0.29880322189456776,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391679980","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08663566,0.008165049,0.000033230634,0.41184962,0.0056989146,0.00084481086,0.0000018170659,0.0001647475,0.48660615],"genre_scores_gemma":[0.9929847,0.00040029068,0.00047938622,0.0015300672,0.0015466104,0.00039373,0.000025234758,0.000010224337,0.0026297478],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99693954,0.0013193865,0.00032152463,0.00034339726,0.00084069686,0.00023546998],"domain_scores_gemma":[0.99793166,0.0012573013,0.0000945641,0.000093772716,0.0005365318,0.00008619758],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0044604344,0.0001206155,0.00013336276,0.00012059745,0.00042900722,0.00018766373,0.00016927443,0.00033802394,0.0014145646],"category_scores_gemma":[0.0012647226,0.00010517713,0.000049296108,0.0008066698,0.00012430076,0.0005982053,0.000047848946,0.0015515168,0.0001909817],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000005396417,0.00010828423,0.0038220002,0.000029528459,0.000029397695,8.9364477e-7,0.02624493,0.00001066469,0.000054927375,0.91228175,0.0038174763,0.053594764],"study_design_scores_gemma":[0.00012351481,0.00004325195,0.05003035,0.0002956852,0.00014508495,0.0000014646811,0.014285817,0.0004223363,0.0000056073563,0.19000933,0.74441427,0.00022330561],"about_ca_topic_score_codex":0.001272348,"about_ca_topic_score_gemma":0.00036476206,"teacher_disagreement_score":0.90634906,"about_ca_system_score_codex":0.0016733999,"about_ca_system_score_gemma":0.0028958032,"threshold_uncertainty_score":0.99949825},"labels":[],"label_agreement":null},{"id":"W4396978195","doi":"10.1080/02602938.2024.2350682","title":"Conflict management styles assessment and feedback for student self-awareness and team development","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Conflict Management and Negotiation","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Windsor; University of Ottawa; University of Winnipeg; University of Calgary","funders":"","keywords":"Psychology; Conflict management; Applied psychology; Medical education; Political science; Medicine","score_opus":0.09251670516487259,"score_gpt":0.48036438080568167,"score_spread":0.3878476756408091,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396978195","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8818933,0.002272813,0.0016900293,0.010949979,0.0035467506,0.0071322424,0.000004418446,0.00029556602,0.09221495],"genre_scores_gemma":[0.9847453,0.0008227784,0.006736162,0.00018216253,0.00025296042,0.0024187549,0.00012332255,0.000019872157,0.0046986896],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9977034,0.00027545795,0.0004100971,0.00050966576,0.0008421639,0.0002592548],"domain_scores_gemma":[0.9992368,0.00021054281,0.00011107674,0.00015304859,0.00020522793,0.000083306964],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0031169453,0.00017902078,0.00016197433,0.00037202955,0.0004317271,0.0005940514,0.00012862647,0.00007879656,0.00030434423],"category_scores_gemma":[0.000007363335,0.0001886134,0.000028531249,0.00036745393,0.000046052875,0.0005218552,0.000074318465,0.00009155897,0.000007142474],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001214944,0.00072217226,0.1902592,0.0005376711,0.0002508477,6.127558e-7,0.022695767,0.00005269271,0.000049666927,0.50795746,0.004362877,0.27309886],"study_design_scores_gemma":[0.0005150768,0.00003023323,0.71444607,0.00013078394,0.00014643981,1.3277351e-7,0.006795662,0.0018064085,0.000005442542,0.0013847252,0.2745493,0.00018969567],"about_ca_topic_score_codex":0.00012573559,"about_ca_topic_score_gemma":0.00024558432,"teacher_disagreement_score":0.5241869,"about_ca_system_score_codex":0.001234422,"about_ca_system_score_gemma":0.0011009754,"threshold_uncertainty_score":0.76914316},"labels":[],"label_agreement":null},{"id":"W4402198693","doi":"10.1080/02602938.2024.2388699","title":"‘There was very little room for me to be me’: the lived tensions between assessment standardisation and student diversity","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"Curtin University of Technology","keywords":"Diversity (politics); Pedagogy; Psychology; Higher education; Lived experience; Sociology; Mathematics education; Political science","score_opus":0.19275646327551255,"score_gpt":0.5260661712831748,"score_spread":0.33330970800766224,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402198693","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8306583,0.00052314746,0.001073691,0.15552235,0.0037889176,0.0027797709,0.000021236154,0.00010411814,0.005528467],"genre_scores_gemma":[0.99034727,0.00009046143,0.0026015996,0.0005755392,0.00083756266,0.000998892,0.00014838608,0.000022367167,0.0043779006],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9959916,0.0013876369,0.00038849752,0.0005123046,0.0014355469,0.00028440313],"domain_scores_gemma":[0.99682313,0.0018813275,0.00018900087,0.0002956666,0.00066403934,0.00014680535],"candidate_categories":["sts","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006760556,0.00017176996,0.00018878558,0.00020890753,0.0013149554,0.00064924464,0.0002803476,0.00011712248,0.0010440027],"category_scores_gemma":[0.00032058934,0.00015138541,0.00006248107,0.0006728994,0.000083154526,0.0008505575,0.00013638815,0.00027470628,0.00002498394],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018668952,0.0004992527,0.77830595,0.00007268939,0.00018401713,3.0371373e-7,0.12669231,0.00064344885,0.00015264536,0.054114304,0.020691007,0.018625397],"study_design_scores_gemma":[0.00021184854,0.0000689933,0.7762248,0.00007533116,0.00023855988,1.3156567e-7,0.026371002,0.00020025586,0.0000048604525,0.0027771862,0.19366978,0.00015727572],"about_ca_topic_score_codex":0.0011532315,"about_ca_topic_score_gemma":0.00048064478,"teacher_disagreement_score":0.17297877,"about_ca_system_score_codex":0.0019346577,"about_ca_system_score_gemma":0.0028569652,"threshold_uncertainty_score":0.9999852},"labels":[],"label_agreement":null},{"id":"W4402556220","doi":"10.1080/02602938.2024.2400349","title":"Feedback practices in clinical placement: how students come to understand how they are progressing","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Psychology; Mathematics education; Pedagogy; Higher education; Advanced Placement; Medical education; Medicine; Political science","score_opus":0.287976591638158,"score_gpt":0.57342609451059,"score_spread":0.28544950287243204,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402556220","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8896041,0.0006415185,0.000058434838,0.08524258,0.0061541144,0.0027050844,0.0000029512473,0.000093815266,0.015497406],"genre_scores_gemma":[0.9903855,0.00018152456,0.0009283224,0.0003360614,0.000891719,0.0004965505,0.00005191211,0.000026528327,0.00670192],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9942298,0.0017463308,0.0005285681,0.00068745826,0.0023764533,0.0004313903],"domain_scores_gemma":[0.9980715,0.000673508,0.0005196324,0.00027659745,0.0003086214,0.00015016938],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.009019918,0.00021948613,0.00028030275,0.00047899422,0.0002606064,0.002221458,0.00046027254,0.0001740526,0.0006111505],"category_scores_gemma":[0.00037594148,0.0002189967,0.000074570904,0.0011099642,0.000086821325,0.0015338789,0.00010074737,0.00038849848,0.000054196396],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027178205,0.0010510334,0.96253765,0.0000513131,0.000056099558,0.0000025271158,0.01423205,0.00005388881,0.00002129914,0.0075367154,0.006110632,0.0083196],"study_design_scores_gemma":[0.0007651558,0.00009047908,0.90904063,0.00048767915,0.000088551686,1.9586116e-7,0.06875982,0.0003584632,0.0000013886926,0.0018407118,0.018311245,0.00025564962],"about_ca_topic_score_codex":0.0002773158,"about_ca_topic_score_gemma":0.0027121753,"teacher_disagreement_score":0.10078136,"about_ca_system_score_codex":0.0021975574,"about_ca_system_score_gemma":0.0018963037,"threshold_uncertainty_score":0.99881434},"labels":[],"label_agreement":null},{"id":"W4402798678","doi":"10.1080/02602938.2024.2402956","title":"Connecting students’ descriptions of classroom assessment in higher education with wellness","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Higher education; Pedagogy; Mathematics education; Medical education; Medicine","score_opus":0.1083461691642832,"score_gpt":0.47227263569398775,"score_spread":0.36392646652970456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402798678","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.86613864,0.00076284097,0.00015973777,0.0098765865,0.0081937285,0.002107327,0.000004345092,0.00010768058,0.112649135],"genre_scores_gemma":[0.98518676,0.000106212785,0.0021385744,0.00014955146,0.0006376542,0.0012285751,0.0001705372,0.000038655293,0.010343486],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9951348,0.0009686089,0.0007892225,0.0006196498,0.0020809537,0.00040678936],"domain_scores_gemma":[0.9983947,0.0003643012,0.00030643193,0.0003337024,0.00049588445,0.000104948944],"candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003671447,0.00026161992,0.0003158285,0.00080089155,0.00026579606,0.00046824574,0.0004209279,0.00017562449,0.0058099977],"category_scores_gemma":[0.00003322743,0.00026180787,0.000074778516,0.0019734967,0.00009748338,0.0011681685,0.000060987993,0.00042289976,0.000029194964],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017140175,0.0021997332,0.7342402,0.00010667238,0.0000713238,9.113738e-7,0.0075757294,0.00038741573,0.00034124692,0.23644115,0.0015240429,0.017094433],"study_design_scores_gemma":[0.0007089061,0.000088669294,0.9484112,0.00061993714,0.00015267977,5.0229784e-7,0.021160338,0.00041965878,0.000014683249,0.006377486,0.021731105,0.0003147954],"about_ca_topic_score_codex":0.0020853889,"about_ca_topic_score_gemma":0.0018798994,"teacher_disagreement_score":0.23006366,"about_ca_system_score_codex":0.002864198,"about_ca_system_score_gemma":0.0060352376,"threshold_uncertainty_score":0.99998343},"labels":[],"label_agreement":null},{"id":"W4402918453","doi":"10.1080/02602938.2024.2404634","title":"Authentic assessment: from panacea to criticality","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Panacea (medicine); Psychology; Criticality; Pedagogy; Mathematics education","score_opus":0.0740774940749726,"score_gpt":0.5029731262640805,"score_spread":0.4288956321891079,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402918453","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8427647,0.0005035729,0.0046500857,0.10484077,0.011611842,0.0022903935,0.000014589315,0.00022778369,0.033096276],"genre_scores_gemma":[0.9645715,0.000012847621,0.023642832,0.004905685,0.0011342085,0.0014946435,0.0012368187,0.000040730676,0.0029607415],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9965072,0.0003008,0.0007675593,0.0006770948,0.0014385706,0.00030872985],"domain_scores_gemma":[0.99827975,0.00020883462,0.00007415086,0.0006040218,0.0007093041,0.00012392443],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0023435405,0.00021565698,0.00026821057,0.00068072637,0.000093907394,0.00018661057,0.00014365707,0.0001664734,0.011819931],"category_scores_gemma":[0.000382907,0.00021483054,0.000071401526,0.0017290588,0.000054981178,0.00037328055,0.0000396454,0.00050171616,0.0003609374],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004532932,0.004238204,0.15538587,0.0006849106,0.00024037117,0.000008485202,0.0026046732,0.00013267071,0.005715792,0.24279395,0.14524515,0.4429046],"study_design_scores_gemma":[0.00049603137,0.00009284514,0.9086346,0.00069315225,0.0003225308,0.00000338665,0.0006947176,0.016447373,0.000069868795,0.01967533,0.052646495,0.00022367985],"about_ca_topic_score_codex":0.00024225097,"about_ca_topic_score_gemma":0.000011530622,"teacher_disagreement_score":0.75324875,"about_ca_system_score_codex":0.0026480644,"about_ca_system_score_gemma":0.0036421455,"threshold_uncertainty_score":0.9890834},"labels":[],"label_agreement":null},{"id":"W4404012066","doi":"10.1080/02602938.2024.2419604","title":"The paradox of inclusive assessment","year":2024,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Psychology; Pedagogy; Higher education; Mathematics education; Evaluation methods; Political science","score_opus":0.07941909821958311,"score_gpt":0.5416770865350428,"score_spread":0.46225798831545967,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404012066","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.32581407,0.0047082263,0.00025956272,0.20873062,0.021312833,0.0024927366,0.000004018771,0.00018838079,0.43648952],"genre_scores_gemma":[0.9847703,0.00030766102,0.0014186072,0.00015774863,0.0006265626,0.00074125646,0.000037358954,0.000017179316,0.011923318],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.99545264,0.001981338,0.00051823497,0.00032362549,0.0014719364,0.00025220448],"domain_scores_gemma":[0.9968857,0.001829531,0.00029538068,0.0003164332,0.0006019811,0.000070974114],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.007689833,0.00012205594,0.00013034958,0.00019044928,0.0005600943,0.0003870326,0.0003253153,0.00009628107,0.0028607233],"category_scores_gemma":[0.00030781457,0.000103954306,0.000056891866,0.0011023392,0.00018728053,0.000808847,0.000039966482,0.00032323008,0.00006946822],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000006071895,0.000427131,0.06372272,0.000043453732,0.000054502547,3.1589863e-7,0.019090462,0.0004294134,0.00015413699,0.8260987,0.011740587,0.078232534],"study_design_scores_gemma":[0.00010961193,0.000026266487,0.39641654,0.00008942908,0.000061650506,2.6384268e-7,0.0076439586,0.00069191755,0.000010283996,0.016040707,0.5787998,0.00010955412],"about_ca_topic_score_codex":0.0012336815,"about_ca_topic_score_gemma":0.00036327465,"teacher_disagreement_score":0.81005794,"about_ca_system_score_codex":0.0016079021,"about_ca_system_score_gemma":0.009480536,"threshold_uncertainty_score":0.9980508},"labels":[],"label_agreement":null},{"id":"W4407793778","doi":"10.1080/02602938.2025.2467647","title":"Self-assessment design in a digital world: centring student agency","year":2025,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ontario Tech University","funders":"","keywords":"Centring; Agency (philosophy); Pedagogy; Psychology; Higher education; Mathematics education; Sociology; Engineering; Political science; Social science","score_opus":0.10112696830078945,"score_gpt":0.5256117004833262,"score_spread":0.42448473218253674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407793778","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6884228,0.00040693715,0.028097916,0.0026527888,0.0075917426,0.0022900214,0.0000014229903,0.00016086924,0.2703755],"genre_scores_gemma":[0.9612868,0.000005218812,0.026161242,0.00031703582,0.00014929802,0.0013237674,0.00006875086,0.000021622978,0.0106663015],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9949208,0.0027674134,0.000754438,0.00058404694,0.00060087966,0.00037242364],"domain_scores_gemma":[0.9984362,0.0005690297,0.00025909895,0.00041579842,0.0002805237,0.000039328574],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0072933636,0.00022338147,0.00025367585,0.0013270217,0.0001166877,0.00016842218,0.00024785238,0.000101807294,0.0019259205],"category_scores_gemma":[0.000065735505,0.00024020518,0.000047727277,0.001990041,0.00002158104,0.00031323812,0.000053814787,0.000577623,0.000045309742],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021823536,0.002664872,0.79104424,0.000028257133,0.00007605596,0.000001627482,0.0044116885,0.0011196132,0.00012464679,0.1135345,0.0014900333,0.08548266],"study_design_scores_gemma":[0.0012998193,0.000056829394,0.98052114,0.00013946381,0.0000472568,4.7339483e-7,0.0010236453,0.001135451,0.000008332275,0.008272449,0.00729104,0.00020409166],"about_ca_topic_score_codex":0.00009443933,"about_ca_topic_score_gemma":0.000011763733,"teacher_disagreement_score":0.27286392,"about_ca_system_score_codex":0.0027184645,"about_ca_system_score_gemma":0.001492627,"threshold_uncertainty_score":0.9989865},"labels":[],"label_agreement":null},{"id":"W4408054097","doi":"10.1080/02602938.2025.2468848","title":"Enhancing the structure of feedback forms increases trustworthiness and usefulness of peer feedback","year":2025,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Peer feedback; Trustworthiness; Psychology; Peer evaluation; Peer review; Higher education; Negative feedback; Social psychology; Mathematics education; Political science; Engineering","score_opus":0.039775825084288216,"score_gpt":0.4023611073920225,"score_spread":0.3625852823077343,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408054097","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9792331,0.0002961981,0.00005870659,0.0039882,0.001201815,0.0009705766,0.0000057092807,0.00001423138,0.01423148],"genre_scores_gemma":[0.99663484,0.000058791065,0.00042442128,0.00013312936,0.00014464316,0.00008417249,0.000052888558,0.000008691559,0.0024584348],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9972959,0.00049681606,0.0005790443,0.00026113057,0.0011648891,0.00020218661],"domain_scores_gemma":[0.9978659,0.00047022867,0.00039309348,0.00026212947,0.00097041676,0.000038242782],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002363922,0.00014878636,0.00026094905,0.0002338015,0.00024567952,0.00008624493,0.00028708755,0.00012289212,0.0010055809],"category_scores_gemma":[0.00021942731,0.00011661757,0.000051116094,0.0009868171,0.00021792972,0.0004477041,0.00006857739,0.00015374256,0.0000010104732],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000032735523,0.00040305246,0.78573704,0.00010938087,0.00007255455,4.722954e-8,0.008328027,0.00010976548,0.0026676161,0.19364256,0.0009346095,0.007962586],"study_design_scores_gemma":[0.0006179263,0.000021576247,0.9642157,0.0001597776,0.00013209369,1.08123174e-7,0.016296698,0.000070713315,0.00094266154,0.015874553,0.0015504136,0.0001177659],"about_ca_topic_score_codex":0.0021561747,"about_ca_topic_score_gemma":0.0021649667,"teacher_disagreement_score":0.17847864,"about_ca_system_score_codex":0.00028596475,"about_ca_system_score_gemma":0.0016851535,"threshold_uncertainty_score":0.9999076},"labels":[],"label_agreement":null},{"id":"W4411690222","doi":"10.1080/02602938.2025.2523598","title":"Open book assessment: performance and student perceptions in a UK veterinary programme","year":2025,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Perception; Psychology; Medical education; Pedagogy; Medicine","score_opus":0.08350136233889723,"score_gpt":0.5151767344608628,"score_spread":0.43167537212196555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411690222","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9310804,0.00026565389,0.00007955747,0.018691959,0.0013552435,0.0033931707,7.935266e-7,0.000028369748,0.04510486],"genre_scores_gemma":[0.9749726,0.00016847845,0.009934559,0.0033999933,0.00012243183,0.0050217025,0.00037080905,0.000016287588,0.0059931567],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9975964,0.00026405754,0.0007063995,0.00049143034,0.0006626579,0.0002790808],"domain_scores_gemma":[0.99892145,0.0000652323,0.00016181721,0.0004350228,0.0003673306,0.000049124803],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002597244,0.00018416325,0.000268631,0.0007841303,0.00014255261,0.0001605941,0.00023086858,0.00013174675,0.0020151236],"category_scores_gemma":[0.00009309645,0.0001906094,0.000025414463,0.0013163463,0.00008087508,0.0005603049,0.000154206,0.00042306865,0.00001170052],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003226612,0.0022591338,0.82033235,0.00016807957,0.000029440474,6.6275413e-7,0.00080497726,0.000031248062,0.00037854855,0.0059923627,0.008815789,0.16115515],"study_design_scores_gemma":[0.002100653,0.0002477451,0.9699165,0.00064921135,0.00009601682,0.000005865448,0.0020898487,0.0040732888,0.0000052570585,0.0007303096,0.019934487,0.00015078494],"about_ca_topic_score_codex":0.00011952064,"about_ca_topic_score_gemma":0.000019582616,"teacher_disagreement_score":0.16100436,"about_ca_system_score_codex":0.0022562037,"about_ca_system_score_gemma":0.0031391606,"threshold_uncertainty_score":0.9988972},"labels":[],"label_agreement":null},{"id":"W4411702131","doi":"10.1080/02602938.2025.2524095","title":"The role of digital technology in authentic assessment: perspectives of university educators","year":2025,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Higher education; Pedagogy; Technology integration; Psychology; Technological literacy; Mathematics education; Educational technology; Sociology; Engineering ethics; Medical education; Engineering; Political science; Medicine","score_opus":0.020031202184941713,"score_gpt":0.4013315858594782,"score_spread":0.3813003836745365,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411702131","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.79778945,0.00057103613,0.000015314086,0.007047066,0.0007123656,0.0006962886,0.000002003636,0.00001606967,0.1931504],"genre_scores_gemma":[0.9967351,0.00013591564,0.0001916124,0.000006614732,0.000034772074,0.000050936185,0.000011111792,0.000004499231,0.0028294502],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9982409,0.00043433785,0.0003429535,0.00022701464,0.00056676887,0.00018803349],"domain_scores_gemma":[0.9987124,0.00036372413,0.00025903867,0.00022239778,0.00041998,0.000022462806],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017370684,0.0000935868,0.00016374444,0.00064958,0.00018595006,0.0000528939,0.00036377882,0.0000991389,0.0002252865],"category_scores_gemma":[0.000117662385,0.00009138804,0.000049026385,0.0018698239,0.00030045825,0.00036620733,0.00005930271,0.00015088065,0.0000018842925],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000070386723,0.00044266504,0.5181791,0.0000052576506,0.000017753255,2.6659682e-8,0.0040310114,0.000009446898,0.00011155532,0.4704014,0.000048084992,0.006746643],"study_design_scores_gemma":[0.00030978987,0.000019934289,0.69859797,0.00005164179,0.000035421206,1.8504936e-8,0.22888938,0.00009931581,0.000028510418,0.06774849,0.0041573653,0.00006216076],"about_ca_topic_score_codex":0.00052096753,"about_ca_topic_score_gemma":0.00033725597,"teacher_disagreement_score":0.40265292,"about_ca_system_score_codex":0.0011892088,"about_ca_system_score_gemma":0.003611867,"threshold_uncertainty_score":0.6407297},"labels":[],"label_agreement":null},{"id":"W4416613148","doi":"10.1080/02602938.2025.2587246","title":"What should we be assessing exactly? Higher education staff narratives on gen AI integration of assessment in a postplagiarism era","year":2025,"lang":"en","type":"article","venue":"Assessment & Evaluation in Higher Education","topic":"Academic integrity and plagiarism","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Calgary Laboratory Services; Brock University; University of Calgary","funders":"","keywords":"Higher education; Narrative; Qualitative research; Professional development; Semi-structured interview; Focus group; Educational assessment","score_opus":0.11281127226536691,"score_gpt":0.49540280718684077,"score_spread":0.38259153492147385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416613148","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53853786,0.0014194744,0.00089997135,0.29815227,0.01665317,0.003179238,0.000011040987,0.00008901145,0.14105797],"genre_scores_gemma":[0.9815541,0.00056327065,0.00313843,0.0036911825,0.0004630474,0.0008635154,0.0003892441,0.00002335266,0.0093138525],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9930885,0.002947254,0.0011204153,0.00068602496,0.001734975,0.00042283113],"domain_scores_gemma":[0.9970575,0.00069619576,0.000623391,0.00041199586,0.0011117116,0.0000991878],"candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.006313928,0.00033902223,0.0004279833,0.0011753656,0.00036146192,0.0005433304,0.0004172318,0.00093246327,0.0024617892],"category_scores_gemma":[0.00020460229,0.00035055084,0.000099831996,0.0016572009,0.0001776194,0.0032974926,0.000047911093,0.0021052673,0.000007966151],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006242713,0.0035811204,0.024454437,0.000115605966,0.00006542753,4.5838098e-7,0.03261785,0.00058768335,0.0030473762,0.78374994,0.022506768,0.12921092],"study_design_scores_gemma":[0.0010320309,0.00015924264,0.6647667,0.0026193396,0.00012121153,3.421576e-7,0.115502164,0.0012776474,0.0004483693,0.14779405,0.06572633,0.0005525155],"about_ca_topic_score_codex":0.0029456455,"about_ca_topic_score_gemma":0.0012349014,"teacher_disagreement_score":0.6403123,"about_ca_system_score_codex":0.0036980554,"about_ca_system_score_gemma":0.011979359,"threshold_uncertainty_score":0.9998946},"labels":[],"label_agreement":null}]}