{"meta":{"query_hash":"7a547a233b1e","filters":{"venue":"International Journal of Assessment Tools in Education"},"cohort_total":15,"direct_labels_cover":0,"predictions_cover":15,"exported":15,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/7a547a233b1e","api":"https://metacan.xera.ac/api/v1/cohort?venue=International+Journal+of+Assessment+Tools+in+Education"},"results":[{"id":"W2945378006","doi":"10.21449/ijate.515085","title":"Explanatory Item Response Models for Polytomous Item Responses","year":2019,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Polytomous Rasch model; Item response theory; Rating scale; Psychology; Econometrics; Psychometrics; Scale (ratio); Statistics; Clinical psychology; Mathematics; Developmental psychology","score_opus":0.41021247149057005,"score_gpt":0.560403145971193,"score_spread":0.15019067448062295,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2945378006","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9596146,0.00028887947,0.025928121,0.0032080098,0.008268993,0.00025453675,0.0000193284,0.000009500668,0.0024080295],"genre_scores_gemma":[0.9488794,0.000049375307,0.048786163,0.0004271049,0.00035036705,0.000018747753,0.000004207101,0.0000111640675,0.0014734351],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9957319,0.0008536553,0.0012945876,0.00027897858,0.0016556674,0.0001852478],"domain_scores_gemma":[0.9390366,0.057617124,0.001243131,0.00029896514,0.0017210912,0.00008308731],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01428318,0.00012532585,0.00030732062,0.0019452102,0.000039047263,0.00038917054,0.0012442077,0.00008060416,0.00017023402],"category_scores_gemma":[0.060425255,0.00009976969,0.00017186013,0.00060243125,0.00002597383,0.0013111933,0.00007715668,0.00023740307,0.000012485327],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0059539992,0.0008894622,0.42506132,0.00001200646,0.00012957538,0.000022869659,0.0011532597,0.005783636,0.007133816,0.013196617,0.013670032,0.5269934],"study_design_scores_gemma":[0.0017316401,0.0005559199,0.8536826,0.00022610836,0.000013258382,0.00019269658,0.0077574034,0.010863916,0.0006158293,0.09712924,0.026983475,0.00024793425],"about_ca_topic_score_codex":0.000009939391,"about_ca_topic_score_gemma":0.0000020660086,"teacher_disagreement_score":0.5267455,"about_ca_system_score_codex":0.0005061216,"about_ca_system_score_gemma":0.001361358,"threshold_uncertainty_score":0.9474892},"labels":[],"label_agreement":null},{"id":"W2993603560","doi":"10.21449/ijate.627361","title":"Educational data mining: A tutorial for the rattle package in R","year":2019,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Educational data mining; Context (archaeology); Data mining; Flexibility (engineering); Graphical user interface; Field (mathematics); Big data; Data science; Software; Data stream mining; Random forest; Set (abstract data type); Interface (matter); Machine learning","score_opus":0.044753681496017134,"score_gpt":0.42481545407214394,"score_spread":0.38006177257612683,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2993603560","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.61255985,0.00046602282,0.08229443,0.2540977,0.047426526,0.0006426629,0.000035921643,0.000016001763,0.0024608877],"genre_scores_gemma":[0.9467265,0.000050945688,0.05066529,0.00029088047,0.001695489,0.0000096279255,0.000051386174,0.0000057075385,0.0005041799],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9987737,0.000063380794,0.0004391097,0.00016970192,0.00044679592,0.0001073095],"domain_scores_gemma":[0.99795306,0.0009645981,0.00035475803,0.0003399741,0.00035955745,0.000028058446],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011240321,0.00007241771,0.00011382672,0.00024947524,0.000022625987,0.0002944275,0.0019215107,0.000033483673,0.00003440801],"category_scores_gemma":[0.00048808035,0.000056392262,0.000046602516,0.00018867875,0.000013816406,0.0010628523,0.00014615506,0.00021060325,0.0000063924126],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011138166,0.0028966132,0.37890577,0.000034276287,0.00022278652,0.0000069240027,0.0023089172,0.008437789,0.000988471,0.24476102,0.034248292,0.32707775],"study_design_scores_gemma":[0.003310015,0.0002970522,0.5814169,0.0005173823,0.000031431693,0.000105401115,0.0024389252,0.2783603,0.00008301911,0.0374153,0.09565817,0.0003661389],"about_ca_topic_score_codex":0.000018665063,"about_ca_topic_score_gemma":0.000018913133,"teacher_disagreement_score":0.33416665,"about_ca_system_score_codex":0.00024593805,"about_ca_system_score_gemma":0.0019550717,"threshold_uncertainty_score":0.3570677},"labels":[],"label_agreement":null},{"id":"W3121031673","doi":"10.21449/ijate.705426","title":"Examining the Measurement Invariance of TIMSS 2015 Mathematics Liking Scale through Different Methods","year":2021,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Rasch model; Measurement invariance; Mathematics; Statistics; Metric (unit); Homogeneity (statistics); Scale (ratio); Level of measurement; Descriptive statistics; Polytomous Rasch model; Scale invariance; Econometrics; Psychology; Structural equation modeling; Mathematics education; Social psychology; Item response theory; Confirmatory factor analysis; Psychometrics; Geography; Engineering","score_opus":0.6714988377706251,"score_gpt":0.597784172952272,"score_spread":0.07371466481835309,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3121031673","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.47132543,0.00084897515,0.50868154,0.0037468185,0.0067747,0.0001153877,0.0000023370217,0.0000043139858,0.008500516],"genre_scores_gemma":[0.5875616,0.00013582583,0.41189492,0.00012657813,0.00021398066,0.0000053773247,0.0000010047634,0.000004818962,0.000055908047],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99466753,0.00093826855,0.00153472,0.00020293501,0.0025251296,0.00013140503],"domain_scores_gemma":[0.98264,0.012059889,0.0018358849,0.00034294333,0.0030812914,0.000039939623],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014422799,0.00010983237,0.0003433503,0.00031808825,0.0000483165,0.00030278077,0.0010973971,0.000047842303,0.00017592228],"category_scores_gemma":[0.037111863,0.000068408095,0.00011401016,0.0007532737,0.00004763619,0.00053426955,0.00017299465,0.00028995052,0.0000014420008],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027022064,0.0011863293,0.11740626,0.000016405795,0.00015567403,0.000008258029,0.0036932933,0.0015565936,0.01422622,0.0074429153,0.0024011997,0.85187984],"study_design_scores_gemma":[0.000676987,0.00012273055,0.73414475,0.0009177495,0.00004705096,0.00015302969,0.028425716,0.0028674407,0.013437243,0.21420504,0.004822533,0.00017974747],"about_ca_topic_score_codex":0.000008941636,"about_ca_topic_score_gemma":0.000009260283,"teacher_disagreement_score":0.85170007,"about_ca_system_score_codex":0.00034833432,"about_ca_system_score_gemma":0.0007300253,"threshold_uncertainty_score":0.97099894},"labels":[],"label_agreement":null},{"id":"W4309232742","doi":"10.21449/ijate.1124382","title":"Automatic story and item generation for reading comprehension assessments with transformers","year":2022,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University of Edmonton; University of Alberta","funders":"University of Alberta","keywords":"Fluency; Reading comprehension; Computer science; Comprehension; Literacy; Reading (process); Mathematics education; Multimedia; Psychology; Pedagogy; Linguistics","score_opus":0.04583102892320276,"score_gpt":0.3728767693014102,"score_spread":0.32704574037820744,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309232742","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.54040986,0.000045863595,0.4559708,0.0017121627,0.0015538542,0.00016717109,0.0000018044134,0.000007893243,0.00013060111],"genre_scores_gemma":[0.8342242,0.000020746307,0.16529842,0.0002271676,0.00013749348,0.00004987362,0.000011497235,0.000005787927,0.000024788194],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99856496,0.00011133183,0.00035599992,0.00016227263,0.0007102661,0.000095143885],"domain_scores_gemma":[0.9990141,0.00017503553,0.00032916846,0.0000906258,0.00035250123,0.00003859523],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009490437,0.00007957001,0.00011961938,0.0003330865,0.00010951929,0.00017013187,0.00036850566,0.0000178409,0.000012494188],"category_scores_gemma":[0.00003357241,0.00007575766,0.000033353102,0.000104775136,0.000010191344,0.0010947724,0.00005216962,0.00019925053,1.0406878e-7],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004061078,0.0004935503,0.009560666,0.00002838621,0.00009814185,0.000008712844,0.0015504368,0.014266699,0.0071439743,0.023378775,0.0006650601,0.942765],"study_design_scores_gemma":[0.0013998131,0.00048097153,0.027617468,0.000117723706,0.000019052448,0.0002372603,0.0015938093,0.9626383,0.00023863325,0.0015342212,0.003955268,0.00016749026],"about_ca_topic_score_codex":0.000007716158,"about_ca_topic_score_gemma":0.000004812533,"teacher_disagreement_score":0.9483716,"about_ca_system_score_codex":0.0008839525,"about_ca_system_score_gemma":0.0007870984,"threshold_uncertainty_score":0.30893078},"labels":[],"label_agreement":null},{"id":"W4323355119","doi":"10.21449/ijate.1212539","title":"The bibliometric journey of IJATE from local to global","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Publish or perish; Citation impact; Citation; Library science; Bibliometrics; Web of science; Impact factor; Publication; Visibility; Scopus; Citation analysis; Computer science; Political science; Publishing; Geography; MEDLINE; Law","score_opus":0.1766345083418794,"score_gpt":0.5757474426469738,"score_spread":0.3991129343050943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323355119","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9104001,0.00019562023,0.0035508121,0.07004661,0.007639609,0.00015191079,0.000013589468,0.000013405521,0.00798832],"genre_scores_gemma":[0.9951742,0.0007954164,0.0029428399,0.00022765243,0.0006520259,0.000006933775,0.0000076220667,0.0000045079864,0.00018878841],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9969222,0.0005047163,0.0006096244,0.00009465864,0.0017307901,0.00013800623],"domain_scores_gemma":[0.9964287,0.001561189,0.0007613204,0.00009864586,0.0010624821,0.00008763283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0054256106,0.000061506435,0.00011342389,0.003567566,0.00012879026,0.00033621286,0.0008541103,0.00004419923,0.000107158776],"category_scores_gemma":[0.004546916,0.000050281218,0.00006935142,0.007909968,0.000072348616,0.00094448717,0.00006582499,0.00019853713,0.000031006777],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000774389,0.00035891752,0.19142163,0.0000022903291,0.0001310298,0.0000047865337,0.003542532,0.0034575614,0.00015598787,0.0370072,0.019165635,0.744675],"study_design_scores_gemma":[0.00025783834,0.000058491187,0.88092035,0.00009911922,0.00001596233,0.000002927453,0.01480893,0.00020798338,0.000047238,0.01486382,0.0886532,0.00006414841],"about_ca_topic_score_codex":0.001960207,"about_ca_topic_score_gemma":0.00062918686,"teacher_disagreement_score":0.74461085,"about_ca_system_score_codex":0.00093781453,"about_ca_system_score_gemma":0.002034094,"threshold_uncertainty_score":0.54434115},"labels":[],"label_agreement":null},{"id":"W4379929879","doi":"10.21449/ijate.1249297","title":"Automatic item generation for online measurement and evaluation: Turkish literature items","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Turkish; Computer science; Field (mathematics); Test (biology); Item bank; Item analysis; Subject-matter expert; Subject matter; Item response theory; Data science; Artificial intelligence; Statistics; Psychometrics; Curriculum; Expert system; Psychology; Mathematics","score_opus":0.10478086886775079,"score_gpt":0.4375316124065066,"score_spread":0.33275074353875583,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379929879","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92398113,0.0009832768,0.03083363,0.03464446,0.008637823,0.00065875414,0.000015672094,0.00006538253,0.00017985761],"genre_scores_gemma":[0.92310154,0.00028500543,0.07509401,0.00031950878,0.00083972473,0.00014777269,0.00015216609,0.000008043725,0.000052227642],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997789,0.00012256518,0.0006206944,0.00023519513,0.0010924733,0.00014001806],"domain_scores_gemma":[0.9967818,0.00023820765,0.00045035026,0.00018324473,0.0022926934,0.00005365008],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024757271,0.00012408386,0.00015668289,0.0007369982,0.00006557767,0.0003837192,0.00060103106,0.00008690094,0.000013278097],"category_scores_gemma":[0.00056182046,0.000116546216,0.000064045504,0.0004929319,0.000021782284,0.0012515684,0.00007170073,0.00020968064,0.0000018727577],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009323131,0.0009933664,0.008049111,0.00004219787,0.00014868515,0.0000059619033,0.0010186839,0.0007767601,0.0045546116,0.057077833,0.013062028,0.91426146],"study_design_scores_gemma":[0.001914018,0.00029808737,0.425685,0.0008492951,0.00005199415,0.00018308262,0.0011421373,0.50388867,0.00094141735,0.05187443,0.012830901,0.00034098225],"about_ca_topic_score_codex":0.0000026050866,"about_ca_topic_score_gemma":0.00001106591,"teacher_disagreement_score":0.91392046,"about_ca_system_score_codex":0.0006997156,"about_ca_system_score_gemma":0.0015346261,"threshold_uncertainty_score":0.47526166},"labels":[],"label_agreement":null},{"id":"W4389915531","doi":"10.21449/ijate.1321061","title":"A data pipeline for e-large-scale assessments: Better automation, quality assurance, and efficiency","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Mental Health Research Topics","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Northern Alberta Institute of Technology","funders":"International Atomic Energy Agency","keywords":"Workflow; Computer science; Pipeline (software); Scale (ratio); Quality assurance; Automation; Interface (matter); Quality (philosophy); Data science; Data quality; Risk analysis (engineering); Data mining; Database; Engineering; Operations management; Metric (unit)","score_opus":0.16544671294430488,"score_gpt":0.5848930899388985,"score_spread":0.41944637699459364,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389915531","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8682092,0.000267524,0.09822825,0.017924018,0.009729646,0.00088926806,0.0005611629,0.00003949528,0.004151454],"genre_scores_gemma":[0.9689685,0.00026945965,0.026223067,0.0010384163,0.0011731667,0.00014321064,0.0009858629,0.00002315677,0.0011751294],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99743164,0.0002816895,0.0009689483,0.0002811443,0.0007636819,0.00027288744],"domain_scores_gemma":[0.9975968,0.0007812559,0.00054897513,0.0003349976,0.0006378611,0.00010008811],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004148097,0.00011019514,0.0002076919,0.000515919,0.00006500458,0.0001396438,0.00073702005,0.000074201016,0.00024146604],"category_scores_gemma":[0.00048708322,0.00010555156,0.000042954383,0.0002906015,0.000029675135,0.00079614174,0.00015165847,0.00027435683,0.000019122166],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035221103,0.0038006355,0.25430873,0.00024777086,0.00021066413,0.000020550502,0.0023863653,0.00008419181,0.00047993578,0.017668206,0.18498339,0.5354573],"study_design_scores_gemma":[0.0022194807,0.00016156767,0.90993786,0.00019957277,0.000015324122,0.00003504232,0.0042043533,0.012587165,0.000019387155,0.0035826557,0.06689382,0.00014375932],"about_ca_topic_score_codex":0.000043212858,"about_ca_topic_score_gemma":0.0000860449,"teacher_disagreement_score":0.6556291,"about_ca_system_score_codex":0.00039431517,"about_ca_system_score_gemma":0.00078286405,"threshold_uncertainty_score":0.43042678},"labels":[],"label_agreement":null},{"id":"W4389923528","doi":"10.21449/ijate.1394194","title":"Language models in automated essay scoring: Insights for the Turkish language","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Turkish; Transformative learning; Computer science; Language model; Artificial intelligence; Natural language processing; Intersection (aeronautics); Transformer; Linguistics; Sociology; Engineering","score_opus":0.0407797704534206,"score_gpt":0.39349338321123484,"score_spread":0.3527136127578142,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389923528","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74063826,0.00031539696,0.2491135,0.0045079403,0.00427133,0.0002527768,0.0000024394344,0.00007713098,0.0008212284],"genre_scores_gemma":[0.97952545,0.00006187551,0.01969401,0.00021411202,0.00032719498,0.00003938811,0.000009102637,0.0000069435896,0.00012193131],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987821,0.00006065214,0.00043919747,0.00014420784,0.0004429922,0.00013087218],"domain_scores_gemma":[0.9989404,0.00037672295,0.00023720418,0.00018849394,0.00022789987,0.000029268891],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007599398,0.000078455916,0.000110634435,0.000465956,0.00002811712,0.00024259777,0.001018761,0.00003735898,0.0000042128972],"category_scores_gemma":[0.00013726481,0.000060843566,0.00005530272,0.00033268603,0.000009985926,0.0010916764,0.00009946679,0.00018301215,0.000002599705],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036746533,0.0006429817,0.006981761,0.000035021516,0.00012302092,0.00010327471,0.04177945,0.36620116,0.0036417593,0.15149677,0.0023769815,0.42658108],"study_design_scores_gemma":[0.0005118141,0.000024429177,0.034185804,0.00015596322,0.000003899799,0.000020463774,0.002923865,0.95160836,0.00029440023,0.009800757,0.0003855107,0.00008474191],"about_ca_topic_score_codex":0.000075988464,"about_ca_topic_score_gemma":0.000049417285,"teacher_disagreement_score":0.5854072,"about_ca_system_score_codex":0.0004066423,"about_ca_system_score_gemma":0.0006739506,"threshold_uncertainty_score":0.24811286},"labels":[],"label_agreement":null},{"id":"W4389925007","doi":"10.21449/ijate.1359348","title":"Automatic item generation for non-verbal reasoning items","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Verbal reasoning; Computer science; Test (biology); Item analysis; Cognition; Item response theory; Homogeneous; Subject-matter expert; Natural language processing; Artificial intelligence; Subject matter; Psychology; Cognitive psychology; Psychometrics; Pedagogy; Expert system; Mathematics; Developmental psychology","score_opus":0.08817372903914804,"score_gpt":0.4972479358557301,"score_spread":0.4090742068165821,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389925007","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9672849,0.00004639044,0.0018692047,0.0117305415,0.012119925,0.00037142815,0.000009762377,0.000029461553,0.0065383487],"genre_scores_gemma":[0.984994,0.00023825321,0.008437909,0.00024457462,0.004483025,0.00009518263,0.00017760816,0.0000135242635,0.0013159155],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99804175,0.00014474413,0.0006226869,0.00015202981,0.000829571,0.00020919797],"domain_scores_gemma":[0.99763733,0.00066206563,0.00053752685,0.000084771185,0.0009941524,0.000084147],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002158554,0.00010064467,0.00016097576,0.0005695026,0.00016259587,0.00032305426,0.0004285853,0.00008128827,0.00019288214],"category_scores_gemma":[0.00065305503,0.00010517188,0.00010869552,0.00038909103,0.000041301162,0.001060117,0.000023484676,0.00015960021,0.000014935144],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005804947,0.0014013308,0.2410517,0.00005184151,0.00026893156,0.000012211935,0.025236318,0.00127119,0.0046739723,0.16172951,0.09701511,0.4672298],"study_design_scores_gemma":[0.0020731427,0.00028570383,0.6352672,0.00066528143,0.00008064404,0.00003102356,0.056332443,0.055470027,0.00030345426,0.019787742,0.22910899,0.0005943266],"about_ca_topic_score_codex":0.00019847254,"about_ca_topic_score_gemma":0.00029674545,"teacher_disagreement_score":0.4666355,"about_ca_system_score_codex":0.0009298055,"about_ca_system_score_gemma":0.0036438587,"threshold_uncertainty_score":0.64640486},"labels":[],"label_agreement":null},{"id":"W4390145121","doi":"10.21449/ijate.1266808","title":"The complexity of the grading system in Turkish higher education","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Grading (engineering); Higher education; Concordance; Turkish; Mathematics education; Quarter (Canadian coin); Psychology; Medical education; Medicine; Engineering; Political science","score_opus":0.06223638291745076,"score_gpt":0.39129005974478154,"score_spread":0.3290536768273308,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390145121","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88285094,0.00032998025,0.0018382682,0.08020889,0.028697658,0.00038175343,0.0000038263006,0.00004163897,0.0056470283],"genre_scores_gemma":[0.9951675,0.000076709264,0.0039717825,0.00014178263,0.00025426204,0.000042020067,0.0000050361095,0.0000049502664,0.00033598303],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.9982439,0.00017887173,0.0006742019,0.00014029902,0.0006242435,0.00013844403],"domain_scores_gemma":[0.99819887,0.0003241654,0.00066278025,0.0002862141,0.00050067366,0.000027270471],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012062222,0.00009005745,0.00013156004,0.0004148519,0.000086368986,0.00014696554,0.0018042016,0.00005523327,0.00000666491],"category_scores_gemma":[0.000104605075,0.00006150488,0.00007920875,0.0008250377,0.00011517709,0.0006097783,0.00017088327,0.00032876193,0.0000035363591],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000005160424,0.00036866975,0.07599884,0.000012378804,0.000027797172,0.0000012968931,0.00024045884,0.0002578574,0.00030035077,0.8865238,0.0024864804,0.033776894],"study_design_scores_gemma":[0.0001957541,0.000022142007,0.91561735,0.00035743768,0.0000044499307,0.0000319112,0.0017382505,0.0009565426,0.00036081314,0.07492572,0.005721999,0.00006762793],"about_ca_topic_score_codex":0.000049444654,"about_ca_topic_score_gemma":0.000027546754,"teacher_disagreement_score":0.8396185,"about_ca_system_score_codex":0.00084785675,"about_ca_system_score_gemma":0.0020484652,"threshold_uncertainty_score":0.36338893},"labels":[],"label_agreement":null},{"id":"W4390265350","doi":"10.21449/ijate.1406304","title":"A dialectic on validity: Explanation-focused and the many ways of being human","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Canada Research Chairs; University of Cambridge","keywords":"Test (biology); External validity; Classical test theory; Test validity; Epistemology; Construct validity; Psychology; Variation (astronomy); Dialectic; Predictive validity; Cognitive psychology; Item response theory; Social psychology; Psychometrics","score_opus":0.10254702649635265,"score_gpt":0.44019930065982515,"score_spread":0.3376522741634725,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390265350","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95731854,0.00001967547,0.000231523,0.03894414,0.00095365435,0.00014025498,0.0000017388722,0.000006872956,0.0023835858],"genre_scores_gemma":[0.99710524,0.0001796507,0.0004955107,0.0015895814,0.0005075996,0.0000137068155,0.00002255016,0.0000067252195,0.000079444675],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9987117,0.00008475195,0.00050524325,0.00008572622,0.0005383704,0.00007420889],"domain_scores_gemma":[0.99284613,0.006334102,0.00034606768,0.00008928182,0.00034454366,0.00003986158],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0011588655,0.0000626771,0.00018975312,0.00031433656,0.000029105768,0.00004413916,0.00012301617,0.000041569263,0.000032737793],"category_scores_gemma":[0.009446448,0.000043545962,0.000075199896,0.0001649813,0.000058080022,0.00010330923,0.000021545098,0.0002544397,0.0000029532093],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013104354,0.0025637697,0.5657342,0.00006549699,0.000550053,0.00016204652,0.0038356676,0.00032624463,0.0017146745,0.21957794,0.019562136,0.18459736],"study_design_scores_gemma":[0.004426183,0.00051884213,0.95576,0.0019905153,0.00009262128,0.0000764872,0.00072220265,0.0005209153,0.00048220856,0.03468309,0.0006538078,0.000073098316],"about_ca_topic_score_codex":0.000034504337,"about_ca_topic_score_gemma":0.0000034948396,"teacher_disagreement_score":0.39002585,"about_ca_system_score_codex":0.00014401494,"about_ca_system_score_gemma":0.00032051996,"threshold_uncertainty_score":0.99889743},"labels":[],"label_agreement":null},{"id":"W4396978444","doi":"10.21449/ijate.1376160","title":"The difference between estimated and perceived item difficulty: An empirical study","year":2024,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Statistics; Empirical research; Econometrics; Mathematics education; Mathematics","score_opus":0.08975046533286718,"score_gpt":0.5107784354602035,"score_spread":0.42102797012733634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396978444","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99013275,0.00016271582,0.00015145601,0.0056187944,0.002500994,0.00022585124,0.0000026399891,0.000018239249,0.0011865712],"genre_scores_gemma":[0.99776673,0.00038784338,0.00048436414,0.000048127462,0.00096488144,0.000014998167,0.000009163883,0.000007304943,0.0003165973],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99808455,0.00031847184,0.00042534224,0.00015383569,0.00087601616,0.00014180451],"domain_scores_gemma":[0.9984225,0.0009151304,0.00015337832,0.000074779986,0.00034723524,0.000087001805],"candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0012243337,0.000095337855,0.00013362317,0.00019084424,0.0001962712,0.0012563962,0.00049963366,0.00004666487,0.000035927198],"category_scores_gemma":[0.00024548694,0.000066311695,0.00004395479,0.00020179545,0.00009665512,0.0007077001,0.000050294446,0.00028364995,0.0000026391258],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00001230981,0.00042878254,0.84924406,0.0000019191916,0.000080952545,0.000007748953,0.014091996,0.0000023516475,0.0001133291,0.0011073234,0.0002475331,0.13466166],"study_design_scores_gemma":[0.00021891436,0.00013354531,0.9597335,0.00009437176,0.00003134816,0.0000038333515,0.03638928,0.00020765902,0.0000025406569,0.0010265815,0.0020846042,0.000073847106],"about_ca_topic_score_codex":0.00018694773,"about_ca_topic_score_gemma":0.00038017845,"teacher_disagreement_score":0.13458782,"about_ca_system_score_codex":0.00044958587,"about_ca_system_score_gemma":0.00082517386,"threshold_uncertainty_score":0.9997804},"labels":[],"label_agreement":null},{"id":"W4401920683","doi":"10.21449/ijate.1475980","title":"The mental imagery scale for art students: Building and validating a short form","year":2024,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Creativity in Education and Neuroscience","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Creativity; Scale (ratio); Psychology; Reliability (semiconductor); Construct (python library); Cognition; Mental image; Content validity; Construct validity; Visual arts education; Criterion validity; Computer science; Cognitive psychology; Applied psychology; Mathematics education; Psychometrics; Social psychology; Clinical psychology","score_opus":0.04279848651872874,"score_gpt":0.5057800219547152,"score_spread":0.4629815354359864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401920683","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97138983,0.0002728757,0.004107632,0.006838623,0.014821494,0.00024547102,0.000009942494,0.000009239531,0.002304909],"genre_scores_gemma":[0.9952417,0.00019404436,0.0025463398,0.00024263182,0.00056359236,0.0000738926,0.000008517213,0.000009497242,0.0011197693],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99884844,0.00006114368,0.00040409045,0.00015249364,0.00041184612,0.00012198087],"domain_scores_gemma":[0.998824,0.000737858,0.00012851253,0.00008108787,0.00018601115,0.0000425462],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012610686,0.00007720597,0.00008659142,0.00021780738,0.00009279771,0.00072417286,0.0003441405,0.000025818734,0.000049399223],"category_scores_gemma":[0.00016047257,0.000060175575,0.00005894497,0.000119030345,0.00005711215,0.0006084475,0.00004404282,0.00017559642,0.0000030731176],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012726472,0.0015712894,0.16487,0.000027144108,0.00017052273,0.00001065919,0.007601304,0.00001535888,0.01567179,0.02863789,0.037994813,0.743302],"study_design_scores_gemma":[0.00096156396,0.0003835488,0.7285887,0.00079322665,0.00007050457,0.0008333742,0.026207842,0.001214589,0.0022626605,0.009481194,0.22888036,0.00032240487],"about_ca_topic_score_codex":0.000005331945,"about_ca_topic_score_gemma":0.000007914995,"teacher_disagreement_score":0.7429796,"about_ca_system_score_codex":0.0003053785,"about_ca_system_score_gemma":0.00029242193,"threshold_uncertainty_score":0.6983216},"labels":[],"label_agreement":null},{"id":"W4410004709","doi":"10.21449/ijate.1602294","title":"A review of automatic item generation techniques leveraging large language models","year":2025,"lang":"en","type":"review","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Data science","score_opus":0.07488583176221998,"score_gpt":0.4522201303716182,"score_spread":0.3773342986093982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410004709","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000008662538,0.6902927,0.30687407,0.00027386987,0.0015037506,0.0003276632,0.000005881699,0.000015761158,0.00069760927],"genre_scores_gemma":[0.00018899319,0.88404113,0.114864446,0.0003720572,0.0003522904,0.00006228051,0.000052004667,0.000009553406,0.00005724613],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9969439,0.00030319055,0.0017111525,0.00024128257,0.0006693978,0.00013110401],"domain_scores_gemma":[0.9967942,0.0002792411,0.0018030992,0.00037576247,0.00071124354,0.000036447735],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016350148,0.00020775716,0.0008400228,0.000914934,0.00002004316,0.0001573619,0.0015229647,0.00010614035,0.000018695617],"category_scores_gemma":[0.00030808203,0.00018422931,0.0003067066,0.00036686676,0.000008141082,0.0011610747,0.00020179256,0.00038832004,8.3464647e-7],"study_design_candidate":"design_other","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[2.1840273e-7,0.00017575936,0.0000039129604,0.014733081,0.00006589167,0.0000047466424,0.00014442616,0.00002694609,0.000002633165,0.008879484,0.0007667648,0.9751961],"study_design_scores_gemma":[0.00025303534,0.00005357057,0.0000146161665,0.7045194,0.0002617082,0.00021546967,0.00010476238,0.0517667,0.000031023446,0.0024799074,0.23988122,0.0004185924],"about_ca_topic_score_codex":0.000011119177,"about_ca_topic_score_gemma":0.000001990696,"teacher_disagreement_score":0.9747775,"about_ca_system_score_codex":0.0008648163,"about_ca_system_score_gemma":0.003529035,"threshold_uncertainty_score":0.75126535},"labels":[],"label_agreement":null},{"id":"W4414775956","doi":"10.21449/ijate.1566093","title":"Construction and validation of a multilingual diagnostic instrument for neuromyths and their origins","year":2025,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Neuroscience, Education and Cognitive Function","field":"Neuroscience","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Türkiye Bilimsel ve Teknolojik Araştırma Kurumu","keywords":"Relevance (law); Identification (biology); Robustness (evolution); Process (computing); Adaptation (eye); Key (lock); Qualitative research","score_opus":0.035631585674076316,"score_gpt":0.3779109001011886,"score_spread":0.3422793144271123,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414775956","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"codex-gemma-dda1882f352a","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98476666,0.0000308882,0.008108705,0.0016129204,0.0047192615,0.00028562424,0.000013939001,0.0000045735496,0.00045744624],"genre_scores_gemma":[0.99712545,0.00031305224,0.0020063925,0.00036890196,0.00010162726,0.000029308314,0.0000050626586,0.0000041956855,0.000046023662],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99903315,0.00009370364,0.00042360497,0.000177478,0.00019778417,0.00007429579],"domain_scores_gemma":[0.998049,0.0010790756,0.00038245518,0.000054982298,0.00040104208,0.000033416258],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002930239,0.00008429573,0.00012032358,0.00046637727,0.000049166312,0.00010431071,0.00012404539,0.000029333973,0.000009481837],"category_scores_gemma":[0.0021889296,0.00007615141,0.00003335135,0.00018933274,0.000104419676,0.00052178645,0.000026982369,0.00010990077,1.8245309e-7],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001769613,0.0010171947,0.08091438,0.000057700825,0.000022547667,0.0000012146705,0.00082999835,0.00009045007,0.3052131,0.043172292,0.000120743156,0.56838346],"study_design_scores_gemma":[0.0023831683,0.00045545446,0.24385649,0.00049912854,0.00004276631,0.00017879075,0.0046576965,0.0011798482,0.7204567,0.018350337,0.0077511733,0.0001884074],"about_ca_topic_score_codex":0.0000082890165,"about_ca_topic_score_gemma":0.0000018762267,"teacher_disagreement_score":0.56819504,"about_ca_system_score_codex":0.00015942262,"about_ca_system_score_gemma":0.00061946467,"threshold_uncertainty_score":0.3105364},"labels":[],"label_agreement":null}]}