{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":15,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":15,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"7a547a233b1e","filters":{"venue":"International Journal of Assessment Tools in Education"}},"results":[{"id":"W4309232742","doi":"10.21449/ijate.1124382","title":"Automatic story and item generation for reading comprehension assessments with transformers","year":2022,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University of Edmonton; University of Alberta","funders":"University of Alberta","keywords":"Fluency; Reading comprehension; Computer science; Comprehension; Literacy; Reading (process); Mathematics education; Multimedia; Psychology; Pedagogy; Linguistics","authors":[{"name":"Okan Bulut","is_ca":true},{"name":"Seyma N. Yildirim‐Erbasli","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04583102892320276,"gpt":0.3728767693014102,"spread":0.3270457403782074,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009490437,0.00007957001,0.0001196194,0.0003330865,0.0001095193,0.0001701319,0.0003685057,0.0000178409,0.00001249419],"category_scores_gemma":[0.00003357241,0.00007575766,0.0000333531,0.0001047751,0.00001019134,0.001094772,0.00005216962,0.0001992505,1.040688e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008839525,"about_ca_system_score_gemma":0.0007870984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007716158,"about_ca_topic_score_gemma":0.000004812533,"domain_scores_codex":[0.998565,0.0001113318,0.0003559999,0.0001622726,0.0007102661,0.00009514389],"domain_scores_gemma":[0.9990141,0.0001750355,0.0003291685,0.0000906258,0.0003525012,0.00003859523],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004061078,0.0004935503,0.009560666,0.00002838621,0.00009814185,0.000008712844,0.001550437,0.0142667,0.007143974,0.02337877,0.0006650601,0.942765],"study_design_scores_gemma":[0.001399813,0.0004809715,0.02761747,0.0001177237,0.00001905245,0.0002372603,0.001593809,0.9626383,0.0002386332,0.001534221,0.003955268,0.0001674903],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5404099,0.0000458636,0.4559708,0.001712163,0.001553854,0.0001671711,0.000001804413,0.000007893243,0.0001306011],"genre_scores_gemma":[0.8342242,0.00002074631,0.1652984,0.0002271676,0.0001374935,0.00004987362,0.00001149724,0.000005787927,0.00002478819],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9483716,"threshold_uncertainty_score":0.3089308,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2945378006","doi":"10.21449/ijate.515085","title":"Explanatory Item Response Models for Polytomous Item Responses","year":2019,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Polytomous Rasch model; Item response theory; Rating scale; Psychology; Econometrics; Psychometrics; Scale (ratio); Statistics; Clinical psychology; Mathematics; Developmental psychology","authors":[{"name":"Luke Stanke","is_ca":false},{"name":"Okan Bulut","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4102124714905701,"gpt":0.560403145971193,"spread":0.150190674480623,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01428318,0.0001253259,0.0003073206,0.00194521,0.00003904726,0.0003891705,0.001244208,0.00008060416,0.000170234],"category_scores_gemma":[0.06042526,0.00009976969,0.0001718601,0.0006024312,0.00002597383,0.001311193,0.00007715668,0.0002374031,0.00001248533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005061216,"about_ca_system_score_gemma":0.001361358,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000009939391,"about_ca_topic_score_gemma":0.000002066009,"domain_scores_codex":[0.9957319,0.0008536553,0.001294588,0.0002789786,0.001655667,0.0001852478],"domain_scores_gemma":[0.9390366,0.05761712,0.001243131,0.0002989651,0.001721091,0.00008308731],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005953999,0.0008894622,0.4250613,0.00001200646,0.0001295754,0.00002286966,0.00115326,0.005783636,0.007133816,0.01319662,0.01367003,0.5269934],"study_design_scores_gemma":[0.00173164,0.0005559199,0.8536826,0.0002261084,0.00001325838,0.0001926966,0.007757403,0.01086392,0.0006158293,0.09712924,0.02698348,0.0002479343],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9596146,0.0002888795,0.02592812,0.00320801,0.008268993,0.0002545367,0.0000193284,0.000009500668,0.00240803],"genre_scores_gemma":[0.9488794,0.00004937531,0.04878616,0.0004271049,0.000350367,0.00001874775,0.000004207101,0.00001116407,0.001473435],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5267455,"threshold_uncertainty_score":0.9474892,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4390265350","doi":"10.21449/ijate.1406304","title":"A dialectic on validity: Explanation-focused and the many ways of being human","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Canada Research Chairs; University of Cambridge","keywords":"Test (biology); External validity; Classical test theory; Test validity; Epistemology; Construct validity; Psychology; Variation (astronomy); Dialectic; Predictive validity; Cognitive psychology; Item response theory; Social psychology; Psychometrics","authors":[{"name":"Bruno D. Zumbo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1025470264963527,"gpt":0.4401993006598252,"spread":0.3376522741634725,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.001158866,0.0000626771,0.0001897531,0.0003143366,0.00002910577,0.00004413916,0.0001230162,0.00004156926,0.00003273779],"category_scores_gemma":[0.009446448,0.00004354596,0.0000751999,0.0001649813,0.00005808002,0.0001033092,0.0000215451,0.0002544397,0.000002953209],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001440149,"about_ca_system_score_gemma":0.00032052,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003450434,"about_ca_topic_score_gemma":0.00000349484,"domain_scores_codex":[0.9987117,0.00008475195,0.0005052433,0.00008572622,0.0005383704,0.00007420889],"domain_scores_gemma":[0.9928461,0.006334102,0.0003460677,0.00008928182,0.0003445437,0.00003986158],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001310435,0.00256377,0.5657342,0.00006549699,0.000550053,0.0001620465,0.003835668,0.0003262446,0.001714675,0.2195779,0.01956214,0.1845974],"study_design_scores_gemma":[0.004426183,0.0005188421,0.95576,0.001990515,0.00009262128,0.0000764872,0.0007222026,0.0005209153,0.0004822086,0.03468309,0.0006538078,0.00007309832],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9573185,0.00001967547,0.000231523,0.03894414,0.0009536544,0.000140255,0.000001738872,0.000006872956,0.002383586],"genre_scores_gemma":[0.9971052,0.0001796507,0.0004955107,0.001589581,0.0005075996,0.00001370682,0.00002255016,0.000006725219,0.00007944468],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3900259,"threshold_uncertainty_score":0.9988974,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4410004709","doi":"10.21449/ijate.1602294","title":"A review of automatic item generation techniques leveraging large language models","year":2025,"lang":"en","type":"review","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Natural language processing; Data science","authors":[{"name":"Bin Tan","is_ca":true},{"name":"Nour Armoush","is_ca":true},{"name":"Elisabetta Mazzullo","is_ca":true},{"name":"Okan Bulut","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07488583176221998,"gpt":0.4522201303716182,"spread":0.3773342986093982,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001635015,0.0002077572,0.0008400228,0.000914934,0.00002004316,0.0001573619,0.001522965,0.0001061404,0.00001869562],"category_scores_gemma":[0.000308082,0.0001842293,0.0003067066,0.0003668668,0.000008141082,0.001161075,0.0002017926,0.00038832,8.346465e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008648163,"about_ca_system_score_gemma":0.003529035,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001111918,"about_ca_topic_score_gemma":0.000001990696,"domain_scores_codex":[0.9969439,0.0003031906,0.001711152,0.0002412826,0.0006693978,0.000131104],"domain_scores_gemma":[0.9967942,0.0002792411,0.001803099,0.0003757625,0.0007112435,0.00003644773],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[2.184027e-7,0.0001757594,0.00000391296,0.01473308,0.00006589167,0.000004746642,0.0001444262,0.00002694609,0.000002633165,0.008879484,0.0007667648,0.9751961],"study_design_scores_gemma":[0.0002530353,0.00005357057,0.00001461617,0.7045194,0.0002617082,0.0002154697,0.0001047624,0.0517667,0.00003102345,0.002479907,0.2398812,0.0004185924],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.000008662538,0.6902927,0.3068741,0.0002738699,0.001503751,0.0003276632,0.000005881699,0.00001576116,0.0006976093],"genre_scores_gemma":[0.0001889932,0.8840411,0.1148644,0.0003720572,0.0003522904,0.00006228051,0.00005200467,0.000009553406,0.00005724613],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9747775,"threshold_uncertainty_score":0.7512653,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2993603560","doi":"10.21449/ijate.627361","title":"Educational data mining: A tutorial for the rattle package in R","year":2019,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Educational data mining; Context (archaeology); Data mining; Flexibility (engineering); Graphical user interface; Field (mathematics); Big data; Data science; Software; Data stream mining; Random forest; Set (abstract data type); Interface (matter); Machine learning","authors":[{"name":"Okan Bulut","is_ca":true},{"name":"Hatice Çiğdem Yavuz","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04475368149601713,"gpt":0.4248154540721439,"spread":0.3800617725761268,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001124032,0.00007241771,0.0001138267,0.0002494752,0.00002262599,0.0002944275,0.001921511,0.00003348367,0.00003440801],"category_scores_gemma":[0.0004880803,0.00005639226,0.00004660252,0.0001886788,0.00001381641,0.001062852,0.0001461551,0.0002106032,0.000006392413],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000245938,"about_ca_system_score_gemma":0.001955072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001866506,"about_ca_topic_score_gemma":0.00001891313,"domain_scores_codex":[0.9987737,0.00006338079,0.0004391097,0.0001697019,0.0004467959,0.0001073095],"domain_scores_gemma":[0.9979531,0.0009645981,0.000354758,0.0003399741,0.0003595575,0.00002805845],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001113817,0.002896613,0.3789058,0.00003427629,0.0002227865,0.000006924003,0.002308917,0.008437789,0.000988471,0.244761,0.03424829,0.3270777],"study_design_scores_gemma":[0.003310015,0.0002970522,0.5814169,0.0005173823,0.00003143169,0.0001054011,0.002438925,0.2783603,0.00008301911,0.0374153,0.09565817,0.0003661389],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6125599,0.0004660228,0.08229443,0.2540977,0.04742653,0.0006426629,0.00003592164,0.00001600176,0.002460888],"genre_scores_gemma":[0.9467265,0.00005094569,0.05066529,0.0002908805,0.001695489,0.000009627925,0.00005138617,0.000005707539,0.0005041799],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3341666,"threshold_uncertainty_score":0.3570677,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4389923528","doi":"10.21449/ijate.1394194","title":"Language models in automated essay scoring: Insights for the Turkish language","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Turkish; Transformative learning; Computer science; Language model; Artificial intelligence; Natural language processing; Intersection (aeronautics); Transformer; Linguistics; Sociology; Engineering","authors":[{"name":"Tahereh Firoozi","is_ca":true},{"name":"Okan Bulut","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0407797704534206,"gpt":0.3934933832112348,"spread":0.3527136127578142,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007599398,0.00007845592,0.0001106344,0.000465956,0.00002811712,0.0002425978,0.001018761,0.00003735898,0.000004212897],"category_scores_gemma":[0.0001372648,0.00006084357,0.00005530272,0.000332686,0.000009985926,0.001091676,0.00009946679,0.0001830121,0.000002599705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004066423,"about_ca_system_score_gemma":0.0006739506,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007598846,"about_ca_topic_score_gemma":0.00004941729,"domain_scores_codex":[0.9987821,0.00006065214,0.0004391975,0.0001442078,0.0004429922,0.0001308722],"domain_scores_gemma":[0.9989404,0.000376723,0.0002372042,0.0001884939,0.0002278999,0.00002926889],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003674653,0.0006429817,0.006981761,0.00003502152,0.0001230209,0.0001032747,0.04177945,0.3662012,0.003641759,0.1514968,0.002376982,0.4265811],"study_design_scores_gemma":[0.0005118141,0.00002442918,0.0341858,0.0001559632,0.000003899799,0.00002046377,0.002923865,0.9516084,0.0002944002,0.009800757,0.0003855107,0.00008474191],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7406383,0.000315397,0.2491135,0.00450794,0.00427133,0.0002527768,0.000002439434,0.00007713098,0.0008212284],"genre_scores_gemma":[0.9795254,0.00006187551,0.01969401,0.000214112,0.000327195,0.00003938811,0.000009102637,0.00000694359,0.0001219313],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5854072,"threshold_uncertainty_score":0.2481129,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4389915531","doi":"10.21449/ijate.1321061","title":"A data pipeline for e-large-scale assessments: Better automation, quality assurance, and efficiency","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Mental Health Research Topics","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Northern Alberta Institute of Technology","funders":"International Atomic Energy Agency","keywords":"Workflow; Computer science; Pipeline (software); Scale (ratio); Quality assurance; Automation; Interface (matter); Quality (philosophy); Data science; Data quality; Risk analysis (engineering); Data mining; Database; Engineering; Operations management; Metric (unit)","authors":[{"name":"R.K. Schwarz","is_ca":false},{"name":"Hatice Cigdem Bulut","is_ca":true},{"name":"Charles ANİFOWOSE","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1654467129443049,"gpt":0.5848930899388985,"spread":0.4194463769945936,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004148097,0.0001101951,0.0002076919,0.000515919,0.00006500458,0.0001396438,0.00073702,0.00007420102,0.000241466],"category_scores_gemma":[0.0004870832,0.0001055516,0.00004295438,0.0002906015,0.00002967514,0.0007961417,0.0001516585,0.0002743568,0.00001912217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003943152,"about_ca_system_score_gemma":0.0007828641,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004321286,"about_ca_topic_score_gemma":0.0000860449,"domain_scores_codex":[0.9974316,0.0002816895,0.0009689483,0.0002811443,0.0007636819,0.0002728874],"domain_scores_gemma":[0.9975968,0.0007812559,0.0005489751,0.0003349976,0.0006378611,0.0001000881],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.000352211,0.003800635,0.2543087,0.0002477709,0.0002106641,0.0000205505,0.002386365,0.00008419181,0.0004799358,0.01766821,0.1849834,0.5354573],"study_design_scores_gemma":[0.002219481,0.0001615677,0.9099379,0.0001995728,0.00001532412,0.00003504232,0.004204353,0.01258717,0.00001938715,0.003582656,0.06689382,0.0001437593],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8682092,0.000267524,0.09822825,0.01792402,0.009729646,0.0008892681,0.0005611629,0.00003949528,0.004151454],"genre_scores_gemma":[0.9689685,0.0002694596,0.02622307,0.001038416,0.001173167,0.0001432106,0.0009858629,0.00002315677,0.001175129],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6556291,"threshold_uncertainty_score":0.4304268,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4396978444","doi":"10.21449/ijate.1376160","title":"The difference between estimated and perceived item difficulty: An empirical study","year":2024,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Statistics; Empirical research; Econometrics; Mathematics education; Mathematics","authors":[{"name":"Ayfer SAYIN","is_ca":false},{"name":"Okan Bulut","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08975046533286718,"gpt":0.5107784354602035,"spread":0.4210279701273363,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001224334,0.00009533785,0.0001336232,0.0001908442,0.0001962712,0.001256396,0.0004996337,0.00004666487,0.0000359272],"category_scores_gemma":[0.0002454869,0.0000663117,0.00004395479,0.0002017955,0.00009665512,0.0007077001,0.00005029445,0.0002836499,0.000002639126],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004495859,"about_ca_system_score_gemma":0.0008251739,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001869477,"about_ca_topic_score_gemma":0.0003801784,"domain_scores_codex":[0.9980845,0.0003184718,0.0004253422,0.0001538357,0.0008760162,0.0001418045],"domain_scores_gemma":[0.9984225,0.0009151304,0.0001533783,0.00007477999,0.0003472352,0.0000870018],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00001230981,0.0004287825,0.8492441,0.000001919192,0.00008095254,0.000007748953,0.014092,0.000002351647,0.0001133291,0.001107323,0.0002475331,0.1346617],"study_design_scores_gemma":[0.0002189144,0.0001335453,0.9597335,0.00009437176,0.00003134816,0.000003833351,0.03638928,0.000207659,0.000002540657,0.001026582,0.002084604,0.00007384711],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9901327,0.0001627158,0.000151456,0.005618794,0.002500994,0.0002258512,0.000002639989,0.00001823925,0.001186571],"genre_scores_gemma":[0.9977667,0.0003878434,0.0004843641,0.00004812746,0.0009648814,0.00001499817,0.000009163883,0.000007304943,0.0003165973],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1345878,"threshold_uncertainty_score":0.9997804,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4390145121","doi":"10.21449/ijate.1266808","title":"The complexity of the grading system in Turkish higher education","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Grading (engineering); Higher education; Concordance; Turkish; Mathematics education; Quarter (Canadian coin); Psychology; Medical education; Medicine; Engineering; Political science","authors":[{"name":"Recep GÜR","is_ca":false},{"name":"Mustafa Köroğlu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06223638291745076,"gpt":0.3912900597447815,"spread":0.3290536768273308,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001206222,0.00009005745,0.00013156,0.0004148519,0.00008636899,0.0001469655,0.001804202,0.00005523327,0.00000666491],"category_scores_gemma":[0.0001046051,0.00006150488,0.00007920875,0.0008250377,0.0001151771,0.0006097783,0.0001708833,0.0003287619,0.000003536359],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008478567,"about_ca_system_score_gemma":0.002048465,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004944465,"about_ca_topic_score_gemma":0.00002754675,"domain_scores_codex":[0.9982439,0.0001788717,0.0006742019,0.000140299,0.0006242435,0.000138444],"domain_scores_gemma":[0.9981989,0.0003241654,0.0006627803,0.0002862141,0.0005006737,0.00002727047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.000005160424,0.0003686697,0.07599884,0.0000123788,0.00002779717,0.000001296893,0.0002404588,0.0002578574,0.0003003508,0.8865238,0.00248648,0.03377689],"study_design_scores_gemma":[0.0001957541,0.00002214201,0.9156173,0.0003574377,0.000004449931,0.0000319112,0.00173825,0.0009565426,0.0003608131,0.07492572,0.005721999,0.00006762793],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8828509,0.0003299802,0.001838268,0.08020889,0.02869766,0.0003817534,0.000003826301,0.00004163897,0.005647028],"genre_scores_gemma":[0.9951675,0.00007670926,0.003971783,0.0001417826,0.000254262,0.00004202007,0.00000503611,0.000004950266,0.000335983],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8396185,"threshold_uncertainty_score":0.3633889,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3121031673","doi":"10.21449/ijate.705426","title":"Examining the Measurement Invariance of TIMSS 2015 Mathematics Liking Scale through Different Methods","year":2021,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Rasch model; Measurement invariance; Mathematics; Statistics; Metric (unit); Homogeneity (statistics); Scale (ratio); Level of measurement; Descriptive statistics; Polytomous Rasch model; Scale invariance; Econometrics; Psychology; Structural equation modeling; Mathematics education; Social psychology; Item response theory; Confirmatory factor analysis; Psychometrics; Geography; Engineering","authors":[{"name":"Zafer Ertürk","is_ca":false},{"name":"Esra Oyar","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.6714988377706251,"gpt":0.597784172952272,"spread":0.07371466481835309,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0144228,0.0001098324,0.0003433503,0.0003180882,0.0000483165,0.0003027808,0.001097397,0.0000478423,0.0001759223],"category_scores_gemma":[0.03711186,0.0000684081,0.0001140102,0.0007532737,0.00004763619,0.0005342696,0.0001729947,0.0002899505,0.000001442001],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003483343,"about_ca_system_score_gemma":0.0007300253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008941636,"about_ca_topic_score_gemma":0.000009260283,"domain_scores_codex":[0.9946675,0.0009382686,0.00153472,0.000202935,0.00252513,0.000131405],"domain_scores_gemma":[0.98264,0.01205989,0.001835885,0.0003429433,0.003081291,0.00003993962],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00002702206,0.001186329,0.1174063,0.00001640579,0.000155674,0.000008258029,0.003693293,0.001556594,0.01422622,0.007442915,0.0024012,0.8518798],"study_design_scores_gemma":[0.000676987,0.0001227305,0.7341447,0.0009177495,0.00004705096,0.0001530297,0.02842572,0.002867441,0.01343724,0.214205,0.004822533,0.0001797475],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4713254,0.0008489751,0.5086815,0.003746819,0.0067747,0.0001153877,0.000002337022,0.000004313986,0.008500516],"genre_scores_gemma":[0.5875616,0.0001358258,0.4118949,0.0001265781,0.0002139807,0.000005377325,0.000001004763,0.000004818962,0.00005590805],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8517001,"threshold_uncertainty_score":0.9709989,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4379929879","doi":"10.21449/ijate.1249297","title":"Automatic item generation for online measurement and evaluation: Turkish literature items","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Turkish; Computer science; Field (mathematics); Test (biology); Item bank; Item analysis; Subject-matter expert; Subject matter; Item response theory; Data science; Artificial intelligence; Statistics; Psychometrics; Curriculum; Expert system; Psychology; Mathematics","authors":[{"name":"Ayfer SAYIN","is_ca":false},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1047808688677508,"gpt":0.4375316124065066,"spread":0.3327507435387558,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002475727,0.0001240839,0.0001566829,0.0007369982,0.00006557767,0.0003837192,0.0006010311,0.00008690094,0.0000132781],"category_scores_gemma":[0.0005618205,0.0001165462,0.0000640455,0.0004929319,0.00002178228,0.001251568,0.00007170073,0.0002096806,0.000001872758],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006997156,"about_ca_system_score_gemma":0.001534626,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000002605087,"about_ca_topic_score_gemma":0.00001106591,"domain_scores_codex":[0.997789,0.0001225652,0.0006206944,0.0002351951,0.001092473,0.0001400181],"domain_scores_gemma":[0.9967818,0.0002382076,0.0004503503,0.0001832447,0.002292693,0.00005365008],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009323131,0.0009933664,0.008049111,0.00004219787,0.0001486852,0.000005961903,0.001018684,0.0007767601,0.004554612,0.05707783,0.01306203,0.9142615],"study_design_scores_gemma":[0.001914018,0.0002980874,0.425685,0.0008492951,0.00005199415,0.0001830826,0.001142137,0.5038887,0.0009414174,0.05187443,0.0128309,0.0003409823],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9239811,0.0009832768,0.03083363,0.03464446,0.008637823,0.0006587541,0.00001567209,0.00006538253,0.0001798576],"genre_scores_gemma":[0.9231015,0.0002850054,0.07509401,0.0003195088,0.0008397247,0.0001477727,0.0001521661,0.000008043725,0.00005222764],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9139205,"threshold_uncertainty_score":0.4752617,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4323355119","doi":"10.21449/ijate.1212539","title":"The bibliometric journey of IJATE from local to global","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Publish or perish; Citation impact; Citation; Library science; Bibliometrics; Web of science; Impact factor; Publication; Visibility; Scopus; Citation analysis; Computer science; Political science; Publishing; Geography; MEDLINE; Law","authors":[{"name":"Orhan Karamustafaoğlu","is_ca":false},{"name":"Metin Orbay","is_ca":false},{"name":"İzzet Kara","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1766345083418794,"gpt":0.5757474426469738,"spread":0.3991129343050943,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005425611,0.00006150643,0.0001134239,0.003567566,0.0001287903,0.0003362129,0.0008541103,0.00004419923,0.0001071588],"category_scores_gemma":[0.004546916,0.00005028122,0.00006935142,0.007909968,0.00007234862,0.0009444872,0.00006582499,0.0001985371,0.00003100678],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009378145,"about_ca_system_score_gemma":0.002034094,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001960207,"about_ca_topic_score_gemma":0.0006291869,"domain_scores_codex":[0.9969222,0.0005047163,0.0006096244,0.00009465864,0.00173079,0.0001380062],"domain_scores_gemma":[0.9964287,0.001561189,0.0007613204,0.00009864586,0.001062482,0.00008763283],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0000774389,0.0003589175,0.1914216,0.000002290329,0.0001310298,0.000004786534,0.003542532,0.003457561,0.0001559879,0.0370072,0.01916564,0.744675],"study_design_scores_gemma":[0.0002578383,0.00005849119,0.8809204,0.00009911922,0.00001596233,0.000002927453,0.01480893,0.0002079834,0.000047238,0.01486382,0.0886532,0.00006414841],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9104001,0.0001956202,0.003550812,0.07004661,0.007639609,0.0001519108,0.00001358947,0.00001340552,0.00798832],"genre_scores_gemma":[0.9951742,0.0007954164,0.00294284,0.0002276524,0.0006520259,0.000006933775,0.000007622067,0.000004507986,0.0001887884],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7446108,"threshold_uncertainty_score":0.5443411,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4414775956","doi":"10.21449/ijate.1566093","title":"Construction and validation of a multilingual diagnostic instrument for neuromyths and their origins","year":2025,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Neuroscience, Education and Cognitive Function","field":"Neuroscience","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université du Québec à Montréal","funders":"Türkiye Bilimsel ve Teknolojik Araştırma Kurumu","keywords":"Relevance (law); Identification (biology); Robustness (evolution); Process (computing); Adaptation (eye); Key (lock); Qualitative research","authors":[{"name":"Oktay Cem Adıgüzel","is_ca":false},{"name":"Patrice Potvin","is_ca":true},{"name":"Sibel Küçükkayhan","is_ca":false},{"name":"Derya Atik Kara","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03563158567407632,"gpt":0.3779109001011886,"spread":0.3422793144271123,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002930239,0.00008429573,0.0001203236,0.0004663773,0.00004916631,0.0001043107,0.0001240454,0.00002933397,0.000009481837],"category_scores_gemma":[0.00218893,0.00007615141,0.00003335135,0.0001893327,0.0001044197,0.0005217865,0.00002698237,0.0001099008,1.824531e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001594226,"about_ca_system_score_gemma":0.0006194647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000008289016,"about_ca_topic_score_gemma":0.000001876227,"domain_scores_codex":[0.9990332,0.00009370364,0.000423605,0.000177478,0.0001977842,0.00007429579],"domain_scores_gemma":[0.998049,0.001079076,0.0003824552,0.0000549823,0.0004010421,0.00003341626],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001769613,0.001017195,0.08091438,0.00005770082,0.00002254767,0.000001214671,0.0008299983,0.00009045007,0.3052131,0.04317229,0.0001207432,0.5683835],"study_design_scores_gemma":[0.002383168,0.0004554545,0.2438565,0.0004991285,0.00004276631,0.0001787908,0.004657696,0.001179848,0.7204567,0.01835034,0.007751173,0.0001884074],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9847667,0.0000308882,0.008108705,0.00161292,0.004719262,0.0002856242,0.000013939,0.00000457355,0.0004574462],"genre_scores_gemma":[0.9971254,0.0003130522,0.002006392,0.000368902,0.0001016273,0.00002930831,0.000005062659,0.000004195686,0.00004602366],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.568195,"threshold_uncertainty_score":0.3105364,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4401920683","doi":"10.21449/ijate.1475980","title":"The mental imagery scale for art students: Building and validating a short form","year":2024,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Creativity in Education and Neuroscience","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Northern Alberta Institute of Technology","funders":"","keywords":"Creativity; Scale (ratio); Psychology; Reliability (semiconductor); Construct (python library); Cognition; Mental image; Content validity; Construct validity; Visual arts education; Criterion validity; Computer science; Cognitive psychology; Applied psychology; Mathematics education; Psychometrics; Social psychology; Clinical psychology","authors":[{"name":"Handan Narin","is_ca":false},{"name":"Hatice Cigdem Bulut","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04279848651872874,"gpt":0.5057800219547152,"spread":0.4629815354359864,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001261069,0.00007720597,0.00008659142,0.0002178074,0.00009279771,0.0007241729,0.0003441405,0.00002581873,0.00004939922],"category_scores_gemma":[0.0001604726,0.00006017557,0.00005894497,0.0001190303,0.00005711215,0.0006084475,0.00004404282,0.0001755964,0.000003073118],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003053785,"about_ca_system_score_gemma":0.0002924219,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000005331945,"about_ca_topic_score_gemma":0.000007914995,"domain_scores_codex":[0.9988484,0.00006114368,0.0004040904,0.0001524936,0.0004118461,0.0001219809],"domain_scores_gemma":[0.998824,0.000737858,0.0001285125,0.00008108787,0.0001860112,0.0000425462],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001272647,0.001571289,0.16487,0.00002714411,0.0001705227,0.00001065919,0.007601304,0.00001535888,0.01567179,0.02863789,0.03799481,0.743302],"study_design_scores_gemma":[0.000961564,0.0003835488,0.7285887,0.0007932266,0.00007050457,0.0008333742,0.02620784,0.001214589,0.002262661,0.009481194,0.2288804,0.0003224049],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9713898,0.0002728757,0.004107632,0.006838623,0.01482149,0.000245471,0.000009942494,0.000009239531,0.002304909],"genre_scores_gemma":[0.9952417,0.0001940444,0.00254634,0.0002426318,0.0005635924,0.0000738926,0.000008517213,0.000009497242,0.001119769],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7429796,"threshold_uncertainty_score":0.6983216,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4389925007","doi":"10.21449/ijate.1359348","title":"Automatic item generation for non-verbal reasoning items","year":2023,"lang":"en","type":"article","venue":"International Journal of Assessment Tools in Education","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Verbal reasoning; Computer science; Test (biology); Item analysis; Cognition; Item response theory; Homogeneous; Subject-matter expert; Natural language processing; Artificial intelligence; Subject matter; Psychology; Cognitive psychology; Psychometrics; Pedagogy; Expert system; Mathematics; Developmental psychology","authors":[{"name":"Ayfer SAYIN","is_ca":false},{"name":"Sabiha BOZDAĞ","is_ca":false},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08817372903914804,"gpt":0.4972479358557301,"spread":0.4090742068165821,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002158554,0.0001006447,0.0001609758,0.0005695026,0.0001625959,0.0003230543,0.0004285853,0.00008128827,0.0001928821],"category_scores_gemma":[0.000653055,0.0001051719,0.0001086955,0.000389091,0.00004130116,0.001060117,0.00002348468,0.0001596002,0.00001493514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009298055,"about_ca_system_score_gemma":0.003643859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001984725,"about_ca_topic_score_gemma":0.0002967454,"domain_scores_codex":[0.9980417,0.0001447441,0.0006226869,0.0001520298,0.000829571,0.000209198],"domain_scores_gemma":[0.9976373,0.0006620656,0.0005375269,0.00008477119,0.0009941524,0.000084147],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00005804947,0.001401331,0.2410517,0.00005184151,0.0002689316,0.00001221194,0.02523632,0.00127119,0.004673972,0.1617295,0.09701511,0.4672298],"study_design_scores_gemma":[0.002073143,0.0002857038,0.6352672,0.0006652814,0.00008064404,0.00003102356,0.05633244,0.05547003,0.0003034543,0.01978774,0.229109,0.0005943266],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9672849,0.00004639044,0.001869205,0.01173054,0.01211992,0.0003714281,0.000009762377,0.00002946155,0.006538349],"genre_scores_gemma":[0.984994,0.0002382532,0.008437909,0.0002445746,0.004483025,0.00009518263,0.0001776082,0.00001352426,0.001315915],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4666355,"threshold_uncertainty_score":0.6464049,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}