{"meta":{"query_hash":"eaebb99799eb","filters":{"venue":"Educational Measurement Issues and Practice"},"cohort_total":38,"direct_labels_cover":0,"predictions_cover":38,"exported":38,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/eaebb99799eb","api":"https://metacan.xera.ac/api/v1/cohort?venue=Educational+Measurement+Issues+and+Practice"},"results":[{"id":"W1876976565","doi":"10.1111/j.1745-3992.2010.00198.x","title":"Reporting the Percentage of Students above a Cut Score: The Effect of Group Size","year":2011,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Mathematics education; Scale (ratio); Statistics; Psychology; Reading (process); Mathematics; Geography; Cartography; Political science","score_opus":0.3899556216284745,"score_gpt":0.5438855402313928,"score_spread":0.15392991860291827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1876976565","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.896157,0.004903854,0.07722037,0.0025676156,0.00085169,0.0019253639,0.0014560975,0.00075104024,0.014166996],"genre_scores_gemma":[0.97779137,0.0001695992,0.019132927,0.000475804,0.00008253072,0.00086963025,0.00043484892,0.00020089031,0.00084242213],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.50722843,0.37688854,0.034211084,0.027825925,0.051181443,0.002664583],"domain_scores_gemma":[0.084211424,0.7973049,0.046946935,0.050093524,0.020103931,0.0013393229],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3075692,0.0012232442,0.0019453272,0.004067761,0.0024099387,0.0030705053,0.003489143,0.0021046314,0.0017821981],"category_scores_gemma":[0.654336,0.0011103648,0.0030871578,0.0047990573,0.0053040176,0.0035528147,0.004268879,0.00250008,0.00060329895],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.010017557,0.00060517364,0.87799066,0.00074892957,0.0067188595,0.0002510549,0.011039487,0.0033055525,0.0027347158,0.0023644636,0.0048349868,0.0793885],"study_design_scores_gemma":[0.00031836712,0.004281472,0.96641994,0.00039375172,0.002109097,0.00043080677,0.0026204342,0.009466684,0.006431779,0.0022492134,0.0050903196,0.00018820762],"about_ca_topic_score_codex":0.010496132,"about_ca_topic_score_gemma":0.009130489,"teacher_disagreement_score":0.3075692,"about_ca_system_score_codex":0.002641438,"about_ca_system_score_gemma":0.0017077944,"threshold_uncertainty_score":0.8538904},"labels":[],"label_agreement":null},{"id":"W1985695119","doi":"10.1111/j.1745-3992.2004.tb00164.x","title":"Avoiding Misconception, Misuse, and Missed Opportunities: The Collection of Verbal Reports in Educational Achievement Testing","year":2004,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":121,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Cognition; Data collection; Trustworthiness; Nonverbal communication; Test (biology); Cognitive psychology; Developmental psychology; Applied psychology; Social psychology; Social science","score_opus":0.32269388672061894,"score_gpt":0.43327184522770973,"score_spread":0.11057795850709079,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1985695119","genre_codex":"commentary","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":"methods","prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10212946,0.07097492,0.294193,0.50397587,0.011429384,0.0010146467,0.00023385843,0.0010723246,0.014976599],"genre_scores_gemma":[0.60753155,0.026943296,0.20119101,0.14614353,0.011020377,0.0025003857,0.00016277871,0.00079737155,0.0037096415],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.2746556,0.602209,0.04299914,0.0061054532,0.07232972,0.0017010858],"domain_scores_gemma":[0.10678922,0.74237454,0.05484158,0.043321844,0.05040879,0.0022640007],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.509223,0.0014898246,0.0017667296,0.009384218,0.006096591,0.016445916,0.007318873,0.009632415,0.0005064935],"category_scores_gemma":[0.7473314,0.0024643277,0.0010475003,0.007213284,0.0706352,0.020039476,0.011015025,0.017273905,0.00094814505],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040789947,0.00020191178,0.032183316,0.0051351255,0.0003016947,0.0020155145,0.35690987,0.00080788904,0.0020439234,0.10007951,0.046698563,0.4532148],"study_design_scores_gemma":[0.00023241583,0.0014565475,0.038146492,0.057223916,0.000724687,0.014260673,0.23524502,0.009047845,0.015405633,0.308111,0.3188248,0.0013209796],"about_ca_topic_score_codex":0.005293658,"about_ca_topic_score_gemma":0.00732099,"teacher_disagreement_score":0.49077702,"about_ca_system_score_codex":0.006880463,"about_ca_system_score_gemma":0.012338069,"threshold_uncertainty_score":0.60521543},"labels":[],"label_agreement":null},{"id":"W2002384848","doi":"10.1111/j.1745-3992.2002.tb00103.x","title":"What Do School‐Level Scores From Large‐Scale Assessments Really Measure?","year":2002,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Public Health","funders":"","keywords":"Reading (process); Scale (ratio); Variance (accounting); Psychology; Cognition; Measure (data warehouse); Subject (documents); Illusion; Mathematics education; Cognitive psychology; Computer science; Linguistics; Data mining","score_opus":0.20664888336625692,"score_gpt":0.420967314211943,"score_spread":0.21431843084568608,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002384848","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8612076,0.010086661,0.041565124,0.019047821,0.0023643714,0.0005587204,0.003760676,0.00087395695,0.060535047],"genre_scores_gemma":[0.97981346,0.0020291558,0.012646136,0.0018646168,0.0007024847,0.00037338547,0.0012243129,0.00017306356,0.0011733872],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98395175,0.007113536,0.0019389018,0.0016937727,0.004821048,0.0004810106],"domain_scores_gemma":[0.8693184,0.0658543,0.02274621,0.013584694,0.025485411,0.0030108923],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.032688312,0.00092926546,0.0017780538,0.005404849,0.0006705684,0.0039714645,0.0019245864,0.0028541826,0.0015134319],"category_scores_gemma":[0.17000917,0.0004342166,0.0008685076,0.00557355,0.0034045728,0.0064821183,0.0015061271,0.0018263515,0.0016172172],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009423312,0.00018862635,0.8025778,0.0006706264,0.0006675349,0.000073439376,0.004396125,0.00064423244,0.00050506066,0.003374386,0.011129508,0.17567836],"study_design_scores_gemma":[0.00004715119,0.00045655482,0.9618918,0.0007461768,0.00031172243,0.00037252257,0.003961861,0.0020189984,0.0009979142,0.01644321,0.01265361,0.000098702636],"about_ca_topic_score_codex":0.004373445,"about_ca_topic_score_gemma":0.008270957,"teacher_disagreement_score":0.032688312,"about_ca_system_score_codex":0.0011348828,"about_ca_system_score_gemma":0.001299076,"threshold_uncertainty_score":0.17287433},"labels":[],"label_agreement":null},{"id":"W2006618588","doi":"10.1111/j.1745-3992.2000.tb00036.x","title":"An NCME Instructional Module on Exploring the Logic of Tatsuoka's Rule‐Space Model for Test Development and Analysis","year":2000,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Social Sciences and Humanities Research Council; Alberta Advanced Education; University of Alberta","funders":"","keywords":"Test (biology); Set (abstract data type); Space (punctuation); Blueprint; Computer science; Cognition; Artificial intelligence; Rule-based system; Machine learning; Psychology; Programming language; Engineering","score_opus":0.1492879822400197,"score_gpt":0.36563478447414144,"score_spread":0.21634680223412173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2006618588","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007197671,0.0041104835,0.6451235,0.0547293,0.011723276,0.0035380162,0.0014959952,0.005768808,0.26631296],"genre_scores_gemma":[0.02983825,0.009140808,0.5270944,0.02928029,0.010489052,0.00539524,0.0018525139,0.0015758746,0.38533354],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9989348,0.00041538686,0.00007606396,0.000118890755,0.00039155624,0.00006328975],"domain_scores_gemma":[0.9925776,0.004504138,0.00033244627,0.00050829194,0.0015609724,0.00051648053],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035135448,0.0012369762,0.00089811307,0.00243312,0.0009792079,0.001969221,0.0022639818,0.0023680574,0.06937638],"category_scores_gemma":[0.015877979,0.0006016945,0.0008627636,0.001239904,0.0012883567,0.003360235,0.0029671495,0.0031658467,0.024891574],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054747805,0.0010351405,0.0013958754,0.00034059124,0.000019410052,0.00042268643,0.00041956326,0.0028535908,0.0029969783,0.045962133,0.5493872,0.3951121],"study_design_scores_gemma":[0.00006840963,0.00024918516,0.005196165,0.0008781927,0.000014471873,0.0008054688,0.0003928946,0.009013141,0.0016169974,0.065823495,0.9158693,0.00007227068],"about_ca_topic_score_codex":0.0020957123,"about_ca_topic_score_gemma":0.008023508,"teacher_disagreement_score":0.06937638,"about_ca_system_score_codex":0.0017641381,"about_ca_system_score_gemma":0.0024865037,"threshold_uncertainty_score":0.23208714},"labels":[],"label_agreement":null},{"id":"W2018691396","doi":"10.1111/j.1745-3992.2010.00173.x","title":"Application of Think Aloud Protocols for Examining and Confirming Sources of Differential Item Functioning Identified by Expert Reviews","year":2010,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":96,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of New Brunswick; University of British Columbia","funders":"","keywords":"Differential item functioning; Think aloud protocol; Psychology; Empirical evidence; Cognitive psychology; Expert opinion; Applied psychology; Social psychology; Item response theory; Computer science; Developmental psychology; Psychometrics; Epistemology; Medicine; Human–computer interaction","score_opus":0.20234959071078357,"score_gpt":0.47284411788600883,"score_spread":0.27049452717522526,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2018691396","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.157169,0.0015582879,0.76707333,0.0014731261,0.0009516002,0.055628795,0.0011378275,0.002285451,0.012722663],"genre_scores_gemma":[0.08723245,0.00071401225,0.8424261,0.0004201975,0.00015580647,0.06655532,0.0003914263,0.0002678598,0.0018367526],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.6542087,0.25671312,0.042871185,0.011139799,0.03349718,0.0015700303],"domain_scores_gemma":[0.42794895,0.36509138,0.04525364,0.04969629,0.11023131,0.0017784424],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.21492966,0.0022662717,0.0018340059,0.009078635,0.0028958113,0.0029459933,0.0027803564,0.001528558,0.0023299286],"category_scores_gemma":[0.39165473,0.0014064601,0.0012852497,0.004398292,0.002474734,0.0028717222,0.004249005,0.0023936566,0.0020937787],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001991789,0.0010223424,0.011671003,0.007714304,0.00061411015,0.0009419167,0.11774579,0.0023270969,0.07361138,0.008690528,0.009687918,0.7639818],"study_design_scores_gemma":[0.002536988,0.013749873,0.07995953,0.010489034,0.0017185038,0.004535196,0.0980249,0.044386562,0.36387914,0.09725705,0.28076276,0.0027004732],"about_ca_topic_score_codex":0.0008753639,"about_ca_topic_score_gemma":0.002535016,"teacher_disagreement_score":0.21492966,"about_ca_system_score_codex":0.00187649,"about_ca_system_score_gemma":0.007874333,"threshold_uncertainty_score":0.9681315},"labels":[],"label_agreement":null},{"id":"W2034257998","doi":"10.1111/emip.12052","title":"What Role Does, and Should, the Test <i>Standards</i> Play Outside of the United States of America?","year":2014,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":19,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Citation; Library science; Test (biology); Sociology; Political science; History; Computer science","score_opus":0.33892563741599147,"score_gpt":0.4786025746879919,"score_spread":0.13967693727200042,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034257998","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.053321894,0.020515451,0.020353988,0.8009057,0.004389107,0.00008118586,0.0002476212,0.00016187843,0.10002319],"genre_scores_gemma":[0.8874976,0.012591274,0.02431007,0.06801231,0.0024847281,0.00020776593,0.00016813159,0.00018409798,0.0045440043],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9397777,0.041200712,0.002558421,0.0029571005,0.010716549,0.0027894403],"domain_scores_gemma":[0.74580896,0.16317564,0.02298673,0.0090086935,0.047977023,0.011042999],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.10820245,0.0005544831,0.0017104399,0.0030777857,0.004811467,0.023275945,0.0028549712,0.0054717786,0.003963138],"category_scores_gemma":[0.2443413,0.00050356187,0.000763663,0.004107104,0.020555612,0.028416444,0.0035868764,0.008169869,0.000970325],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013683426,0.00035406646,0.067851156,0.000713132,0.00018607487,0.00014868836,0.008935169,0.0005545637,0.00047544806,0.57114506,0.051346704,0.29815316],"study_design_scores_gemma":[0.00009303438,0.0003434945,0.09206208,0.01329591,0.00045711448,0.0008771653,0.08229807,0.0062132417,0.003236937,0.48939967,0.31132782,0.0003955287],"about_ca_topic_score_codex":0.046072666,"about_ca_topic_score_gemma":0.05189088,"teacher_disagreement_score":0.10820245,"about_ca_system_score_codex":0.009331195,"about_ca_system_score_gemma":0.028607257,"threshold_uncertainty_score":0.57223606},"labels":[],"label_agreement":null},{"id":"W2034878828","doi":"10.1111/j.1745-3992.2002.tb00087.x","title":"Scoring Examinee Responses for Multiple Inferences: Multiple Scoring in Assessments","year":2002,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Scoring system; Inference; Scale (ratio); Computer science; Psychology; Artificial intelligence; Medicine; Geography","score_opus":0.8659566286900213,"score_gpt":0.5782094015970195,"score_spread":0.2877472270930018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034878828","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0377364,0.003521816,0.9341734,0.004891808,0.0012544716,0.0033046894,0.00021424855,0.0010439998,0.013859197],"genre_scores_gemma":[0.21965076,0.0024321123,0.7662473,0.0022554807,0.00070006563,0.005131784,0.00025371485,0.00039843185,0.002930346],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.46100184,0.41843694,0.033506047,0.012605627,0.0722926,0.0021569969],"domain_scores_gemma":[0.3675581,0.4660519,0.045494307,0.052649572,0.06584612,0.0023999745],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.24177332,0.0018239979,0.0027152502,0.008021637,0.0035836673,0.0062776464,0.0044373027,0.0036818483,0.003011699],"category_scores_gemma":[0.6199554,0.001485733,0.0016697025,0.009639783,0.0050266054,0.008386918,0.009490683,0.005786226,0.0017098002],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00061824545,0.0005661629,0.05209879,0.003940347,0.0011126084,0.00073683675,0.024575384,0.0019129214,0.0045564203,0.056795377,0.02009414,0.8329927],"study_design_scores_gemma":[0.0008950499,0.002880175,0.208109,0.014931077,0.0023335693,0.009783708,0.022525486,0.068446,0.05486838,0.42120224,0.19204566,0.0019796956],"about_ca_topic_score_codex":0.0020234403,"about_ca_topic_score_gemma":0.0043668374,"teacher_disagreement_score":0.24177332,"about_ca_system_score_codex":0.0023441403,"about_ca_system_score_gemma":0.0052352743,"threshold_uncertainty_score":0.9350285},"labels":[],"label_agreement":null},{"id":"W2052069113","doi":"10.1111/emip.12003","title":"Validating Student Score Inferences With Person‐Fit Statistic and Verbal Reports: A Person‐Fit Study for Cognitive Diagnostic Assessment","year":2013,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Science Education and Pedagogy","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Statistic; Test (biology); Cognition; Consistency (knowledge bases); Psychology; Test statistic; Cognitive psychology; Artificial intelligence; Statistical hypothesis testing; Computer science; Statistics; Mathematics","score_opus":0.33467354375654246,"score_gpt":0.504415793140177,"score_spread":0.16974224938363458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052069113","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9732102,0.00003408569,0.02506634,0.000051324587,0.000020062995,0.00038299296,0.000052324995,0.000051806812,0.0011308491],"genre_scores_gemma":[0.9825916,0.000020755397,0.016499765,0.00004412394,0.00000960189,0.0005984916,0.0000712949,0.000020919979,0.00014335586],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.92390984,0.05379995,0.005892722,0.0027484929,0.01261903,0.0010299416],"domain_scores_gemma":[0.64100826,0.2684951,0.030500866,0.024870887,0.033170052,0.0019547977],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08252496,0.0007105224,0.00066519214,0.003607291,0.00093512115,0.0026933914,0.0011928992,0.001055906,0.0008172539],"category_scores_gemma":[0.29450274,0.00035560539,0.0012814178,0.0017690513,0.0016666189,0.002989163,0.0025084063,0.0014713521,0.0002624767],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00092941034,0.002313661,0.8463715,0.00019334907,0.00034115204,0.0002560475,0.042197324,0.002293357,0.0035711145,0.003513831,0.00048797362,0.09753134],"study_design_scores_gemma":[0.00040075235,0.012084866,0.7891465,0.00032649524,0.00036374162,0.0018289441,0.0551682,0.09181581,0.0325597,0.01002203,0.0058519207,0.00043108634],"about_ca_topic_score_codex":0.00077605335,"about_ca_topic_score_gemma":0.0009467518,"teacher_disagreement_score":0.08252496,"about_ca_system_score_codex":0.0011664418,"about_ca_system_score_gemma":0.0013865782,"threshold_uncertainty_score":0.43643892},"labels":[],"label_agreement":null},{"id":"W2055319746","doi":"10.1111/j.1745-3992.2004.tb00161.x","title":"Modeling Passing Rates on a Computer‐Based Medical Licensing Examination: An Application of Survival Data Analysis","year":2004,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Medical Education and Admissions","field":"Medicine","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Covariate; United States Medical Licensing Examination; Proportional hazards model; Survival analysis; Medical school; Variable (mathematics); Medicine; Medical education; Computer science; Psychology; Statistics; Surgery; Mathematics","score_opus":0.22741617092731478,"score_gpt":0.4633927414507003,"score_spread":0.23597657052338553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2055319746","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70602995,0.00047939975,0.28507373,0.001922567,0.00016151476,0.0010969808,0.0023695962,0.00068748824,0.0021786941],"genre_scores_gemma":[0.94370216,0.0003169373,0.049719654,0.00013917872,0.00007581282,0.0011457152,0.00129503,0.000072830364,0.003532695],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9889443,0.00837154,0.00049202744,0.0009090631,0.000742442,0.00054061576],"domain_scores_gemma":[0.9444001,0.045678467,0.004589197,0.0028975355,0.0018793889,0.000555339],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027185747,0.0010205602,0.0013710911,0.0030942513,0.0007219603,0.0013494638,0.0023410665,0.0013954386,0.0046369135],"category_scores_gemma":[0.06481484,0.0005277035,0.0033364678,0.0028188156,0.0010143796,0.0014221139,0.001695725,0.0022104932,0.0008208221],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015781727,0.0007841812,0.6381362,0.0003170576,0.0016488923,0.00047202827,0.0019202137,0.21563244,0.00052813586,0.018885322,0.0027959899,0.117301404],"study_design_scores_gemma":[0.00016229472,0.0012406916,0.08098049,0.00009483654,0.00046676619,0.00036085065,0.00071110786,0.9003795,0.00077330245,0.011896109,0.00284679,0.0000873312],"about_ca_topic_score_codex":0.022099048,"about_ca_topic_score_gemma":0.011186604,"teacher_disagreement_score":0.027185747,"about_ca_system_score_codex":0.0014363929,"about_ca_system_score_gemma":0.002685471,"threshold_uncertainty_score":0.14377373},"labels":[],"label_agreement":null},{"id":"W2057691077","doi":"10.1111/j.1745-3992.2009.01133.x","title":"Inclusive Achievement Testing for Linguistically and Culturally Diverse Test Takers: Essential Considerations for Test Developers and Decision Makers","year":2009,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Disability Education and Employment","field":"Social Sciences","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Carleton University","funders":"","keywords":"Ell; Accountability; Test (biology); Government (linguistics); Standardized test; No child left behind; Equity (law); Achievement test; Political science; Language assessment; Psychology; English language; Public relations; Pedagogy; Mathematics education; Teaching method","score_opus":0.125304747687676,"score_gpt":0.42539812488639406,"score_spread":0.3000933771987181,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2057691077","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15403928,0.010691897,0.07893905,0.69865537,0.0019617437,0.0015790144,0.00031581701,0.00074018486,0.053077746],"genre_scores_gemma":[0.74040264,0.006544151,0.19128352,0.052806225,0.001148641,0.0018337541,0.0002543662,0.0001854433,0.0055412934],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.90001684,0.049104955,0.012575465,0.002040456,0.03237109,0.0038911358],"domain_scores_gemma":[0.6223534,0.26265958,0.01590485,0.0103075355,0.055414755,0.033359922],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.13678075,0.0007513547,0.0015070969,0.0037395696,0.005422009,0.011827229,0.004550952,0.00537501,0.0017911189],"category_scores_gemma":[0.3013574,0.00077078113,0.0007344472,0.0019350216,0.008276172,0.0069038183,0.011902112,0.008514394,0.0007071162],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029060716,0.001453971,0.13573547,0.00071035314,0.00011186758,0.0029774425,0.06001225,0.001978099,0.0019282955,0.034767594,0.066896714,0.69313735],"study_design_scores_gemma":[0.00029388155,0.002278003,0.22013852,0.0111275455,0.00040130015,0.008861032,0.26511875,0.015199717,0.009707529,0.19629213,0.2698099,0.00077179325],"about_ca_topic_score_codex":0.027575169,"about_ca_topic_score_gemma":0.06652948,"teacher_disagreement_score":0.13678075,"about_ca_system_score_codex":0.004775884,"about_ca_system_score_gemma":0.037339773,"threshold_uncertainty_score":0.72337437},"labels":[],"label_agreement":null},{"id":"W2061554266","doi":"10.1111/j.1745-3992.2004.tb00165.x","title":"Assessing School Readiness: Validity and Bias in Preschool and Kindergarten Teachers' Ratings","year":2004,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Education and Early Childhood Development","funders":"","keywords":"Psychology; Head start; Vocabulary; Developmental psychology; Association (psychology); Early childhood education; Early childhood; Preschool education; Academic skills; Mathematics education","score_opus":0.18145666075735387,"score_gpt":0.4182170293389201,"score_spread":0.2367603685815662,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061554266","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9870833,0.0006533156,0.005169633,0.00013953916,0.000081213984,0.00026549137,0.00024866863,0.00008650907,0.0062723947],"genre_scores_gemma":[0.99619544,0.00019586273,0.002140437,0.00008709445,0.000023145374,0.00020525923,0.00039532318,0.000046123096,0.00071129785],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9745207,0.0093122125,0.00381864,0.0025186073,0.008847356,0.0009824472],"domain_scores_gemma":[0.8942074,0.05027697,0.018400388,0.010755409,0.024452347,0.0019074045],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03359967,0.00040466836,0.00077644875,0.0027188575,0.0007916318,0.0015483814,0.00090393244,0.000623143,0.0008959404],"category_scores_gemma":[0.11514715,0.00060108583,0.00067274127,0.0017006714,0.0016939105,0.0013695414,0.0018919134,0.00085453334,0.00051933853],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019417366,0.000062650404,0.97368217,0.00008850202,0.00014209976,0.00003874136,0.004839264,0.00017627371,0.001192358,0.0002035295,0.00030305886,0.019077126],"study_design_scores_gemma":[0.000026974698,0.00019429174,0.99232167,0.00010920451,0.000058890855,0.0001862474,0.0029715127,0.0010687821,0.0013407735,0.00037242533,0.0013251461,0.000024137109],"about_ca_topic_score_codex":0.011275134,"about_ca_topic_score_gemma":0.01428836,"teacher_disagreement_score":0.03359967,"about_ca_system_score_codex":0.00097380625,"about_ca_system_score_gemma":0.0011612041,"threshold_uncertainty_score":0.1776942},"labels":[],"label_agreement":null},{"id":"W2061917416","doi":"10.1111/emip.12015","title":"The Multiple‐Use of Accountability Assessments: Implications for the Process of Validation","year":2013,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Manitoba","funders":"","keywords":"Accountability; Process (computing); Argument (complex analysis); Quality (philosophy); Management science; Process management; Computer science; Best practice; Psychology; Political science; Medicine; Business; Engineering; Epistemology","score_opus":0.4021609414534796,"score_gpt":0.5386433098125486,"score_spread":0.13648236835906902,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2061917416","genre_codex":"methods","genre_gemma":"empirical","domain_codex":"methods","domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06633445,0.007421612,0.57096046,0.2760496,0.0014087249,0.003918594,0.00009930593,0.00040155204,0.07340566],"genre_scores_gemma":[0.7337202,0.0011239542,0.25199687,0.006191339,0.00029883021,0.0037244556,0.00003307662,0.00014144172,0.0027699205],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.14510398,0.7647716,0.023175031,0.011550083,0.04986543,0.0055338503],"domain_scores_gemma":[0.060075797,0.84924513,0.02181414,0.032324973,0.032921284,0.0036187416],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.721591,0.0017519172,0.003008878,0.01031651,0.025498599,0.032467946,0.009584797,0.013945655,0.0029684461],"category_scores_gemma":[0.7751388,0.0024462882,0.0025744396,0.009981265,0.123463735,0.04980734,0.027844708,0.020417111,0.0007283381],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008476834,0.00016421414,0.011585039,0.00069222925,0.00007893084,0.00055744505,0.12454645,0.001250456,0.0003783683,0.79713196,0.002924558,0.060605604],"study_design_scores_gemma":[0.000117067226,0.00022517811,0.007423201,0.005104395,0.000046848,0.0007739118,0.069485195,0.007897135,0.0012365006,0.8553298,0.052017517,0.00034336676],"about_ca_topic_score_codex":0.023752661,"about_ca_topic_score_gemma":0.023912162,"teacher_disagreement_score":0.278409,"about_ca_system_score_codex":0.039878536,"about_ca_system_score_gemma":0.09643989,"threshold_uncertainty_score":0.34332794},"labels":[],"label_agreement":null},{"id":"W2064530600","doi":"10.1111/j.1745-3992.2009.01135.x","title":"An NCME Instructional Module on Using Differential Step Functioning to Refine the Analysis of DIF in Polytomous Items","year":2009,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Polytomous Rasch model; Differential item functioning; Item response theory; Psychology; Rasch model; Task (project management); Test (biology); Statistics; Differential (mechanical device); Psychometrics; Mathematics; Clinical psychology; Developmental psychology","score_opus":0.5327361304287398,"score_gpt":0.5248611500943309,"score_spread":0.007874980334408921,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2064530600","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025414241,0.0034680306,0.32939118,0.07664284,0.013777438,0.0062043937,0.0034652075,0.008415297,0.5332215],"genre_scores_gemma":[0.05541914,0.006143339,0.39096016,0.032342743,0.0038763094,0.005339121,0.0028074377,0.0013821222,0.5017296],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99898165,0.00031099565,0.00007903302,0.00009136874,0.00047448085,0.00006247413],"domain_scores_gemma":[0.9933363,0.0030151904,0.0002437275,0.00042228523,0.0023961957,0.0005862679],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029794446,0.00080052816,0.00051489746,0.0017653166,0.0008079801,0.0011791481,0.0016142996,0.0015796535,0.07287819],"category_scores_gemma":[0.012818728,0.00029508988,0.0004678074,0.0010749113,0.00060602254,0.0022174362,0.0021791437,0.0020509134,0.024864968],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006798976,0.0011746244,0.0026968664,0.00048935215,0.000013343263,0.00034693928,0.0005107,0.0010929211,0.0044024102,0.010526389,0.5799014,0.39877695],"study_design_scores_gemma":[0.00006463368,0.00032315077,0.013218527,0.00074476807,0.000016770078,0.0007423854,0.00047217024,0.0046338774,0.0050702435,0.016461315,0.9581938,0.000058247995],"about_ca_topic_score_codex":0.002459819,"about_ca_topic_score_gemma":0.012289136,"teacher_disagreement_score":0.07287819,"about_ca_system_score_codex":0.0010672712,"about_ca_system_score_gemma":0.002345973,"threshold_uncertainty_score":0.24380183},"labels":[],"label_agreement":null},{"id":"W2109560181","doi":"10.1111/j.1745-3992.2005.00002.x","title":"Using Dimensionality‐Based DIF Analyses to Identify and Interpret Constructs That Elicit Group Differences","year":2005,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":59,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Matching (statistics); Psychology; Selection (genetic algorithm); Curse of dimensionality; Contrast (vision); Cognitive psychology; Test (biology); Social psychology; Computer science; Statistics; Artificial intelligence; Mathematics","score_opus":0.8614623269368624,"score_gpt":0.6322567921565106,"score_spread":0.22920553478035177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109560181","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1318298,0.000633041,0.84566146,0.0016719627,0.00021197446,0.0020090528,0.0009850083,0.00046030697,0.016537366],"genre_scores_gemma":[0.44201577,0.00056406227,0.5514025,0.00050440873,0.00007143705,0.003917148,0.00082706776,0.000089913294,0.0006077797],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97055805,0.021901838,0.0019701764,0.0016160553,0.0034399503,0.0005139546],"domain_scores_gemma":[0.8828627,0.09527585,0.0074712583,0.007210835,0.006579368,0.00059998245],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.036832623,0.0017280269,0.0011564179,0.009132141,0.001509923,0.0035391035,0.0008308985,0.0007620077,0.0030967102],"category_scores_gemma":[0.1251279,0.00035332,0.0012633692,0.0060453126,0.002220669,0.003797907,0.0040673767,0.0019803604,0.00068392226],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004098389,0.00057531864,0.12290894,0.0018426233,0.00084639055,0.00039675998,0.03623871,0.0054057506,0.009955879,0.18092027,0.0070869983,0.6334125],"study_design_scores_gemma":[0.0002763071,0.0010204442,0.17208126,0.001325104,0.0005699541,0.0013395001,0.033932894,0.06313404,0.014430869,0.6734259,0.03787801,0.0005856522],"about_ca_topic_score_codex":0.00090443296,"about_ca_topic_score_gemma":0.001251526,"teacher_disagreement_score":0.036832623,"about_ca_system_score_codex":0.0017001955,"about_ca_system_score_gemma":0.001358172,"threshold_uncertainty_score":0.1947918},"labels":[],"label_agreement":null},{"id":"W2111648217","doi":"10.1111/j.1745-3992.2010.00181.x","title":"Developing Score Reports for Cognitive Diagnostic Assessments","year":2010,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Context (archaeology); Test (biology); Cognition; Diagnostic test; Hierarchy; Sample (material); Knowledge management; Data science; Psychology; Medicine","score_opus":0.16578009314480702,"score_gpt":0.4552123800807414,"score_spread":0.28943228693593437,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111648217","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0093471445,0.0005963463,0.96610326,0.0011672686,0.000483208,0.0025882148,0.003194943,0.00552154,0.010998072],"genre_scores_gemma":[0.03763359,0.00046682823,0.95272267,0.00017754849,0.00018816274,0.0033871138,0.0035068616,0.00052646827,0.0013908643],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9008323,0.04256547,0.021962048,0.003385955,0.03004692,0.0012072437],"domain_scores_gemma":[0.66289043,0.14653063,0.042054515,0.032202173,0.11389152,0.0024307044],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.088678226,0.0025300172,0.0012940599,0.022055928,0.0016680356,0.0071118497,0.0038599924,0.0013581461,0.0050469297],"category_scores_gemma":[0.305648,0.0008466556,0.0017971017,0.008175002,0.0017542295,0.007168976,0.006005995,0.0030223757,0.0043460056],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024174055,0.00025506437,0.021729223,0.0015805666,0.00020101634,0.00032945842,0.0057943393,0.0061383112,0.004438039,0.09379523,0.04053209,0.82496494],"study_design_scores_gemma":[0.00030374562,0.0018644696,0.031607546,0.0055067753,0.0005169026,0.0026578512,0.011699458,0.067750484,0.056895334,0.2650497,0.5551157,0.0010319991],"about_ca_topic_score_codex":0.0023910734,"about_ca_topic_score_gemma":0.0022147633,"teacher_disagreement_score":0.088678226,"about_ca_system_score_codex":0.0025692594,"about_ca_system_score_gemma":0.005630372,"threshold_uncertainty_score":0.4689809},"labels":[],"label_agreement":null},{"id":"W2119322254","doi":"10.1111/j.1745-3992.2001.tb00060.x","title":"Illustrating the Utility of Differential Bundle Functioning Analyses to Identify and Interpret Group Differences on Achievement Tests","year":2001,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":69,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Bundle; Differential (mechanical device); Interpretation (philosophy); Psychology; Computer science; Engineering","score_opus":0.30470277556456604,"score_gpt":0.5091262291350933,"score_spread":0.20442345357052727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2119322254","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.22853869,0.00041579906,0.7554831,0.0027741636,0.00011222169,0.00035423719,0.00049919565,0.00096523156,0.010857334],"genre_scores_gemma":[0.6833574,0.00009672875,0.31460768,0.00036897126,0.00004414696,0.00048157398,0.00020125252,0.00020092593,0.0006412465],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.93619585,0.056028333,0.0017322263,0.0015905934,0.003525241,0.0009278486],"domain_scores_gemma":[0.7806054,0.19314954,0.004073358,0.012639397,0.008916161,0.0006161845],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05314831,0.0015650406,0.0009288082,0.0052740844,0.0014616151,0.0034262077,0.0014044879,0.0018813413,0.0028285095],"category_scores_gemma":[0.15934637,0.0007971666,0.0021246648,0.0047413525,0.0036462883,0.004466166,0.0044736117,0.0027831024,0.00064664404],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017129707,0.0006129591,0.2631093,0.0006447136,0.0012484347,0.0011364309,0.029964091,0.017541882,0.011606988,0.18954913,0.007067406,0.47580573],"study_design_scores_gemma":[0.00031819352,0.0012122302,0.2632987,0.0003051484,0.00056274055,0.0020351512,0.007897206,0.18404229,0.009613662,0.51907325,0.011353333,0.0002880679],"about_ca_topic_score_codex":0.0027960541,"about_ca_topic_score_gemma":0.0029899508,"teacher_disagreement_score":0.9468517,"about_ca_system_score_codex":0.00085349166,"about_ca_system_score_gemma":0.0011880621,"threshold_uncertainty_score":0.28107852},"labels":[],"label_agreement":null},{"id":"W2139397945","doi":"10.1111/j.1745-3992.2007.00090.x","title":"Defining and Evaluating Models of Cognition Used in Educational Measurement to Make Inferences About Examinees' Thinking Processes","year":2007,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":166,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Cognition; Identification (biology); Psychology; Cognitive psychology; Computer science; Management science","score_opus":0.27323320056062017,"score_gpt":0.4452723569892246,"score_spread":0.17203915642860446,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2139397945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21980107,0.0074920077,0.7244892,0.0068009356,0.00035597602,0.0017133616,0.0004846891,0.0007429126,0.038119785],"genre_scores_gemma":[0.6899065,0.0015617443,0.30346763,0.0005606907,0.000080776,0.0034726488,0.00037290042,0.000085699234,0.00049143285],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.84854823,0.10497342,0.012679234,0.004977285,0.026660046,0.002161763],"domain_scores_gemma":[0.63619155,0.2805467,0.029679205,0.023885272,0.027798334,0.0018988722],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.10115656,0.002005675,0.0016750938,0.018495103,0.0021711178,0.011610904,0.0038190265,0.0033545725,0.000994357],"category_scores_gemma":[0.3076984,0.00084756536,0.0031924557,0.010892763,0.010911144,0.013870235,0.007644782,0.0033284652,0.00032977463],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00046017536,0.0005141106,0.1483617,0.0028249251,0.0007726769,0.00024241768,0.037091084,0.017139686,0.0018233192,0.41936368,0.003351011,0.3680552],"study_design_scores_gemma":[0.0002161946,0.0011984672,0.12872697,0.005336896,0.0007576543,0.00090994465,0.036023665,0.09918209,0.006669031,0.69500065,0.025298946,0.0006794511],"about_ca_topic_score_codex":0.006243146,"about_ca_topic_score_gemma":0.0062749013,"teacher_disagreement_score":0.10115656,"about_ca_system_score_codex":0.010582954,"about_ca_system_score_gemma":0.007580674,"threshold_uncertainty_score":0.53497344},"labels":[],"label_agreement":null},{"id":"W2170853093","doi":"10.1111/j.1745-3992.2003.tb00136.x","title":"Using Multidimensional Item Response Theory to Evaluate Educational and Psychological Tests","year":2003,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":192,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Item response theory; Test (biology); Measure (data warehouse); Ninth; Computer science; Process (computing); Meaning (existential); Multidimensional analysis; Psychology; Mathematics education; Psychometrics; Management science; Econometrics; Mathematics; Data mining; Developmental psychology; Psychotherapist","score_opus":0.8128600435609681,"score_gpt":0.6116511900217015,"score_spread":0.20120885353926654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2170853093","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23461099,0.002348512,0.7009992,0.0031979475,0.00045696102,0.004598121,0.0017242043,0.0012877635,0.05077626],"genre_scores_gemma":[0.48268554,0.0011586351,0.50617105,0.0005711629,0.0001095543,0.0052661547,0.001834797,0.00013508412,0.0020680125],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9479811,0.036452495,0.0038253998,0.0013803805,0.009746374,0.00061419146],"domain_scores_gemma":[0.8472014,0.11999087,0.008324559,0.0069608074,0.01671018,0.0008121657],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043396488,0.0010338633,0.0012264409,0.009163097,0.0006829954,0.0046707676,0.0014713921,0.0014522598,0.0034206489],"category_scores_gemma":[0.17624004,0.00033707762,0.0015708624,0.0073220646,0.0016104564,0.004032248,0.0024289987,0.0019249382,0.0013516273],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002041943,0.0011391921,0.18032166,0.0010751887,0.0007259836,0.0001570436,0.0048143924,0.030452687,0.0017890774,0.08034903,0.008778763,0.6901928],"study_design_scores_gemma":[0.00035619806,0.004224301,0.32091403,0.0028572446,0.00043187093,0.0011214706,0.01565909,0.2767255,0.008473374,0.299978,0.06858903,0.0006698622],"about_ca_topic_score_codex":0.0013050956,"about_ca_topic_score_gemma":0.0016072822,"teacher_disagreement_score":0.043396488,"about_ca_system_score_codex":0.0021301382,"about_ca_system_score_gemma":0.0017659898,"threshold_uncertainty_score":0.2295053},"labels":[],"label_agreement":null},{"id":"W2172179933","doi":"10.1111/emip.12018","title":"Instructional Topics in Educational Measurement (ITEMS) Module: Using Automated Processes to Generate Test Items","year":2013,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Rendering (computer graphics); Test (biology); Item bank; Process (computing); Item response theory; Task (project management); Item analysis; Artificial intelligence; Machine learning; Information retrieval; Psychometrics; Programming language; Psychology; Engineering","score_opus":0.09722074847346185,"score_gpt":0.36063916936176055,"score_spread":0.2634184208882987,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2172179933","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06924645,0.00028484507,0.79597163,0.0022822109,0.00076282676,0.014810417,0.008618278,0.05544643,0.052576885],"genre_scores_gemma":[0.05736214,0.00018363986,0.9038378,0.0005819811,0.00017055652,0.0074919485,0.0038146202,0.0018736606,0.024683693],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99753726,0.0011980694,0.00023670241,0.00025183964,0.000657219,0.00011880427],"domain_scores_gemma":[0.98693335,0.006942446,0.00056860683,0.0021802252,0.0027028045,0.0006725545],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005031583,0.0010705206,0.0005474714,0.0018560084,0.00049770594,0.0015123081,0.0013355522,0.0011946589,0.021585006],"category_scores_gemma":[0.01919198,0.00085452275,0.0007501835,0.00094153656,0.00047315628,0.0014724972,0.0018789915,0.00142514,0.017502118],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038173236,0.0018023052,0.009491461,0.0004965241,0.0000769431,0.00017530407,0.0012183773,0.0025513805,0.012672866,0.007427388,0.099663034,0.86404264],"study_design_scores_gemma":[0.0015619729,0.0057087913,0.11837285,0.0007482915,0.00028126518,0.0018836436,0.00076236,0.09066774,0.14447512,0.044675916,0.5904215,0.0004405856],"about_ca_topic_score_codex":0.0010916252,"about_ca_topic_score_gemma":0.0014015656,"teacher_disagreement_score":0.021585006,"about_ca_system_score_codex":0.00056489516,"about_ca_system_score_gemma":0.0015440116,"threshold_uncertainty_score":0.07220906},"labels":[],"label_agreement":null},{"id":"W2279705546","doi":"10.1111/emip.12103","title":"The Role of Socioeconomic Status in SAT–Freshman Grade Relationships Across Gender and Racial Subgroups","year":2016,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"School Choice and Performance","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Socioeconomic status; Ethnic group; Race (biology); Demography; Psychology; Test (biology); Predictive power; Academic achievement; Developmental psychology; Sociology; Gender studies; Population","score_opus":0.1052190430503295,"score_gpt":0.3906782734469965,"score_spread":0.285459230396667,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2279705546","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99803907,0.00019427603,0.00008957341,0.00014458697,0.000008354691,0.0000036612698,0.00012721651,0.0000034211773,0.0013898436],"genre_scores_gemma":[0.999485,0.000040440027,0.000026323087,0.000020062631,0.000004995103,0.0000018197968,0.00013160928,0.0000021724709,0.00028745097],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99930894,0.00022893217,0.000046302735,0.00013690465,0.00013306027,0.00014579373],"domain_scores_gemma":[0.996714,0.0010240034,0.0008876669,0.0002394634,0.00035685423,0.0007780929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001583357,0.00022141956,0.00029729446,0.0011716193,0.00065283926,0.001006812,0.00045210746,0.00035222902,0.0036943199],"category_scores_gemma":[0.0067818877,0.00015166945,0.0004935638,0.00092148763,0.00049555366,0.00069948885,0.0009861281,0.00045456804,0.0003938623],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004629196,0.000020772386,0.9972154,0.0000027827089,0.000051031468,0.000025830679,0.00030064594,0.000019785131,0.000096479074,0.00011931052,0.00006600279,0.0020357512],"study_design_scores_gemma":[8.9077633e-7,0.000033167657,0.9992441,0.000003299699,0.000013478206,0.000016063814,0.00038400653,0.00008881909,0.000039490664,0.000073768555,0.00010139222,0.0000015585009],"about_ca_topic_score_codex":0.012124201,"about_ca_topic_score_gemma":0.020411514,"teacher_disagreement_score":0.012124201,"about_ca_system_score_codex":0.00035561403,"about_ca_system_score_gemma":0.00041814503,"threshold_uncertainty_score":0.024107277},"labels":[],"label_agreement":null},{"id":"W2562551526","doi":"10.1111/emip.12129","title":"A Process for Reviewing and Evaluating Generated Test Items","year":2016,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test (biology); Process (computing); Subject-matter expert; Quality (philosophy); Domain (mathematical analysis); Computerized adaptive testing; Item bank; Item response theory; Data science; Artificial intelligence; Psychometrics; Expert system; Mathematics; Statistics; Programming language","score_opus":0.2464384762232207,"score_gpt":0.4293302723194807,"score_spread":0.18289179609625997,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2562551526","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.014468718,0.0005935131,0.9418218,0.0016754142,0.00048503693,0.023365762,0.0007035795,0.010127468,0.006758645],"genre_scores_gemma":[0.017040437,0.00023852353,0.9703625,0.00026755207,0.00015907787,0.0062428983,0.0006736696,0.0009776956,0.0040376373],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.85789984,0.086290255,0.012086424,0.009734916,0.03284383,0.0011447568],"domain_scores_gemma":[0.53681904,0.18774056,0.021549772,0.07387221,0.17652537,0.003492941],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.13076076,0.0030864081,0.0024193157,0.012725278,0.004675223,0.006173004,0.005034926,0.0026919907,0.010111737],"category_scores_gemma":[0.33770424,0.001743782,0.0021656991,0.0051986496,0.0027710814,0.004597738,0.005206759,0.004426127,0.012909974],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004582713,0.00078045035,0.0036608167,0.0013948225,0.00017938035,0.00063400745,0.014866772,0.0025459263,0.03171945,0.009786946,0.04917552,0.88479775],"study_design_scores_gemma":[0.00086497737,0.003304507,0.031697243,0.0045397277,0.000723179,0.00466403,0.015433859,0.07513206,0.12803343,0.061045226,0.67240614,0.0021557321],"about_ca_topic_score_codex":0.0054712617,"about_ca_topic_score_gemma":0.0076695825,"teacher_disagreement_score":0.13076076,"about_ca_system_score_codex":0.0036435835,"about_ca_system_score_gemma":0.016197309,"threshold_uncertainty_score":0.69153726},"labels":[],"label_agreement":null},{"id":"W2615786270","doi":"10.1111/emip.12150","title":"Differential Prediction in the Use of the SAT and High School Grades in Predicting College Performance: Joint Effects of Race and Language","year":2017,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Higher Education Research Studies","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"University of Minnesota","keywords":"Ethnic group; Language proficiency; Psychology; Race (biology); Standardized test; Affect (linguistics); First language; Language assessment; Mathematics education; Linguistics; Sociology; Gender studies","score_opus":0.09120021442940497,"score_gpt":0.3903774829567992,"score_spread":0.29917726852739424,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2615786270","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9982685,0.00010092691,0.00033744177,0.00010796849,0.000010486649,0.000008567546,0.000052371877,0.000007995858,0.0011056795],"genre_scores_gemma":[0.99946386,0.000035735568,0.00012920582,0.000020559723,0.000006851355,0.000003578029,0.00007328799,0.0000032406654,0.00026357887],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9965185,0.0023052339,0.00017692665,0.0003019686,0.00042955164,0.00026780122],"domain_scores_gemma":[0.97920084,0.0121873375,0.0029696624,0.0015273092,0.0013723248,0.002742662],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007017468,0.00045898947,0.00028551114,0.0011982813,0.0005097538,0.0017488827,0.0006282952,0.00053117843,0.0019868435],"category_scores_gemma":[0.027330242,0.00021930339,0.00058107695,0.000866248,0.00079696,0.00105656,0.0014895465,0.0011399504,0.00062007795],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005990085,0.000050560528,0.9972529,0.0000021760902,0.00003849334,0.000008724865,0.00010234695,0.00006649874,0.000056736568,0.00004657774,0.000044719407,0.0022704368],"study_design_scores_gemma":[0.000004779656,0.00013678816,0.99733007,0.000013589924,0.000036721853,0.000030919153,0.0004564474,0.0015135577,0.00017788129,0.00017207599,0.00012207218,0.00000513246],"about_ca_topic_score_codex":0.0088949995,"about_ca_topic_score_gemma":0.019807484,"teacher_disagreement_score":0.0088949995,"about_ca_system_score_codex":0.00033848954,"about_ca_system_score_gemma":0.0007312242,"threshold_uncertainty_score":0.037112355},"labels":[],"label_agreement":null},{"id":"W2795852592","doi":"10.1111/emip.12198","title":"A Review of Recent Research on Individual‐Level Score Reports","year":2018,"lang":"en","type":"review","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Context (archaeology); Test (biology); Computer science; Focus (optics); Knowledge management; Psychology; Data science","score_opus":0.8619976910036018,"score_gpt":0.6550214592649924,"score_spread":0.20697623173860935,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2795852592","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00027739722,0.99807703,0.00026906317,0.00038991004,0.00011821074,0.000015189383,0.00006126344,0.000009147149,0.0007827061],"genre_scores_gemma":[0.0028540143,0.99595755,0.00057766924,0.00025675143,0.00012063676,0.000032803902,0.00008525572,0.0000075992875,0.000107681066],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9910507,0.0031361291,0.0022439829,0.0009263424,0.0024777951,0.00016498734],"domain_scores_gemma":[0.91936296,0.06690966,0.005721725,0.0012230041,0.0063613867,0.00042121552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01467553,0.0010155535,0.0025997297,0.012818945,0.00058172725,0.003266291,0.0024418414,0.0017893969,0.0056803836],"category_scores_gemma":[0.055087846,0.0008405049,0.0015066082,0.015692385,0.0020255223,0.0042202943,0.001579882,0.0017722186,0.0015635078],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000828938,0.000054096276,0.0012727742,0.13823059,0.00035373875,0.00011426881,0.0007474707,0.00017822505,0.00030815607,0.0036574395,0.011469959,0.8435304],"study_design_scores_gemma":[0.00003127918,0.00017469101,0.008966501,0.27382657,0.0013144317,0.001650974,0.0014505736,0.0001296428,0.0007675228,0.0034744625,0.7081256,0.00008771259],"about_ca_topic_score_codex":0.0026184355,"about_ca_topic_score_gemma":0.0041281125,"teacher_disagreement_score":0.01467553,"about_ca_system_score_codex":0.0016723609,"about_ca_system_score_gemma":0.0059465887,"threshold_uncertainty_score":0.07761252},"labels":[],"label_agreement":null},{"id":"W2804917593","doi":"10.1111/emip.12201","title":"Methodologies for Investigating and Interpreting Student–Teacher Rating Incongruence in Noncognitive Assessment","year":2018,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Education, Achievement, and Giftedness","field":"Psychology","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"American Psychological Association; American Educational Research Association","keywords":"Psychology; Construct (python library); Variety (cybernetics); Congruence (geometry); Predictive validity; Interpretation (philosophy); Divergence (linguistics); Construct validity; Descriptive statistics; Mathematics education; Social psychology; Psychometrics; Developmental psychology; Statistics","score_opus":0.2581577380829526,"score_gpt":0.5537707121933004,"score_spread":0.2956129741103478,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2804917593","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.29296893,0.00047741193,0.68923426,0.0003334179,0.00017976943,0.0046927426,0.00053046405,0.00046873727,0.01111436],"genre_scores_gemma":[0.6659042,0.0002137094,0.3199626,0.00015574403,0.000072798946,0.012210589,0.00042995883,0.00016149812,0.00088895403],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7995602,0.15543,0.014511809,0.007748523,0.021688826,0.0010606768],"domain_scores_gemma":[0.5859731,0.26917434,0.05867685,0.044214565,0.040310185,0.0016509483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.13676068,0.00089383096,0.0010089369,0.005968957,0.0015925037,0.003845818,0.0025756767,0.0010601963,0.0025162115],"category_scores_gemma":[0.36288655,0.0010694447,0.0011493148,0.0050966255,0.002848028,0.0025114145,0.004099731,0.0019051896,0.0006706492],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010401901,0.001119471,0.5121197,0.001965207,0.0010256366,0.0004044708,0.05737766,0.0053798556,0.016106421,0.029173875,0.003458981,0.37082848],"study_design_scores_gemma":[0.00031208017,0.0026906368,0.7540978,0.0019076407,0.00048724667,0.0010745588,0.034026235,0.084483,0.03061602,0.0657859,0.023902424,0.00061658345],"about_ca_topic_score_codex":0.0015368237,"about_ca_topic_score_gemma":0.0025646847,"teacher_disagreement_score":0.13676068,"about_ca_system_score_codex":0.0016274955,"about_ca_system_score_gemma":0.0018546353,"threshold_uncertainty_score":0.7232683},"labels":[],"label_agreement":null},{"id":"W2885169382","doi":"10.1111/emip.12211","title":"How Robust Are Cross‐Country Comparisons of PISA Scores to the Scaling Model Used?","year":2018,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Criticism; Robustness (evolution); Underpinning; Item response theory; Psychology; Test (biology); Cross country; Psychometrics; Political science; Developmental psychology; Economics; Demographic economics; Engineering","score_opus":0.13014986620014385,"score_gpt":0.37875505524570363,"score_spread":0.24860518904555978,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2885169382","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7514302,0.0045236656,0.16090381,0.01039313,0.004449642,0.0010045503,0.0061221947,0.00069208647,0.060480766],"genre_scores_gemma":[0.9857512,0.00024545836,0.009946746,0.0006397519,0.00021157676,0.00033239127,0.0018911434,0.0002320943,0.000749607],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.7958846,0.16351879,0.008506343,0.017147113,0.012100945,0.0028421523],"domain_scores_gemma":[0.46420744,0.38811797,0.041020744,0.07958438,0.024383128,0.00268637],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.17524736,0.001186957,0.0017516237,0.0035527537,0.0018178066,0.007863597,0.004104985,0.0020147574,0.008193031],"category_scores_gemma":[0.5214518,0.0007287199,0.0035883612,0.0064809793,0.0047396226,0.0058714366,0.005877248,0.0042416737,0.002977046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021362,0.0005403552,0.72384566,0.0013431452,0.021690158,0.00049515,0.009539147,0.021014843,0.0016953859,0.041359127,0.017770968,0.15856987],"study_design_scores_gemma":[0.00050922006,0.0025883853,0.78962165,0.0032604926,0.00595405,0.0007606614,0.029365215,0.036502633,0.0080881305,0.07724637,0.045563567,0.0005395883],"about_ca_topic_score_codex":0.0065134615,"about_ca_topic_score_gemma":0.0034319207,"teacher_disagreement_score":0.8247526,"about_ca_system_score_codex":0.0013299932,"about_ca_system_score_gemma":0.0014557311,"threshold_uncertainty_score":0.9268077},"labels":[],"label_agreement":null},{"id":"W3035205464","doi":"10.1111/emip.12353","title":"Exploring the Structure of Teachers’ Emotional Labor in the Classroom: A Multitrait–Multimethod Analysis","year":2020,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Emotional Labor in Professions","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada; Education University of Hong Kong","keywords":"Psychology; Emotional labor; Pride; Valence (chemistry); Social psychology; Anxiety; Negative emotion; School teachers; Emotional expression; Developmental psychology; Mathematics education","score_opus":0.28811363087923314,"score_gpt":0.4361760759650995,"score_spread":0.14806244508586636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3035205464","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.995659,0.000055518954,0.0034008967,0.00003838032,0.0000063231964,0.00011496337,0.000104173814,0.000014020426,0.00060678844],"genre_scores_gemma":[0.99714917,0.000023926874,0.0022336377,0.000016969687,0.000005520458,0.00018850912,0.00014637738,0.0000064036362,0.00022956078],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9920433,0.004269126,0.000611557,0.0009163999,0.0016753265,0.00048431687],"domain_scores_gemma":[0.97731996,0.014580824,0.0034180768,0.0027915332,0.0014112394,0.00047840853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010552227,0.0004190355,0.00074054406,0.002166563,0.0014436253,0.0020961282,0.0009654688,0.0005316353,0.002280953],"category_scores_gemma":[0.026061267,0.00045877052,0.0009081578,0.0016587457,0.0015003291,0.0009506906,0.0017964869,0.0009799527,0.00020309945],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019534145,0.00066929404,0.96106213,0.0000914308,0.00061767164,0.000060455695,0.011320141,0.00091050763,0.0019020981,0.0006977929,0.00031166454,0.022161365],"study_design_scores_gemma":[0.000017498329,0.00024655013,0.98529506,0.000027952383,0.00009871364,0.00004219843,0.0057513416,0.0068522776,0.0006405703,0.0004892698,0.0005133159,0.000025273544],"about_ca_topic_score_codex":0.023337545,"about_ca_topic_score_gemma":0.02455503,"teacher_disagreement_score":0.023337545,"about_ca_system_score_codex":0.0018008215,"about_ca_system_score_gemma":0.0016678149,"threshold_uncertainty_score":0.05580616},"labels":[],"label_agreement":null},{"id":"W3046538170","doi":"10.1111/emip.12382","title":"Synergy and Tension between Large‐Scale and Classroom Assessment: International Trends","year":2020,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University; Brock University","funders":"","keywords":"Scale (ratio); Test (biology); Political science; Mathematics education; Pedagogy; Sociology; Psychology; Geography; Geology; Cartography","score_opus":0.09632131387335943,"score_gpt":0.42067691106516314,"score_spread":0.32435559719180374,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3046538170","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6923557,0.08575698,0.012100718,0.10650138,0.0005356059,0.00008126989,0.0004144431,0.00017276227,0.102081224],"genre_scores_gemma":[0.9889381,0.007133181,0.0019861474,0.0010626396,0.00013192648,0.000022465767,0.0000621931,0.000026247964,0.00063706114],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9779327,0.009496402,0.0020225488,0.003315778,0.005945744,0.0012868799],"domain_scores_gemma":[0.79710925,0.12863964,0.018206926,0.008969309,0.04136628,0.005708543],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.046076596,0.00019608988,0.0005050337,0.004356804,0.00096165075,0.006938803,0.0012651858,0.00094193954,0.0025724433],"category_scores_gemma":[0.05747416,0.00031086826,0.00021280577,0.009207913,0.006573258,0.008068855,0.005446479,0.0026973805,0.00023732554],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021067033,0.00020170554,0.2032793,0.002084328,0.00011087031,0.00022036758,0.041262157,0.0013057038,0.0017646984,0.09733063,0.0064678746,0.64576167],"study_design_scores_gemma":[0.000023682322,0.0004931898,0.7198912,0.004646169,0.000078776975,0.0009358522,0.10282808,0.0022121358,0.0020133997,0.02183255,0.14489502,0.00014998252],"about_ca_topic_score_codex":0.014271814,"about_ca_topic_score_gemma":0.013667175,"teacher_disagreement_score":0.046076596,"about_ca_system_score_codex":0.004967979,"about_ca_system_score_gemma":0.007635783,"threshold_uncertainty_score":0.24367923},"labels":[],"label_agreement":null},{"id":"W4233643233","doi":"10.1111/emip.12321","title":"Digital Module 12: Think‐aloud Interviews and Cognitive Labs https://ncme.elevate.commpartners.com","year":2020,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Cognitive Science and Mapping","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Credibility; Think aloud protocol; Cognitive interview; Glossary; Interview; Cognition; Data collection; Psychology; Computer science; Protocol analysis; Comprehension; Reliability (semiconductor); Test (biology); Multimedia; Human–computer interaction; Cognitive science; Usability; Linguistics","score_opus":0.14978246681435134,"score_gpt":0.35360445438986376,"score_spread":0.20382198757551243,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4233643233","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04837463,0.00079572527,0.24894749,0.0067785913,0.0030669945,0.020919949,0.08242603,0.0817665,0.5069241],"genre_scores_gemma":[0.05459256,0.000855057,0.25292104,0.0026599292,0.0012015682,0.030647703,0.048832722,0.012636923,0.5956525],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.997419,0.00112435,0.00015119102,0.00026057503,0.0008374518,0.00020741594],"domain_scores_gemma":[0.9871149,0.0057660043,0.00044648678,0.0013413599,0.0042075585,0.0011237846],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004738505,0.0007436536,0.0005314983,0.0024733795,0.0006535278,0.0014361652,0.0011996393,0.00081232825,0.39417368],"category_scores_gemma":[0.01440992,0.0004581227,0.00038869958,0.0019159355,0.00034322753,0.0014494756,0.0020315768,0.00079379376,0.23205936],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016909152,0.0002810544,0.0005503812,0.00038546996,0.0000047721774,0.00006734956,0.00089968246,0.00018364095,0.0024134314,0.0011309972,0.6672739,0.32664037],"study_design_scores_gemma":[0.000104395105,0.0003000737,0.005643428,0.00028511864,0.000005181516,0.00018625657,0.00068587664,0.0009531276,0.0053700143,0.0023401722,0.9840949,0.000031324118],"about_ca_topic_score_codex":0.00060694193,"about_ca_topic_score_gemma":0.0013443045,"teacher_disagreement_score":0.39417368,"about_ca_system_score_codex":0.00086765765,"about_ca_system_score_gemma":0.0014526623,"threshold_uncertainty_score":0.8641377},"labels":[],"label_agreement":null},{"id":"W4237928641","doi":"10.1111/emip.12098","title":"On This Issue's Cover","year":2015,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Education Practices and Evaluation","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Scrolling; Test (biology); Depiction; Set (abstract data type); Psychology; Session (web analytics); Think aloud protocol; Multiple choice; Computer science; Applied psychology; Human–computer interaction; World Wide Web; Artificial intelligence; Visual arts; Usability; Linguistics","score_opus":0.2812239541239853,"score_gpt":0.4813383893869291,"score_spread":0.20011443526294376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4237928641","genre_codex":"other","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0005013495,0.0029586211,0.0009023096,0.0146333715,0.29418385,0.0004595345,0.003428284,0.004180555,0.6787521],"genre_scores_gemma":[0.0010845985,0.0013358345,0.00033153265,0.0059998496,0.039009593,0.00012155356,0.0018326407,0.0010710025,0.9492133],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99893016,0.00009282821,0.000058485133,0.00017572023,0.0005614083,0.0001814824],"domain_scores_gemma":[0.9954626,0.00062390586,0.00023343904,0.00047325742,0.0019447178,0.001262072],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00079600635,0.0014385825,0.0013783356,0.0017345763,0.0016763924,0.0066816094,0.002176344,0.003945583,0.83231974],"category_scores_gemma":[0.007495862,0.00055943814,0.0010816766,0.0013357494,0.0006597284,0.003906981,0.003356014,0.0039283237,0.6997379],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000009194225,0.000014762512,0.000023413457,0.000044556957,8.6476626e-7,0.00001655333,0.000003907657,0.0000105551135,0.00004535009,0.00023189338,0.9859541,0.013644991],"study_design_scores_gemma":[0.0000059259246,0.000010679357,0.00017022758,0.00006480882,0.0000010366113,0.000023267627,0.000016375085,0.000023114513,0.00003532639,0.0002013558,0.9994442,0.0000036486354],"about_ca_topic_score_codex":0.00126657,"about_ca_topic_score_gemma":0.0024562425,"teacher_disagreement_score":0.83231974,"about_ca_system_score_codex":0.0011185079,"about_ca_system_score_gemma":0.001445647,"threshold_uncertainty_score":0.2391755},"labels":[],"label_agreement":null},{"id":"W4246593371","doi":"10.1111/emip.12090","title":"Issue Information ‐ TOC &amp; Editorial board","year":2016,"lang":"en","type":"paratext","venue":"Educational Measurement Issues and Practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Editorial board; Computer science; Business; Environmental science; Library science","score_opus":0.16029175209505392,"score_gpt":0.46563718666507947,"score_spread":0.30534543457002555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4246593371","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00023629147,0.0013134925,0.00071883964,0.048008263,0.6745228,0.0008099844,0.0040315697,0.0018668066,0.2684919],"genre_scores_gemma":[0.0010916456,0.0011339644,0.0004884404,0.010667406,0.11474617,0.00033207942,0.0017950402,0.0010963875,0.8686488],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99434954,0.00067758217,0.00045623962,0.0004512422,0.0035817954,0.00048358203],"domain_scores_gemma":[0.9642393,0.0055117444,0.0017341207,0.0020263458,0.020501673,0.005986891],"candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0063879364,0.0023083242,0.0026180733,0.0048429775,0.0034793615,0.013702097,0.0031881537,0.0069365846,0.6071768],"category_scores_gemma":[0.037111208,0.0011184935,0.001323623,0.0029989502,0.0014569161,0.005595725,0.00272363,0.0054667983,0.6013224],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012246384,0.000009700797,0.000010974165,0.0000421254,8.3241025e-7,0.000009521917,0.0000027346962,0.0000062182035,0.00001856611,0.00011401161,0.9943968,0.0053762672],"study_design_scores_gemma":[0.000043693075,0.000018362316,0.00014809481,0.00012760752,0.0000031196132,0.00001401316,0.000025176056,0.000078450146,0.00010447911,0.00055097963,0.99887687,0.0000091305365],"about_ca_topic_score_codex":0.0021966887,"about_ca_topic_score_gemma":0.00473104,"teacher_disagreement_score":0.39282322,"about_ca_system_score_codex":0.0024077194,"about_ca_system_score_gemma":0.0055227005,"threshold_uncertainty_score":0.56031466},"labels":[],"label_agreement":null},{"id":"W4313410276","doi":"10.1111/emip.12537","title":"Using Active Learning Methods to Strategically Select Essays for Automated Scoring","year":2022,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Scalability; Active learning (machine learning); Encoder; Transformer; Database; Engineering","score_opus":0.33659020191225475,"score_gpt":0.5652165547572977,"score_spread":0.22862635284504296,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4313410276","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13467613,0.00013818103,0.8590714,0.00026536084,0.00008059505,0.00034860853,0.00011420997,0.002285823,0.0030196859],"genre_scores_gemma":[0.65617293,0.000055523105,0.33966437,0.00008356924,0.000047577014,0.000362164,0.0002587212,0.00017445853,0.003180752],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99457234,0.002951531,0.0003483021,0.00067407143,0.0012763041,0.00017740008],"domain_scores_gemma":[0.9594585,0.027509792,0.0026947903,0.0024205567,0.0071859625,0.0007304285],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009012679,0.0010242527,0.00082758575,0.0022183827,0.0005658511,0.002347926,0.0020284862,0.0009438706,0.002421166],"category_scores_gemma":[0.03578798,0.0003481179,0.00047528357,0.0010170073,0.00076114555,0.0023394055,0.0017253293,0.0014270779,0.0012893497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00057551643,0.0006788155,0.009924227,0.0001435319,0.000082828614,0.00007601165,0.000559862,0.07586161,0.014858535,0.0057124365,0.002231063,0.8892955],"study_design_scores_gemma":[0.00007734585,0.00026192304,0.0020118307,0.000023791243,0.000021851361,0.000050377785,0.00017450348,0.96861935,0.0202601,0.0068267733,0.001642187,0.000029937219],"about_ca_topic_score_codex":0.001038913,"about_ca_topic_score_gemma":0.0022400816,"teacher_disagreement_score":0.009012679,"about_ca_system_score_codex":0.00084227516,"about_ca_system_score_gemma":0.0011419094,"threshold_uncertainty_score":0.047664225},"labels":[],"label_agreement":null},{"id":"W4319790235","doi":"10.1111/emip.12539","title":"Machine Learning Literacy for Measurement Professionals: A Practical Tutorial","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Medical Council of Canada","funders":"","keywords":"Toolbox; Computer science; Python (programming language); Data science; Context (archaeology); Artificial intelligence","score_opus":0.19356892700389544,"score_gpt":0.4464950114435416,"score_spread":0.25292608443964615,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319790235","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013162168,0.059951063,0.4524196,0.21119657,0.022287296,0.0016048034,0.0040193805,0.022697086,0.21266204],"genre_scores_gemma":[0.07694711,0.07186833,0.34904808,0.06969782,0.014811375,0.0032304772,0.008267229,0.00670344,0.3994262],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9986584,0.0005536935,0.00010052811,0.0001800687,0.00032448518,0.00018275545],"domain_scores_gemma":[0.9959986,0.0024641172,0.00017895958,0.00013764124,0.0007318705,0.0004888736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002324205,0.0018616653,0.0006857439,0.0017290448,0.0012081951,0.0034908492,0.0013541882,0.003310825,0.06550618],"category_scores_gemma":[0.0094993925,0.00053440675,0.0008631657,0.0010878873,0.0009767095,0.0076322826,0.004278048,0.0039274767,0.030561445],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000036335965,0.00018168295,0.0002883685,0.00072717684,0.000010324312,0.00039389316,0.0018313751,0.0007453235,0.0016171242,0.03740408,0.67015,0.2866143],"study_design_scores_gemma":[0.000012527845,0.000058152567,0.00039584757,0.0012115269,0.000004984559,0.0006642115,0.0004905313,0.0010165586,0.0005079714,0.025843047,0.9697658,0.000028877275],"about_ca_topic_score_codex":0.0007332377,"about_ca_topic_score_gemma":0.0014983703,"teacher_disagreement_score":0.06550618,"about_ca_system_score_codex":0.0015992001,"about_ca_system_score_gemma":0.002065645,"threshold_uncertainty_score":0.21914005},"labels":[],"label_agreement":null},{"id":"W4362673096","doi":"10.1111/emip.12553","title":"Validation as Evaluating Desired and Undesired Effects: Insights From Cross‐Classified Mixed Effects Model","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reliability (semiconductor); Variance (accounting); Computer science; Reliability engineering; Variance components; Validity; External validity; Cross-validation; Random effects model; Statistics; Data mining; Econometrics; Psychology; Artificial intelligence; Psychometrics; Mathematics; Engineering; Medicine","score_opus":0.7119997234981603,"score_gpt":0.5622263498475483,"score_spread":0.14977337365061205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362673096","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04507495,0.0011237557,0.9484598,0.0012170022,0.00016263359,0.00084210327,0.0002005038,0.00021074689,0.0027084523],"genre_scores_gemma":[0.50173223,0.00044428415,0.49307534,0.0007310767,0.00012877125,0.00257916,0.00036338772,0.00015124104,0.0007944524],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.4743856,0.49189314,0.009345068,0.010034111,0.013318409,0.0010236725],"domain_scores_gemma":[0.105389714,0.85270506,0.01225304,0.01803736,0.010990795,0.0006240343],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.44658118,0.0024675159,0.004040416,0.0052897525,0.0023449857,0.0072389706,0.005535049,0.0045908685,0.0040248553],"category_scores_gemma":[0.61715347,0.0017451447,0.008743202,0.003791886,0.006440164,0.006994923,0.0058142734,0.0061491933,0.00041220413],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020864953,0.0009936129,0.11546238,0.0026912757,0.010011775,0.0011937629,0.0133450655,0.1259694,0.001318283,0.5263015,0.0035238785,0.19710253],"study_design_scores_gemma":[0.0002996612,0.0012380215,0.015249997,0.001679019,0.0024363634,0.0003821402,0.0014242694,0.71289104,0.0018334945,0.2571212,0.0052065495,0.00023823457],"about_ca_topic_score_codex":0.0061140675,"about_ca_topic_score_gemma":0.0043110573,"teacher_disagreement_score":0.44658118,"about_ca_system_score_codex":0.0038047365,"about_ca_system_score_gemma":0.0050107716,"threshold_uncertainty_score":0.682464},"labels":[],"label_agreement":null},{"id":"W4386482646","doi":"10.1111/emip.12572","title":"Digital Module 33: Fairness in Classroom Assessment: Dimensions and Tensions","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Legitimacy; Perception; Psychology; Disengagement theory; Critical reflection; Social psychology; Pedagogy; Political science","score_opus":0.11055476844730784,"score_gpt":0.42774349726989375,"score_spread":0.3171887288225859,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386482646","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.52514666,0.0029755565,0.047154915,0.039892375,0.0027403748,0.0031314332,0.004875981,0.0022830155,0.37179965],"genre_scores_gemma":[0.78727293,0.004544395,0.0323427,0.004432544,0.0013575036,0.002289751,0.0035202652,0.0004672747,0.16377254],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9985655,0.0005285444,0.00012916826,0.000089909576,0.000504419,0.00018257907],"domain_scores_gemma":[0.99504966,0.0019533108,0.0004819242,0.00036156227,0.0012153484,0.0009381611],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029966389,0.0002959309,0.00035538405,0.0009530309,0.000688381,0.0023664245,0.00058048824,0.00084137305,0.03363624],"category_scores_gemma":[0.008356256,0.00016350555,0.00028358932,0.0011993336,0.0007191504,0.0012692689,0.0020103308,0.0013361571,0.0051337117],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022898993,0.0015218077,0.027258843,0.0008043659,0.000019533243,0.00017276529,0.0060416367,0.0015829927,0.0045642545,0.017206961,0.24581572,0.6947821],"study_design_scores_gemma":[0.00005398588,0.00094114087,0.20306507,0.0015929337,0.000028359682,0.00093166815,0.0077161626,0.0044570626,0.008146683,0.024197143,0.74873686,0.00013304694],"about_ca_topic_score_codex":0.0009364573,"about_ca_topic_score_gemma":0.0019601888,"teacher_disagreement_score":0.03363624,"about_ca_system_score_codex":0.0013228455,"about_ca_system_score_gemma":0.0018882873,"threshold_uncertainty_score":0.11252445},"labels":[],"label_agreement":null},{"id":"W4388223267","doi":"10.1111/emip.12582","title":"Comparing Large‐Scale Assessments in Two Proctoring Modalities with Interactive Log Data Analysis","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Medical Council of Canada","funders":"","keywords":"Modalities; Comparability; Modality (human–computer interaction); Test (biology); Scale (ratio); Medicine; Computer science; Human–computer interaction; Mathematics","score_opus":0.5444063599822654,"score_gpt":0.5751607052320205,"score_spread":0.030754345249755066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388223267","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98864913,0.00010198316,0.00830463,0.00010846747,0.000017541524,0.00031478165,0.0010421317,0.00014830189,0.0013129079],"genre_scores_gemma":[0.9924785,0.00004076877,0.0062312447,0.000033200733,0.000013266504,0.00021297882,0.0006349151,0.00001361715,0.00034160315],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99336785,0.0035814273,0.00051280565,0.00090359175,0.0013441513,0.0002902535],"domain_scores_gemma":[0.9198603,0.0574148,0.009402795,0.0046485327,0.006930978,0.0017426296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068017794,0.00044769933,0.00041032894,0.0023229006,0.0004736032,0.0014581861,0.00082747574,0.0005962356,0.0019482371],"category_scores_gemma":[0.057875402,0.00017720921,0.0006006366,0.002267651,0.0007188527,0.0008613405,0.0014010926,0.0007935964,0.0004936264],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025742596,0.0022818283,0.8246668,0.00057065574,0.00073430064,0.00022999103,0.0041593504,0.014302514,0.0063176104,0.0008249817,0.0023629777,0.14097476],"study_design_scores_gemma":[0.00007520902,0.0012639976,0.9486916,0.00008317943,0.0001431554,0.00013819843,0.0023626452,0.03958053,0.0050226,0.0007554396,0.001811662,0.00007188016],"about_ca_topic_score_codex":0.020247055,"about_ca_topic_score_gemma":0.034983166,"teacher_disagreement_score":0.020247055,"about_ca_system_score_codex":0.0013621002,"about_ca_system_score_gemma":0.0015879347,"threshold_uncertainty_score":0.040258408},"labels":[],"label_agreement":null},{"id":"W4389347402","doi":"10.1111/emip.12585","title":"Digital Module 34: Introduction to Multilevel Measurement Modeling","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Mental Health Research Topics","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Structural equation modeling; Multilevel model; Computer science; Code (set theory); Latent variable; Process (computing); Data mining; Software engineering; Programming language; Artificial intelligence; Machine learning","score_opus":0.3432883525820852,"score_gpt":0.4905485480493123,"score_spread":0.14726019546722713,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389347402","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.001512139,0.0015467219,0.84278125,0.006359137,0.0010682599,0.0037421123,0.029018724,0.03627136,0.07770034],"genre_scores_gemma":[0.01735487,0.0034768912,0.86085206,0.0032427048,0.0014428323,0.011150764,0.021946192,0.021629063,0.05890455],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960145,0.0021336523,0.0003573996,0.0003815401,0.0009902486,0.0001226781],"domain_scores_gemma":[0.98411924,0.01036156,0.0005312773,0.0018307335,0.0027722614,0.0003848687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008192302,0.0010329132,0.0010498731,0.002366054,0.0005967552,0.0023707333,0.0027187916,0.0012658619,0.26481587],"category_scores_gemma":[0.036726847,0.0016469841,0.002291665,0.0032197821,0.00050154823,0.00273433,0.0024899277,0.0030143831,0.12557296],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054192424,0.00011849871,0.0011478307,0.0009207024,0.00007866925,0.0000934648,0.00028773653,0.0035896408,0.0008687341,0.049590666,0.64100295,0.30224687],"study_design_scores_gemma":[0.000091799,0.00011331069,0.0038038723,0.0011385174,0.000044745488,0.00024224247,0.00008117612,0.011484268,0.0015435087,0.07731537,0.9040373,0.000103974875],"about_ca_topic_score_codex":0.0022612691,"about_ca_topic_score_gemma":0.0024948402,"teacher_disagreement_score":0.26481587,"about_ca_system_score_codex":0.0013160463,"about_ca_system_score_gemma":0.0022142523,"threshold_uncertainty_score":0.88589734},"labels":[],"label_agreement":null},{"id":"W4391227797","doi":"10.1111/emip.12590","title":"Using OpenAI GPT to Generate Reading Comprehension Items","year":2024,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading comprehension; Comprehension; Reading (process); Sentence; Computer science; Cognition; Natural language processing; Artificial intelligence; Sample (material); Psychology; Cognitive psychology; Linguistics","score_opus":0.2292006550062576,"score_gpt":0.41553136314358546,"score_spread":0.18633070813732786,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391227797","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23993707,0.00013215606,0.70673996,0.00018279518,0.00025889536,0.008523776,0.004437896,0.026914535,0.012873065],"genre_scores_gemma":[0.2596103,0.00009997491,0.7200114,0.00007469645,0.000043610293,0.008080788,0.006155544,0.0020698656,0.003853862],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9941532,0.0028516012,0.0008039345,0.0008391111,0.0012425118,0.00010965614],"domain_scores_gemma":[0.9454595,0.03644957,0.0019213539,0.0063273935,0.009467513,0.00037468137],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0067087994,0.001260643,0.00068616297,0.0022504346,0.00037822948,0.0016065928,0.0016270935,0.0008019766,0.008932563],"category_scores_gemma":[0.057750527,0.00054971623,0.0008059987,0.0014592459,0.0005323419,0.001530407,0.0017521332,0.0012733733,0.004859292],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007058989,0.0009847138,0.01967767,0.00071022526,0.00011211319,0.0006234278,0.0045867334,0.010896951,0.028456945,0.0030039023,0.010579939,0.9196615],"study_design_scores_gemma":[0.001631092,0.0066235163,0.127087,0.0008522757,0.00043073215,0.0040535014,0.0035609005,0.41513795,0.32435766,0.021067718,0.094490536,0.00070705375],"about_ca_topic_score_codex":0.0008953472,"about_ca_topic_score_gemma":0.0008328997,"teacher_disagreement_score":0.008932563,"about_ca_system_score_codex":0.0006136245,"about_ca_system_score_gemma":0.00067812455,"threshold_uncertainty_score":0.035479963},"labels":[],"label_agreement":null},{"id":"W4405618562","doi":"10.1111/emip.12663","title":"Instruction‐Tuned Large‐Language Models for Quality Control in Automatic Item Generation: A Feasibility Study","year":2024,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Quality (philosophy); Computer science; Control (management); Item response theory; Language proficiency; Mathematics education; Natural language processing; Psychology; Artificial intelligence; Psychometrics; Developmental psychology","score_opus":0.19670056213224912,"score_gpt":0.4278503410420334,"score_spread":0.23114977890978428,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405618562","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85047245,0.00013059871,0.14168076,0.00048335103,0.00006182345,0.0020635654,0.00029381318,0.0034825443,0.0013309424],"genre_scores_gemma":[0.86363566,0.00002986427,0.13459471,0.00011443701,0.000015455598,0.0008314723,0.00023366376,0.0002096155,0.0003351135],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98491645,0.011454172,0.0009187471,0.0011373797,0.0012523144,0.00032097133],"domain_scores_gemma":[0.8163837,0.1545434,0.0035610131,0.012266138,0.011611646,0.0016340928],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.034051638,0.0011578486,0.00075509393,0.000753626,0.000533775,0.0018844286,0.003070298,0.0013260817,0.0027079298],"category_scores_gemma":[0.13227774,0.000945216,0.0005778416,0.0007102212,0.000917786,0.00323792,0.0016269346,0.0021139183,0.0008768544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.011470159,0.020636007,0.086368434,0.0011612705,0.00043527715,0.0009060684,0.007445504,0.17333893,0.062251724,0.0055118543,0.0082976995,0.6221771],"study_design_scores_gemma":[0.0014510036,0.0047098524,0.0165485,0.00009154419,0.00015297103,0.00019765382,0.00053760735,0.94508886,0.026387118,0.0022199687,0.002486016,0.00012889982],"about_ca_topic_score_codex":0.008011228,"about_ca_topic_score_gemma":0.005956651,"teacher_disagreement_score":0.034051638,"about_ca_system_score_codex":0.001603922,"about_ca_system_score_gemma":0.0018809662,"threshold_uncertainty_score":0.18008447},"labels":[],"label_agreement":null}]}