{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":38,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":38,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"eaebb99799eb","filters":{"venue":"Educational Measurement Issues and Practice"}},"results":[{"id":"W2170853093","doi":"10.1111/j.1745-3992.2003.tb00136.x","title":"Using Multidimensional Item Response Theory to Evaluate Educational and Psychological Tests","year":2003,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":192,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Item response theory; Test (biology); Measure (data warehouse); Ninth; Computer science; Process (computing); Meaning (existential); Multidimensional analysis; Psychology; Mathematics education; Psychometrics; Management science; Econometrics; Mathematics; Data mining; Developmental psychology; Psychotherapist","authors":[{"name":"Terry A. Ackerman","is_ca":false},{"name":"Mark J. Gierl","is_ca":true},{"name":"Cindy M. Walker","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8128600435609681,"gpt":0.6116511900217015,"spread":0.2012088535392665,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04339649,0.001033863,0.001226441,0.009163097,0.0006829954,0.004670768,0.001471392,0.00145226,0.003420649],"category_scores_gemma":[0.17624,0.0003370776,0.001570862,0.007322065,0.001610456,0.004032248,0.002428999,0.001924938,0.001351627],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002130138,"about_ca_system_score_gemma":0.00176599,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001305096,"about_ca_topic_score_gemma":0.001607282,"domain_scores_codex":[0.9479811,0.03645249,0.0038254,0.00138038,0.009746374,0.0006141915],"domain_scores_gemma":[0.8472014,0.1199909,0.008324559,0.006960807,0.01671018,0.0008121657],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002041943,0.001139192,0.1803217,0.001075189,0.0007259836,0.0001570436,0.004814392,0.03045269,0.001789077,0.08034903,0.008778763,0.6901928],"study_design_scores_gemma":[0.0003561981,0.004224301,0.320914,0.002857245,0.0004318709,0.001121471,0.01565909,0.2767255,0.008473374,0.299978,0.06858903,0.0006698622],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.234611,0.002348512,0.7009992,0.003197948,0.000456961,0.004598121,0.001724204,0.001287763,0.05077626],"genre_scores_gemma":[0.4826855,0.001158635,0.506171,0.0005711629,0.0001095543,0.005266155,0.001834797,0.0001350841,0.002068012],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04339649,"threshold_uncertainty_score":0.2295053,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2139397945","doi":"10.1111/j.1745-3992.2007.00090.x","title":"Defining and Evaluating Models of Cognition Used in Educational Measurement to Make Inferences About Examinees' Thinking Processes","year":2007,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":166,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Cognition; Identification (biology); Psychology; Cognitive psychology; Computer science; Management science","authors":[{"name":"Jacqueline P. Leighton","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2732332005606202,"gpt":0.4452723569892246,"spread":0.1720391564286045,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1011566,0.002005675,0.001675094,0.0184951,0.002171118,0.0116109,0.003819027,0.003354572,0.000994357],"category_scores_gemma":[0.3076984,0.0008475654,0.003192456,0.01089276,0.01091114,0.01387023,0.007644782,0.003328465,0.0003297746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01058295,"about_ca_system_score_gemma":0.007580674,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006243146,"about_ca_topic_score_gemma":0.006274901,"domain_scores_codex":[0.8485482,0.1049734,0.01267923,0.004977285,0.02666005,0.002161763],"domain_scores_gemma":[0.6361915,0.2805467,0.02967921,0.02388527,0.02779833,0.001898872],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004601754,0.0005141106,0.1483617,0.002824925,0.0007726769,0.0002424177,0.03709108,0.01713969,0.001823319,0.4193637,0.003351011,0.3680552],"study_design_scores_gemma":[0.0002161946,0.001198467,0.128727,0.005336896,0.0007576543,0.0009099446,0.03602367,0.09918209,0.006669031,0.6950006,0.02529895,0.0006794511],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2198011,0.007492008,0.7244892,0.006800936,0.000355976,0.001713362,0.0004846891,0.0007429126,0.03811979],"genre_scores_gemma":[0.6899065,0.001561744,0.3034676,0.0005606907,0.000080776,0.003472649,0.0003729004,0.00008569923,0.0004914328],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1011566,"threshold_uncertainty_score":0.5349734,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1985695119","doi":"10.1111/j.1745-3992.2004.tb00164.x","title":"Avoiding Misconception, Misuse, and Missed Opportunities: The Collection of Verbal Reports in Educational Achievement Testing","year":2004,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":121,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Psychology; Cognition; Data collection; Trustworthiness; Nonverbal communication; Test (biology); Cognitive psychology; Developmental psychology; Applied psychology; Social psychology; Social science","authors":[{"name":"Jacqueline P. Leighton","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3226938867206189,"gpt":0.4332718452277097,"spread":0.1105779585070908,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.509223,0.001489825,0.00176673,0.009384218,0.006096591,0.01644592,0.007318873,0.009632415,0.0005064935],"category_scores_gemma":[0.7473314,0.002464328,0.0010475,0.007213284,0.0706352,0.02003948,0.01101502,0.0172739,0.0009481451],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006880463,"about_ca_system_score_gemma":0.01233807,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005293658,"about_ca_topic_score_gemma":0.00732099,"domain_scores_codex":[0.2746556,0.602209,0.04299914,0.006105453,0.07232972,0.001701086],"domain_scores_gemma":[0.1067892,0.7423745,0.05484158,0.04332184,0.05040879,0.002264001],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004078995,0.0002019118,0.03218332,0.005135125,0.0003016947,0.002015515,0.3569099,0.000807889,0.002043923,0.1000795,0.04669856,0.4532148],"study_design_scores_gemma":[0.0002324158,0.001456548,0.03814649,0.05722392,0.000724687,0.01426067,0.235245,0.009047845,0.01540563,0.308111,0.3188248,0.00132098],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"methods","genre_scores_codex":[0.1021295,0.07097492,0.294193,0.5039759,0.01142938,0.001014647,0.0002338584,0.001072325,0.0149766],"genre_scores_gemma":[0.6075315,0.0269433,0.201191,0.1461435,0.01102038,0.002500386,0.0001627787,0.0007973716,0.003709642],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.490777,"threshold_uncertainty_score":0.6052154,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2018691396","doi":"10.1111/j.1745-3992.2010.00173.x","title":"Application of Think Aloud Protocols for Examining and Confirming Sources of Differential Item Functioning Identified by Expert Reviews","year":2010,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":96,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of New Brunswick; University of British Columbia","funders":"","keywords":"Differential item functioning; Think aloud protocol; Psychology; Empirical evidence; Cognitive psychology; Expert opinion; Applied psychology; Social psychology; Item response theory; Computer science; Developmental psychology; Psychometrics; Epistemology; Medicine; Human–computer interaction","authors":[{"name":"Kadriye Ercikan","is_ca":true},{"name":"Rübab G. Arım","is_ca":true},{"name":"Danielle M. Law","is_ca":true},{"name":"José F. Domene","is_ca":true},{"name":"France Gagnon","is_ca":true},{"name":"Serge Lacroix","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2023495907107836,"gpt":0.4728441178860088,"spread":0.2704945271752253,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2149297,0.002266272,0.001834006,0.009078635,0.002895811,0.002945993,0.002780356,0.001528558,0.002329929],"category_scores_gemma":[0.3916547,0.00140646,0.00128525,0.004398292,0.002474734,0.002871722,0.004249005,0.002393657,0.002093779],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00187649,"about_ca_system_score_gemma":0.007874333,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008753639,"about_ca_topic_score_gemma":0.002535016,"domain_scores_codex":[0.6542087,0.2567131,0.04287118,0.0111398,0.03349718,0.00157003],"domain_scores_gemma":[0.427949,0.3650914,0.04525364,0.04969629,0.1102313,0.001778442],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.001991789,0.001022342,0.011671,0.007714304,0.0006141101,0.0009419167,0.1177458,0.002327097,0.07361138,0.008690528,0.009687918,0.7639818],"study_design_scores_gemma":[0.002536988,0.01374987,0.07995953,0.01048903,0.001718504,0.004535196,0.0980249,0.04438656,0.3638791,0.09725705,0.2807628,0.002700473],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.157169,0.001558288,0.7670733,0.001473126,0.0009516002,0.0556288,0.001137828,0.002285451,0.01272266],"genre_scores_gemma":[0.08723245,0.0007140122,0.8424261,0.0004201975,0.0001558065,0.06655532,0.0003914263,0.0002678598,0.001836753],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2149297,"threshold_uncertainty_score":0.9681315,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2061554266","doi":"10.1111/j.1745-3992.2004.tb00165.x","title":"Assessing School Readiness: Validity and Bias in Preschool and Kindergarten Teachers' Ratings","year":2004,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":88,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Education and Early Childhood Development","funders":"","keywords":"Psychology; Head start; Vocabulary; Developmental psychology; Association (psychology); Early childhood education; Early childhood; Preschool education; Academic skills; Mathematics education","authors":[{"name":"Andrew J. Mashburn","is_ca":false},{"name":"Gary T. Henry","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1814566607573539,"gpt":0.4182170293389201,"spread":0.2367603685815662,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03359967,0.0004046684,0.0007764488,0.002718857,0.0007916318,0.001548381,0.0009039324,0.000623143,0.0008959404],"category_scores_gemma":[0.1151472,0.0006010858,0.0006727413,0.001700671,0.00169391,0.001369541,0.001891913,0.0008545333,0.0005193385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009738063,"about_ca_system_score_gemma":0.001161204,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01127513,"about_ca_topic_score_gemma":0.01428836,"domain_scores_codex":[0.9745207,0.009312212,0.00381864,0.002518607,0.008847356,0.0009824472],"domain_scores_gemma":[0.8942074,0.05027697,0.01840039,0.01075541,0.02445235,0.001907405],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001941737,0.0000626504,0.9736822,0.00008850202,0.0001420998,0.00003874136,0.004839264,0.0001762737,0.001192358,0.0002035295,0.0003030589,0.01907713],"study_design_scores_gemma":[0.0000269747,0.0001942917,0.9923217,0.0001092045,0.00005889086,0.0001862474,0.002971513,0.001068782,0.001340774,0.0003724253,0.001325146,0.00002413711],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9870833,0.0006533156,0.005169633,0.0001395392,0.00008121398,0.0002654914,0.0002486686,0.00008650907,0.006272395],"genre_scores_gemma":[0.9961954,0.0001958627,0.002140437,0.00008709445,0.00002314537,0.0002052592,0.0003953232,0.0000461231,0.0007112978],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03359967,"threshold_uncertainty_score":0.1776942,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2172179933","doi":"10.1111/emip.12018","title":"Instructional Topics in Educational Measurement (ITEMS) Module: Using Automated Processes to Generate Test Items","year":2013,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":71,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Rendering (computer graphics); Test (biology); Item bank; Process (computing); Item response theory; Task (project management); Item analysis; Artificial intelligence; Machine learning; Information retrieval; Psychometrics; Programming language; Psychology; Engineering","authors":[{"name":"Mark J. Gierl","is_ca":true},{"name":"Hollis Lai","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09722074847346185,"gpt":0.3606391693617605,"spread":0.2634184208882987,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005031583,0.001070521,0.0005474714,0.001856008,0.0004977059,0.001512308,0.001335552,0.001194659,0.02158501],"category_scores_gemma":[0.01919198,0.0008545227,0.0007501835,0.0009415366,0.0004731563,0.001472497,0.001878991,0.00142514,0.01750212],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005648952,"about_ca_system_score_gemma":0.001544012,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001091625,"about_ca_topic_score_gemma":0.001401566,"domain_scores_codex":[0.9975373,0.001198069,0.0002367024,0.0002518396,0.000657219,0.0001188043],"domain_scores_gemma":[0.9869334,0.006942446,0.0005686068,0.002180225,0.002702805,0.0006725545],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003817324,0.001802305,0.009491461,0.0004965241,0.0000769431,0.0001753041,0.001218377,0.002551381,0.01267287,0.007427388,0.09966303,0.8640426],"study_design_scores_gemma":[0.001561973,0.005708791,0.1183729,0.0007482915,0.0002812652,0.001883644,0.00076236,0.09066774,0.1444751,0.04467592,0.5904215,0.0004405856],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06924645,0.0002848451,0.7959716,0.002282211,0.0007628268,0.01481042,0.008618278,0.05544643,0.05257688],"genre_scores_gemma":[0.05736214,0.0001836399,0.9038378,0.0005819811,0.0001705565,0.007491949,0.00381462,0.001873661,0.02468369],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02158501,"threshold_uncertainty_score":0.07220906,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2119322254","doi":"10.1111/j.1745-3992.2001.tb00060.x","title":"Illustrating the Utility of Differential Bundle Functioning Analyses to Identify and Interpret Group Differences on Achievement Tests","year":2001,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":69,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Bundle; Differential (mechanical device); Interpretation (philosophy); Psychology; Computer science; Engineering","authors":[{"name":"Mark J. Gierl","is_ca":true},{"name":"Jeffrey Bisanz","is_ca":true},{"name":"Gay L. Bisanz","is_ca":true},{"name":"Keith A. Boughton","is_ca":true},{"name":"Shameem Nyla Khaliq","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.304702775564566,"gpt":0.5091262291350933,"spread":0.2044234535705273,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05314831,0.001565041,0.0009288082,0.005274084,0.001461615,0.003426208,0.001404488,0.001881341,0.00282851],"category_scores_gemma":[0.1593464,0.0007971666,0.002124665,0.004741353,0.003646288,0.004466166,0.004473612,0.002783102,0.000646644],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008534917,"about_ca_system_score_gemma":0.001188062,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002796054,"about_ca_topic_score_gemma":0.002989951,"domain_scores_codex":[0.9361959,0.05602833,0.001732226,0.001590593,0.003525241,0.0009278486],"domain_scores_gemma":[0.7806054,0.1931495,0.004073358,0.0126394,0.008916161,0.0006161845],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001712971,0.0006129591,0.2631093,0.0006447136,0.001248435,0.001136431,0.02996409,0.01754188,0.01160699,0.1895491,0.007067406,0.4758057],"study_design_scores_gemma":[0.0003181935,0.00121223,0.2632987,0.0003051484,0.0005627405,0.002035151,0.007897206,0.1840423,0.009613662,0.5190732,0.01135333,0.0002880679],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2285387,0.0004157991,0.7554831,0.002774164,0.0001122217,0.0003542372,0.0004991957,0.0009652316,0.01085733],"genre_scores_gemma":[0.6833574,0.00009672875,0.3146077,0.0003689713,0.00004414696,0.000481574,0.0002012525,0.0002009259,0.0006412465],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9468517,"threshold_uncertainty_score":0.2810785,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2111648217","doi":"10.1111/j.1745-3992.2010.00181.x","title":"Developing Score Reports for Cognitive Diagnostic Assessments","year":2010,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":68,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Context (archaeology); Test (biology); Cognition; Diagnostic test; Hierarchy; Sample (material); Knowledge management; Data science; Psychology; Medicine","authors":[{"name":"Mary Roduta Roberts","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.165780093144807,"gpt":0.4552123800807414,"spread":0.2894322869359344,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08867823,0.002530017,0.00129406,0.02205593,0.001668036,0.00711185,0.003859992,0.001358146,0.00504693],"category_scores_gemma":[0.305648,0.0008466556,0.001797102,0.008175002,0.00175423,0.007168976,0.006005995,0.003022376,0.004346006],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002569259,"about_ca_system_score_gemma":0.005630372,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002391073,"about_ca_topic_score_gemma":0.002214763,"domain_scores_codex":[0.9008323,0.04256547,0.02196205,0.003385955,0.03004692,0.001207244],"domain_scores_gemma":[0.6628904,0.1465306,0.04205452,0.03220217,0.1138915,0.002430704],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002417405,0.0002550644,0.02172922,0.001580567,0.0002010163,0.0003294584,0.005794339,0.006138311,0.004438039,0.09379523,0.04053209,0.8249649],"study_design_scores_gemma":[0.0003037456,0.00186447,0.03160755,0.005506775,0.0005169026,0.002657851,0.01169946,0.06775048,0.05689533,0.2650497,0.5551157,0.001031999],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009347145,0.0005963463,0.9661033,0.001167269,0.000483208,0.002588215,0.003194943,0.00552154,0.01099807],"genre_scores_gemma":[0.03763359,0.0004668282,0.9527227,0.0001775485,0.0001881627,0.003387114,0.003506862,0.0005264683,0.001390864],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.08867823,"threshold_uncertainty_score":0.4689809,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2109560181","doi":"10.1111/j.1745-3992.2005.00002.x","title":"Using Dimensionality‐Based DIF Analyses to Identify and Interpret Constructs That Elicit Group Differences","year":2005,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":59,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Matching (statistics); Psychology; Selection (genetic algorithm); Curse of dimensionality; Contrast (vision); Cognitive psychology; Test (biology); Social psychology; Computer science; Statistics; Artificial intelligence; Mathematics","authors":[{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8614623269368624,"gpt":0.6322567921565106,"spread":0.2292055347803518,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03683262,0.001728027,0.001156418,0.009132141,0.001509923,0.003539104,0.0008308985,0.0007620077,0.00309671],"category_scores_gemma":[0.1251279,0.00035332,0.001263369,0.006045313,0.002220669,0.003797907,0.004067377,0.00198036,0.0006839223],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001700195,"about_ca_system_score_gemma":0.001358172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000904433,"about_ca_topic_score_gemma":0.001251526,"domain_scores_codex":[0.970558,0.02190184,0.001970176,0.001616055,0.00343995,0.0005139546],"domain_scores_gemma":[0.8828627,0.09527585,0.007471258,0.007210835,0.006579368,0.0005999824],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004098389,0.0005753186,0.1229089,0.001842623,0.0008463906,0.00039676,0.03623871,0.005405751,0.009955879,0.1809203,0.007086998,0.6334125],"study_design_scores_gemma":[0.0002763071,0.001020444,0.1720813,0.001325104,0.0005699541,0.0013395,0.03393289,0.06313404,0.01443087,0.6734259,0.03787801,0.0005856522],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1318298,0.000633041,0.8456615,0.001671963,0.0002119745,0.002009053,0.0009850083,0.000460307,0.01653737],"genre_scores_gemma":[0.4420158,0.0005640623,0.5514025,0.0005044087,0.00007143705,0.003917148,0.0008270678,0.00008991329,0.0006077797],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03683262,"threshold_uncertainty_score":0.1947918,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2006618588","doi":"10.1111/j.1745-3992.2000.tb00036.x","title":"An NCME Instructional Module on Exploring the Logic of Tatsuoka's Rule‐Space Model for Test Development and Analysis","year":2000,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":56,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Social Sciences and Humanities Research Council; Alberta Advanced Education; University of Alberta","funders":"","keywords":"Test (biology); Set (abstract data type); Space (punctuation); Blueprint; Computer science; Cognition; Artificial intelligence; Rule-based system; Machine learning; Psychology; Programming language; Engineering","authors":[{"name":"Mark J. Gierl","is_ca":true},{"name":"Jacqueline P. Leighton","is_ca":true},{"name":"S. Hunka","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1492879822400197,"gpt":0.3656347844741414,"spread":0.2163468022341217,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003513545,0.001236976,0.0008981131,0.00243312,0.0009792079,0.001969221,0.002263982,0.002368057,0.06937638],"category_scores_gemma":[0.01587798,0.0006016945,0.0008627636,0.001239904,0.001288357,0.003360235,0.002967149,0.003165847,0.02489157],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001764138,"about_ca_system_score_gemma":0.002486504,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002095712,"about_ca_topic_score_gemma":0.008023508,"domain_scores_codex":[0.9989348,0.0004153869,0.00007606396,0.0001188908,0.0003915562,0.00006328975],"domain_scores_gemma":[0.9925776,0.004504138,0.0003324463,0.0005082919,0.001560972,0.0005164805],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0000547478,0.00103514,0.001395875,0.0003405912,0.00001941005,0.0004226864,0.0004195633,0.002853591,0.002996978,0.04596213,0.5493872,0.3951121],"study_design_scores_gemma":[0.00006840963,0.0002491852,0.005196165,0.0008781927,0.00001447187,0.0008054688,0.0003928946,0.009013141,0.001616997,0.0658235,0.9158693,0.00007227068],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007197671,0.004110483,0.6451235,0.0547293,0.01172328,0.003538016,0.001495995,0.005768808,0.266313],"genre_scores_gemma":[0.02983825,0.009140808,0.5270944,0.02928029,0.01048905,0.00539524,0.001852514,0.001575875,0.3853335],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.06937638,"threshold_uncertainty_score":0.2320871,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2057691077","doi":"10.1111/j.1745-3992.2009.01133.x","title":"Inclusive Achievement Testing for Linguistically and Culturally Diverse Test Takers: Essential Considerations for Test Developers and Decision Makers","year":2009,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Disability Education and Employment","field":"Social Sciences","cited_by":41,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Carleton University","funders":"","keywords":"Ell; Accountability; Test (biology); Government (linguistics); Standardized test; No child left behind; Equity (law); Achievement test; Political science; Language assessment; Psychology; English language; Public relations; Pedagogy; Mathematics education; Teaching method","authors":[{"name":"Shelley Fairbairn","is_ca":false},{"name":"Janna Fox","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.125304747687676,"gpt":0.4253981248863941,"spread":0.3000933771987181,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1367808,0.0007513547,0.001507097,0.00373957,0.005422009,0.01182723,0.004550952,0.00537501,0.001791119],"category_scores_gemma":[0.3013574,0.0007707811,0.0007344472,0.001935022,0.008276172,0.006903818,0.01190211,0.008514394,0.0007071162],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004775884,"about_ca_system_score_gemma":0.03733977,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02757517,"about_ca_topic_score_gemma":0.06652948,"domain_scores_codex":[0.9000168,0.04910496,0.01257547,0.002040456,0.03237109,0.003891136],"domain_scores_gemma":[0.6223534,0.2626596,0.01590485,0.01030754,0.05541475,0.03335992],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002906072,0.001453971,0.1357355,0.0007103531,0.0001118676,0.002977442,0.06001225,0.001978099,0.001928295,0.03476759,0.06689671,0.6931373],"study_design_scores_gemma":[0.0002938815,0.002278003,0.2201385,0.01112755,0.0004013001,0.008861032,0.2651187,0.01519972,0.009707529,0.1962921,0.2698099,0.0007717932],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"methods","genre_scores_codex":[0.1540393,0.0106919,0.07893905,0.6986554,0.001961744,0.001579014,0.000315817,0.0007401849,0.05307775],"genre_scores_gemma":[0.7404026,0.006544151,0.1912835,0.05280622,0.001148641,0.001833754,0.0002543662,0.0001854433,0.005541293],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.1367808,"threshold_uncertainty_score":0.7233744,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2064530600","doi":"10.1111/j.1745-3992.2009.01135.x","title":"An NCME Instructional Module on Using Differential Step Functioning to Refine the Analysis of DIF in Polytomous Items","year":2009,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Polytomous Rasch model; Differential item functioning; Item response theory; Psychology; Rasch model; Task (project management); Test (biology); Statistics; Differential (mechanical device); Psychometrics; Mathematics; Clinical psychology; Developmental psychology","authors":[{"name":"Randall D. Penfield","is_ca":false},{"name":"Karina A. Gattamorta","is_ca":false},{"name":"Ruth A. Childs","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5327361304287398,"gpt":0.5248611500943309,"spread":0.007874980334408921,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002979445,0.0008005282,0.0005148975,0.001765317,0.0008079801,0.001179148,0.0016143,0.001579653,0.07287819],"category_scores_gemma":[0.01281873,0.0002950899,0.0004678074,0.001074911,0.0006060225,0.002217436,0.002179144,0.002050913,0.02486497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001067271,"about_ca_system_score_gemma":0.002345973,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002459819,"about_ca_topic_score_gemma":0.01228914,"domain_scores_codex":[0.9989817,0.0003109956,0.00007903302,0.00009136874,0.0004744809,0.00006247413],"domain_scores_gemma":[0.9933363,0.00301519,0.0002437275,0.0004222852,0.002396196,0.0005862679],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006798976,0.001174624,0.002696866,0.0004893522,0.00001334326,0.0003469393,0.0005107,0.001092921,0.00440241,0.01052639,0.5799014,0.3987769],"study_design_scores_gemma":[0.00006463368,0.0003231508,0.01321853,0.0007447681,0.00001677008,0.0007423854,0.0004721702,0.004633877,0.005070243,0.01646131,0.9581938,0.000058248],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.02541424,0.003468031,0.3293912,0.07664284,0.01377744,0.006204394,0.003465208,0.008415297,0.5332215],"genre_scores_gemma":[0.05541914,0.006143339,0.3909602,0.03234274,0.003876309,0.005339121,0.002807438,0.001382122,0.5017296],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.07287819,"threshold_uncertainty_score":0.2438018,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4391227797","doi":"10.1111/emip.12590","title":"Using OpenAI GPT to Generate Reading Comprehension Items","year":2024,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Reading comprehension; Comprehension; Reading (process); Sentence; Computer science; Cognition; Natural language processing; Artificial intelligence; Sample (material); Psychology; Cognitive psychology; Linguistics","authors":[{"name":"Ayfer SAYIN","is_ca":false},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2292006550062576,"gpt":0.4155313631435855,"spread":0.1863307081373279,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006708799,0.001260643,0.000686163,0.002250435,0.0003782295,0.001606593,0.001627094,0.0008019766,0.008932563],"category_scores_gemma":[0.05775053,0.0005497162,0.0008059987,0.001459246,0.0005323419,0.001530407,0.001752133,0.001273373,0.004859292],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006136245,"about_ca_system_score_gemma":0.0006781245,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008953472,"about_ca_topic_score_gemma":0.0008328997,"domain_scores_codex":[0.9941532,0.002851601,0.0008039345,0.0008391111,0.001242512,0.0001096561],"domain_scores_gemma":[0.9454595,0.03644957,0.001921354,0.006327393,0.009467513,0.0003746814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007058989,0.0009847138,0.01967767,0.0007102253,0.0001121132,0.0006234278,0.004586733,0.01089695,0.02845694,0.003003902,0.01057994,0.9196615],"study_design_scores_gemma":[0.001631092,0.006623516,0.127087,0.0008522757,0.0004307322,0.004053501,0.0035609,0.4151379,0.3243577,0.02106772,0.09449054,0.0007070538],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2399371,0.0001321561,0.70674,0.0001827952,0.0002588954,0.008523776,0.004437896,0.02691454,0.01287306],"genre_scores_gemma":[0.2596103,0.00009997491,0.7200114,0.00007469645,0.00004361029,0.008080788,0.006155544,0.002069866,0.003853862],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008932563,"threshold_uncertainty_score":0.03547996,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2885169382","doi":"10.1111/emip.12211","title":"How Robust Are Cross‐Country Comparisons of PISA Scores to the Scaling Model Used?","year":2018,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Criticism; Robustness (evolution); Underpinning; Item response theory; Psychology; Test (biology); Cross country; Psychometrics; Political science; Developmental psychology; Economics; Demographic economics; Engineering","authors":[{"name":"John Jerrim","is_ca":false},{"name":"Philip D. Parker","is_ca":false},{"name":"Álvaro Choi","is_ca":false},{"name":"Anna K. Chmielewski","is_ca":true},{"name":"Christine Sälzer","is_ca":false},{"name":"Nikki Shure","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1301498662001438,"gpt":0.3787550552457036,"spread":0.2486051890455598,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1752474,0.001186957,0.001751624,0.003552754,0.001817807,0.007863597,0.004104985,0.002014757,0.008193031],"category_scores_gemma":[0.5214518,0.0007287199,0.003588361,0.006480979,0.004739623,0.005871437,0.005877248,0.004241674,0.002977046],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001329993,"about_ca_system_score_gemma":0.001455731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006513461,"about_ca_topic_score_gemma":0.003431921,"domain_scores_codex":[0.7958846,0.1635188,0.008506343,0.01714711,0.01210095,0.002842152],"domain_scores_gemma":[0.4642074,0.388118,0.04102074,0.07958438,0.02438313,0.00268637],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0021362,0.0005403552,0.7238457,0.001343145,0.02169016,0.00049515,0.009539147,0.02101484,0.001695386,0.04135913,0.01777097,0.1585699],"study_design_scores_gemma":[0.0005092201,0.002588385,0.7896217,0.003260493,0.00595405,0.0007606614,0.02936522,0.03650263,0.008088131,0.07724637,0.04556357,0.0005395883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7514302,0.004523666,0.1609038,0.01039313,0.004449642,0.00100455,0.006122195,0.0006920865,0.06048077],"genre_scores_gemma":[0.9857512,0.0002454584,0.009946746,0.0006397519,0.0002115768,0.0003323913,0.001891143,0.0002320943,0.000749607],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8247526,"threshold_uncertainty_score":0.9268077,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3046538170","doi":"10.1111/emip.12382","title":"Synergy and Tension between Large‐Scale and Classroom Assessment: International Trends","year":2020,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Queen's University; Brock University","funders":"","keywords":"Scale (ratio); Test (biology); Political science; Mathematics education; Pedagogy; Sociology; Psychology; Geography; Geology; Cartography","authors":[{"name":"Louis Volante","is_ca":true},{"name":"Christopher DeLuca","is_ca":true},{"name":"Lenore Adie","is_ca":false},{"name":"Eva L. Baker","is_ca":false},{"name":"Heidi Harju‐Luukkainen","is_ca":false},{"name":"Margaret Heritage","is_ca":false},{"name":"Christoph Schneider","is_ca":false},{"name":"Gordon Stobart","is_ca":false},{"name":"Kelvin Tan","is_ca":false},{"name":"Claire Wyatt‐Smith","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09632131387335943,"gpt":0.4206769110651631,"spread":0.3243555971918037,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0460766,0.0001960899,0.0005050337,0.004356804,0.0009616507,0.006938803,0.001265186,0.0009419395,0.002572443],"category_scores_gemma":[0.05747416,0.0003108683,0.0002128058,0.009207913,0.006573258,0.008068855,0.005446479,0.00269738,0.0002373255],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004967979,"about_ca_system_score_gemma":0.007635783,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01427181,"about_ca_topic_score_gemma":0.01366717,"domain_scores_codex":[0.9779327,0.009496402,0.002022549,0.003315778,0.005945744,0.00128688],"domain_scores_gemma":[0.7971092,0.1286396,0.01820693,0.008969309,0.04136628,0.005708543],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0002106703,0.0002017055,0.2032793,0.002084328,0.0001108703,0.0002203676,0.04126216,0.001305704,0.001764698,0.09733063,0.006467875,0.6457617],"study_design_scores_gemma":[0.00002368232,0.0004931898,0.7198912,0.004646169,0.00007877697,0.0009358522,0.1028281,0.002212136,0.0020134,0.02183255,0.144895,0.0001499825],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6923557,0.08575698,0.01210072,0.1065014,0.0005356059,0.00008126989,0.0004144431,0.0001727623,0.1020812],"genre_scores_gemma":[0.9889381,0.007133181,0.001986147,0.00106264,0.0001319265,0.00002246577,0.0000621931,0.00002624796,0.0006370611],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0460766,"threshold_uncertainty_score":0.2436792,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2562551526","doi":"10.1111/emip.12129","title":"A Process for Reviewing and Evaluating Generated Test Items","year":2016,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test (biology); Process (computing); Subject-matter expert; Quality (philosophy); Domain (mathematical analysis); Computerized adaptive testing; Item bank; Item response theory; Data science; Artificial intelligence; Psychometrics; Expert system; Mathematics; Statistics; Programming language","authors":[{"name":"Mark J. Gierl","is_ca":true},{"name":"Hollis Lai","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2464384762232207,"gpt":0.4293302723194807,"spread":0.18289179609626,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1307608,0.003086408,0.002419316,0.01272528,0.004675223,0.006173004,0.005034926,0.002691991,0.01011174],"category_scores_gemma":[0.3377042,0.001743782,0.002165699,0.00519865,0.002771081,0.004597738,0.005206759,0.004426127,0.01290997],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003643584,"about_ca_system_score_gemma":0.01619731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005471262,"about_ca_topic_score_gemma":0.007669582,"domain_scores_codex":[0.8578998,0.08629026,0.01208642,0.009734916,0.03284383,0.001144757],"domain_scores_gemma":[0.536819,0.1877406,0.02154977,0.07387221,0.1765254,0.003492941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004582713,0.0007804504,0.003660817,0.001394822,0.0001793804,0.0006340074,0.01486677,0.002545926,0.03171945,0.009786946,0.04917552,0.8847978],"study_design_scores_gemma":[0.0008649774,0.003304507,0.03169724,0.004539728,0.000723179,0.00466403,0.01543386,0.07513206,0.1280334,0.06104523,0.6724061,0.002155732],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01446872,0.0005935131,0.9418218,0.001675414,0.0004850369,0.02336576,0.0007035795,0.01012747,0.006758645],"genre_scores_gemma":[0.01704044,0.0002385235,0.9703625,0.0002675521,0.0001590779,0.006242898,0.0006736696,0.0009776956,0.004037637],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1307608,"threshold_uncertainty_score":0.6915373,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4313410276","doi":"10.1111/emip.12537","title":"Using Active Learning Methods to Strategically Select Essays for Automated Scoring","year":2022,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":19,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Scalability; Active learning (machine learning); Encoder; Transformer; Database; Engineering","authors":[{"name":"Tahereh Firoozi","is_ca":true},{"name":"Hamid Mohammadi","is_ca":false},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3365902019122547,"gpt":0.5652165547572977,"spread":0.228626352845043,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009012679,0.001024253,0.0008275858,0.002218383,0.0005658511,0.002347926,0.002028486,0.0009438706,0.002421166],"category_scores_gemma":[0.03578798,0.0003481179,0.0004752836,0.001017007,0.0007611456,0.002339405,0.001725329,0.001427078,0.00128935],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008422752,"about_ca_system_score_gemma":0.001141909,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001038913,"about_ca_topic_score_gemma":0.002240082,"domain_scores_codex":[0.9945723,0.002951531,0.0003483021,0.0006740714,0.001276304,0.0001774001],"domain_scores_gemma":[0.9594585,0.02750979,0.00269479,0.002420557,0.007185963,0.0007304285],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005755164,0.0006788155,0.009924227,0.0001435319,0.00008282861,0.00007601165,0.000559862,0.07586161,0.01485853,0.005712437,0.002231063,0.8892955],"study_design_scores_gemma":[0.00007734585,0.000261923,0.002011831,0.00002379124,0.00002185136,0.00005037778,0.0001745035,0.9686193,0.0202601,0.006826773,0.001642187,0.00002993722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1346761,0.000138181,0.8590714,0.0002653608,0.00008059505,0.0003486085,0.00011421,0.002285823,0.003019686],"genre_scores_gemma":[0.6561729,0.00005552311,0.3396644,0.00008356924,0.00004757701,0.000362164,0.0002587212,0.0001744585,0.003180752],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009012679,"threshold_uncertainty_score":0.04766423,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2034257998","doi":"10.1111/emip.12052","title":"What Role Does, and Should, the Test <i>Standards</i> Play Outside of the United States of America?","year":2014,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":19,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Citation; Library science; Test (biology); Sociology; Political science; History; Computer science","authors":[{"name":"Bruno D. Zumbo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3389256374159915,"gpt":0.4786025746879919,"spread":0.1396769372720004,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1082024,0.0005544831,0.00171044,0.003077786,0.004811467,0.02327595,0.002854971,0.005471779,0.003963138],"category_scores_gemma":[0.2443413,0.0005035619,0.000763663,0.004107104,0.02055561,0.02841644,0.003586876,0.008169869,0.000970325],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009331195,"about_ca_system_score_gemma":0.02860726,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04607267,"about_ca_topic_score_gemma":0.05189088,"domain_scores_codex":[0.9397777,0.04120071,0.002558421,0.002957101,0.01071655,0.00278944],"domain_scores_gemma":[0.745809,0.1631756,0.02298673,0.009008694,0.04797702,0.011043],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001368343,0.0003540665,0.06785116,0.000713132,0.0001860749,0.0001486884,0.008935169,0.0005545637,0.0004754481,0.5711451,0.0513467,0.2981532],"study_design_scores_gemma":[0.00009303438,0.0003434945,0.09206208,0.01329591,0.0004571145,0.0008771653,0.08229807,0.006213242,0.003236937,0.4893997,0.3113278,0.0003955287],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.05332189,0.02051545,0.02035399,0.8009057,0.004389107,0.00008118586,0.0002476212,0.0001618784,0.1000232],"genre_scores_gemma":[0.8874976,0.01259127,0.02431007,0.06801231,0.002484728,0.0002077659,0.0001681316,0.000184098,0.004544004],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1082024,"threshold_uncertainty_score":0.5722361,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2052069113","doi":"10.1111/emip.12003","title":"Validating Student Score Inferences With Person‐Fit Statistic and Verbal Reports: A Person‐Fit Study for Cognitive Diagnostic Assessment","year":2013,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Science Education and Pedagogy","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Statistic; Test (biology); Cognition; Consistency (knowledge bases); Psychology; Test statistic; Cognitive psychology; Artificial intelligence; Statistical hypothesis testing; Computer science; Statistics; Mathematics","authors":[{"name":"Ying Cui","is_ca":true},{"name":"Mary Roduta Roberts","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3346735437565425,"gpt":0.504415793140177,"spread":0.1697422493836346,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08252496,0.0007105224,0.0006651921,0.003607291,0.0009351211,0.002693391,0.001192899,0.001055906,0.0008172539],"category_scores_gemma":[0.2945027,0.0003556054,0.001281418,0.001769051,0.001666619,0.002989163,0.002508406,0.001471352,0.0002624767],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001166442,"about_ca_system_score_gemma":0.001386578,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007760533,"about_ca_topic_score_gemma":0.0009467518,"domain_scores_codex":[0.9239098,0.05379995,0.005892722,0.002748493,0.01261903,0.001029942],"domain_scores_gemma":[0.6410083,0.2684951,0.03050087,0.02487089,0.03317005,0.001954798],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009294103,0.002313661,0.8463715,0.0001933491,0.000341152,0.0002560475,0.04219732,0.002293357,0.003571115,0.003513831,0.0004879736,0.09753134],"study_design_scores_gemma":[0.0004007524,0.01208487,0.7891465,0.0003264952,0.0003637416,0.001828944,0.0551682,0.09181581,0.0325597,0.01002203,0.005851921,0.0004310863],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9732102,0.00003408569,0.02506634,0.00005132459,0.00002006299,0.000382993,0.00005232499,0.00005180681,0.001130849],"genre_scores_gemma":[0.9825916,0.0000207554,0.01649977,0.00004412394,0.00000960189,0.0005984916,0.0000712949,0.00002091998,0.0001433559],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08252496,"threshold_uncertainty_score":0.4364389,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2279705546","doi":"10.1111/emip.12103","title":"The Role of Socioeconomic Status in SAT–Freshman Grade Relationships Across Gender and Racial Subgroups","year":2016,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"School Choice and Performance","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"","keywords":"Socioeconomic status; Ethnic group; Race (biology); Demography; Psychology; Test (biology); Predictive power; Academic achievement; Developmental psychology; Sociology; Gender studies; Population","authors":[{"name":"Jana L. Higdem","is_ca":false},{"name":"Jack W. Kostal","is_ca":false},{"name":"Nathan R. Kuncel","is_ca":false},{"name":"Paul R. Sackett","is_ca":false},{"name":"Winny Shen","is_ca":true},{"name":"Adam Beatty","is_ca":false},{"name":"Thomas Kiger","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1052190430503295,"gpt":0.3906782734469965,"spread":0.285459230396667,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001583357,0.0002214196,0.0002972945,0.001171619,0.0006528393,0.001006812,0.0004521075,0.000352229,0.00369432],"category_scores_gemma":[0.006781888,0.0001516694,0.0004935638,0.0009214876,0.0004955537,0.0006994889,0.0009861281,0.000454568,0.0003938623],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000355614,"about_ca_system_score_gemma":0.000418145,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0121242,"about_ca_topic_score_gemma":0.02041151,"domain_scores_codex":[0.9993089,0.0002289322,0.00004630274,0.0001369047,0.0001330603,0.0001457937],"domain_scores_gemma":[0.996714,0.001024003,0.0008876669,0.0002394634,0.0003568542,0.0007780929],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00004629196,0.00002077239,0.9972154,0.000002782709,0.00005103147,0.00002583068,0.0003006459,0.00001978513,0.00009647907,0.0001193105,0.00006600279,0.002035751],"study_design_scores_gemma":[8.907763e-7,0.00003316766,0.9992441,0.000003299699,0.00001347821,0.00001606381,0.0003840065,0.00008881909,0.00003949066,0.00007376856,0.0001013922,0.000001558501],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9980391,0.000194276,0.00008957341,0.000144587,0.000008354691,0.00000366127,0.0001272165,0.000003421177,0.001389844],"genre_scores_gemma":[0.999485,0.00004044003,0.00002632309,0.00002006263,0.000004995103,0.000001819797,0.0001316093,0.000002172471,0.000287451],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0121242,"threshold_uncertainty_score":0.02410728,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2061917416","doi":"10.1111/emip.12015","title":"The Multiple‐Use of Accountability Assessments: Implications for the Process of Validation","year":2013,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Manitoba","funders":"","keywords":"Accountability; Process (computing); Argument (complex analysis); Quality (philosophy); Management science; Process management; Computer science; Best practice; Psychology; Political science; Medicine; Business; Engineering; Epistemology","authors":[{"name":"Martha Koch","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4021609414534796,"gpt":0.5386433098125486,"spread":0.136482368359069,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.721591,0.001751917,0.003008878,0.01031651,0.0254986,0.03246795,0.009584797,0.01394565,0.002968446],"category_scores_gemma":[0.7751388,0.002446288,0.00257444,0.009981265,0.1234637,0.04980734,0.02784471,0.02041711,0.0007283381],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.03987854,"about_ca_system_score_gemma":0.09643989,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02375266,"about_ca_topic_score_gemma":0.02391216,"domain_scores_codex":[0.145104,0.7647716,0.02317503,0.01155008,0.04986543,0.00553385],"domain_scores_gemma":[0.0600758,0.8492451,0.02181414,0.03232497,0.03292128,0.003618742],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00008476834,0.0001642141,0.01158504,0.0006922293,0.00007893084,0.000557445,0.1245465,0.001250456,0.0003783683,0.797132,0.002924558,0.0606056],"study_design_scores_gemma":[0.0001170672,0.0002251781,0.007423201,0.005104395,0.000046848,0.0007739118,0.06948519,0.007897135,0.001236501,0.8553298,0.05201752,0.0003433668],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06633445,0.007421612,0.5709605,0.2760496,0.001408725,0.003918594,0.00009930593,0.000401552,0.07340566],"genre_scores_gemma":[0.7337202,0.001123954,0.2519969,0.006191339,0.0002988302,0.003724456,0.00003307662,0.0001414417,0.002769921],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.278409,"threshold_uncertainty_score":0.3433279,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2055319746","doi":"10.1111/j.1745-3992.2004.tb00161.x","title":"Modeling Passing Rates on a Computer‐Based Medical Licensing Examination: An Application of Survival Data Analysis","year":2004,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Medical Education and Admissions","field":"Medicine","cited_by":14,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Covariate; United States Medical Licensing Examination; Proportional hazards model; Survival analysis; Medical school; Variable (mathematics); Medicine; Medical education; Computer science; Psychology; Statistics; Surgery; Mathematics","authors":[{"name":"André F. De Champlain","is_ca":false},{"name":"Marcia L. Winward","is_ca":false},{"name":"Gerard F. Dillon","is_ca":false},{"name":"Judy E. de Champlain","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2274161709273148,"gpt":0.4633927414507003,"spread":0.2359765705233855,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02718575,0.00102056,0.001371091,0.003094251,0.0007219603,0.001349464,0.002341066,0.001395439,0.004636914],"category_scores_gemma":[0.06481484,0.0005277035,0.003336468,0.002818816,0.00101438,0.001422114,0.001695725,0.002210493,0.0008208221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001436393,"about_ca_system_score_gemma":0.002685471,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02209905,"about_ca_topic_score_gemma":0.0111866,"domain_scores_codex":[0.9889443,0.00837154,0.0004920274,0.0009090631,0.000742442,0.0005406158],"domain_scores_gemma":[0.9444001,0.04567847,0.004589197,0.002897535,0.001879389,0.000555339],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001578173,0.0007841812,0.6381362,0.0003170576,0.001648892,0.0004720283,0.001920214,0.2156324,0.0005281359,0.01888532,0.00279599,0.1173014],"study_design_scores_gemma":[0.0001622947,0.001240692,0.08098049,0.00009483654,0.0004667662,0.0003608507,0.0007111079,0.9003795,0.0007733025,0.01189611,0.00284679,0.0000873312],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.70603,0.0004793998,0.2850737,0.001922567,0.0001615148,0.001096981,0.002369596,0.0006874882,0.002178694],"genre_scores_gemma":[0.9437022,0.0003169373,0.04971965,0.0001391787,0.00007581282,0.001145715,0.00129503,0.00007283036,0.003532695],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02718575,"threshold_uncertainty_score":0.1437737,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2795852592","doi":"10.1111/emip.12198","title":"A Review of Recent Research on Individual‐Level Score Reports","year":2018,"lang":"en","type":"review","venue":"Educational Measurement Issues and Practice","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":13,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Context (archaeology); Test (biology); Computer science; Focus (optics); Knowledge management; Psychology; Data science","authors":[{"name":"Chad M. Gotch","is_ca":false},{"name":"Mary Roduta Roberts","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8619976910036018,"gpt":0.6550214592649924,"spread":0.2069762317386094,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01467553,0.001015553,0.00259973,0.01281894,0.0005817272,0.003266291,0.002441841,0.001789397,0.005680384],"category_scores_gemma":[0.05508785,0.0008405049,0.001506608,0.01569238,0.002025522,0.004220294,0.001579882,0.001772219,0.001563508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001672361,"about_ca_system_score_gemma":0.005946589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002618436,"about_ca_topic_score_gemma":0.004128112,"domain_scores_codex":[0.9910507,0.003136129,0.002243983,0.0009263424,0.002477795,0.0001649873],"domain_scores_gemma":[0.919363,0.06690966,0.005721725,0.001223004,0.006361387,0.0004212155],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0000828938,0.00005409628,0.001272774,0.1382306,0.0003537388,0.0001142688,0.0007474707,0.000178225,0.0003081561,0.003657439,0.01146996,0.8435304],"study_design_scores_gemma":[0.00003127918,0.000174691,0.008966501,0.2738266,0.001314432,0.001650974,0.001450574,0.0001296428,0.0007675228,0.003474463,0.7081256,0.00008771259],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0002773972,0.998077,0.0002690632,0.00038991,0.0001182107,0.00001518938,0.00006126344,0.000009147149,0.0007827061],"genre_scores_gemma":[0.002854014,0.9959576,0.0005776692,0.0002567514,0.0001206368,0.0000328039,0.00008525572,0.000007599288,0.0001076811],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01467553,"threshold_uncertainty_score":0.07761252,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3035205464","doi":"10.1111/emip.12353","title":"Exploring the Structure of Teachers’ Emotional Labor in the Classroom: A Multitrait–Multimethod Analysis","year":2020,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Emotional Labor in Professions","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"McGill University","funders":"Social Sciences and Humanities Research Council of Canada; Education University of Hong Kong","keywords":"Psychology; Emotional labor; Pride; Valence (chemistry); Social psychology; Anxiety; Negative emotion; School teachers; Emotional expression; Developmental psychology; Mathematics education","authors":[{"name":"Hui Wang","is_ca":false},{"name":"Nathan C. Hall","is_ca":true},{"name":"Ming Ming Chiu","is_ca":false},{"name":"Thomas Goetz","is_ca":false},{"name":"Katarzyna Gogol","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2881136308792331,"gpt":0.4361760759650995,"spread":0.1480624450858664,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01055223,0.0004190355,0.0007405441,0.002166563,0.001443625,0.002096128,0.0009654688,0.0005316353,0.002280953],"category_scores_gemma":[0.02606127,0.0004587705,0.0009081578,0.001658746,0.001500329,0.0009506906,0.001796487,0.0009799527,0.0002030995],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001800822,"about_ca_system_score_gemma":0.001667815,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02333754,"about_ca_topic_score_gemma":0.02455503,"domain_scores_codex":[0.9920433,0.004269126,0.000611557,0.0009163999,0.001675326,0.0004843169],"domain_scores_gemma":[0.97732,0.01458082,0.003418077,0.002791533,0.001411239,0.0004784085],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001953414,0.000669294,0.9610621,0.0000914308,0.0006176716,0.0000604557,0.01132014,0.0009105076,0.001902098,0.0006977929,0.0003116645,0.02216136],"study_design_scores_gemma":[0.00001749833,0.0002465501,0.9852951,0.00002795238,0.00009871364,0.00004219843,0.005751342,0.006852278,0.0006405703,0.0004892698,0.0005133159,0.00002527354],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.995659,0.00005551895,0.003400897,0.00003838032,0.000006323196,0.0001149634,0.0001041738,0.00001402043,0.0006067884],"genre_scores_gemma":[0.9971492,0.00002392687,0.002233638,0.00001696969,0.000005520458,0.0001885091,0.0001463774,0.000006403636,0.0002295608],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02333754,"threshold_uncertainty_score":0.05580616,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2804917593","doi":"10.1111/emip.12201","title":"Methodologies for Investigating and Interpreting Student–Teacher Rating Incongruence in Noncognitive Assessment","year":2018,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Education, Achievement, and Giftedness","field":"Psychology","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"American Psychological Association; American Educational Research Association","keywords":"Psychology; Construct (python library); Variety (cybernetics); Congruence (geometry); Predictive validity; Interpretation (philosophy); Divergence (linguistics); Construct validity; Descriptive statistics; Mathematics education; Social psychology; Psychometrics; Developmental psychology; Statistics","authors":[{"name":"Jessica Kay Flake","is_ca":true},{"name":"Kevin T. Petway","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2581577380829526,"gpt":0.5537707121933004,"spread":0.2956129741103478,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1367607,0.000893831,0.001008937,0.005968957,0.001592504,0.003845818,0.002575677,0.001060196,0.002516211],"category_scores_gemma":[0.3628865,0.001069445,0.001149315,0.005096626,0.002848028,0.002511414,0.004099731,0.00190519,0.0006706492],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001627496,"about_ca_system_score_gemma":0.001854635,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001536824,"about_ca_topic_score_gemma":0.002564685,"domain_scores_codex":[0.7995602,0.15543,0.01451181,0.007748523,0.02168883,0.001060677],"domain_scores_gemma":[0.5859731,0.2691743,0.05867685,0.04421457,0.04031019,0.001650948],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00104019,0.001119471,0.5121197,0.001965207,0.001025637,0.0004044708,0.05737766,0.005379856,0.01610642,0.02917388,0.003458981,0.3708285],"study_design_scores_gemma":[0.0003120802,0.002690637,0.7540978,0.001907641,0.0004872467,0.001074559,0.03402624,0.084483,0.03061602,0.0657859,0.02390242,0.0006165834],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2929689,0.0004774119,0.6892343,0.0003334179,0.0001797694,0.004692743,0.0005304641,0.0004687373,0.01111436],"genre_scores_gemma":[0.6659042,0.0002137094,0.3199626,0.000155744,0.00007279895,0.01221059,0.0004299588,0.0001614981,0.000888954],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1367607,"threshold_uncertainty_score":0.7232683,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2002384848","doi":"10.1111/j.1745-3992.2002.tb00103.x","title":"What Do School‐Level Scores From Large‐Scale Assessments Really Measure?","year":2002,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Cognitive Abilities and Testing","field":"Psychology","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Toronto Public Health","funders":"","keywords":"Reading (process); Scale (ratio); Variance (accounting); Psychology; Cognition; Measure (data warehouse); Subject (documents); Illusion; Mathematics education; Cognitive psychology; Computer science; Linguistics; Data mining","authors":[{"name":"Fiore Sicoly","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2066488833662569,"gpt":0.420967314211943,"spread":0.2143184308456861,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03268831,0.0009292655,0.001778054,0.005404849,0.0006705684,0.003971464,0.001924586,0.002854183,0.001513432],"category_scores_gemma":[0.1700092,0.0004342166,0.0008685076,0.00557355,0.003404573,0.006482118,0.001506127,0.001826352,0.001617217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001134883,"about_ca_system_score_gemma":0.001299076,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004373445,"about_ca_topic_score_gemma":0.008270957,"domain_scores_codex":[0.9839517,0.007113536,0.001938902,0.001693773,0.004821048,0.0004810106],"domain_scores_gemma":[0.8693184,0.0658543,0.02274621,0.01358469,0.02548541,0.003010892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00009423312,0.0001886263,0.8025778,0.0006706264,0.0006675349,0.00007343938,0.004396125,0.0006442324,0.0005050607,0.003374386,0.01112951,0.1756784],"study_design_scores_gemma":[0.00004715119,0.0004565548,0.9618918,0.0007461768,0.0003117224,0.0003725226,0.003961861,0.002018998,0.0009979142,0.01644321,0.01265361,0.00009870264],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8612076,0.01008666,0.04156512,0.01904782,0.002364371,0.0005587204,0.003760676,0.0008739569,0.06053505],"genre_scores_gemma":[0.9798135,0.002029156,0.01264614,0.001864617,0.0007024847,0.0003733855,0.001224313,0.0001730636,0.001173387],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03268831,"threshold_uncertainty_score":0.1728743,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2615786270","doi":"10.1111/emip.12150","title":"Differential Prediction in the Use of the SAT and High School Grades in Predicting College Performance: Joint Effects of Race and Language","year":2017,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Higher Education Research Studies","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Waterloo","funders":"University of Minnesota","keywords":"Ethnic group; Language proficiency; Psychology; Race (biology); Standardized test; Affect (linguistics); First language; Language assessment; Mathematics education; Linguistics; Sociology; Gender studies","authors":[{"name":"Oren R. Shewach","is_ca":false},{"name":"Winny Shen","is_ca":true},{"name":"Paul R. Sackett","is_ca":false},{"name":"Nathan R. Kuncel","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09120021442940497,"gpt":0.3903774829567992,"spread":0.2991772685273942,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007017468,0.0004589895,0.0002855111,0.001198281,0.0005097538,0.001748883,0.0006282952,0.0005311784,0.001986844],"category_scores_gemma":[0.02733024,0.0002193034,0.000581077,0.000866248,0.00079696,0.00105656,0.001489546,0.00113995,0.0006200779],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003384895,"about_ca_system_score_gemma":0.0007312242,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008895,"about_ca_topic_score_gemma":0.01980748,"domain_scores_codex":[0.9965185,0.002305234,0.0001769267,0.0003019686,0.0004295516,0.0002678012],"domain_scores_gemma":[0.9792008,0.01218734,0.002969662,0.001527309,0.001372325,0.002742662],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00005990085,0.00005056053,0.9972529,0.00000217609,0.00003849334,0.000008724865,0.000102347,0.00006649874,0.00005673657,0.00004657774,0.00004471941,0.002270437],"study_design_scores_gemma":[0.000004779656,0.0001367882,0.9973301,0.00001358992,0.00003672185,0.00003091915,0.0004564474,0.001513558,0.0001778813,0.000172076,0.0001220722,0.00000513246],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9982685,0.0001009269,0.0003374418,0.0001079685,0.00001048665,0.000008567546,0.00005237188,0.000007995858,0.00110568],"genre_scores_gemma":[0.9994639,0.00003573557,0.0001292058,0.00002055972,0.000006851355,0.000003578029,0.00007328799,0.000003240665,0.0002635789],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008895,"threshold_uncertainty_score":0.03711236,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1876976565","doi":"10.1111/j.1745-3992.2010.00198.x","title":"Reporting the Percentage of Students above a Cut Score: The Effect of Group Size","year":2011,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Mathematics education; Scale (ratio); Statistics; Psychology; Reading (process); Mathematics; Geography; Cartography; Political science","authors":[{"name":"Lynne Marguerite Hollingshead","is_ca":true},{"name":"Ruth A. Childs","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3899556216284745,"gpt":0.5438855402313928,"spread":0.1539299186029183,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3075692,0.001223244,0.001945327,0.004067761,0.002409939,0.003070505,0.003489143,0.002104631,0.001782198],"category_scores_gemma":[0.654336,0.001110365,0.003087158,0.004799057,0.005304018,0.003552815,0.004268879,0.00250008,0.0006032989],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002641438,"about_ca_system_score_gemma":0.001707794,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01049613,"about_ca_topic_score_gemma":0.009130489,"domain_scores_codex":[0.5072284,0.3768885,0.03421108,0.02782593,0.05118144,0.002664583],"domain_scores_gemma":[0.08421142,0.7973049,0.04694694,0.05009352,0.02010393,0.001339323],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01001756,0.0006051736,0.8779907,0.0007489296,0.00671886,0.0002510549,0.01103949,0.003305553,0.002734716,0.002364464,0.004834987,0.0793885],"study_design_scores_gemma":[0.0003183671,0.004281472,0.9664199,0.0003937517,0.002109097,0.0004308068,0.002620434,0.009466684,0.006431779,0.002249213,0.00509032,0.0001882076],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.896157,0.004903854,0.07722037,0.002567616,0.00085169,0.001925364,0.001456097,0.0007510402,0.014167],"genre_scores_gemma":[0.9777914,0.0001695992,0.01913293,0.000475804,0.00008253072,0.0008696303,0.0004348489,0.0002008903,0.0008424221],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3075692,"threshold_uncertainty_score":0.8538904,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4405618562","doi":"10.1111/emip.12663","title":"Instruction‐Tuned Large‐Language Models for Quality Control in Automatic Item Generation: A Feasibility Study","year":2024,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Quality (philosophy); Computer science; Control (management); Item response theory; Language proficiency; Mathematics education; Natural language processing; Psychology; Artificial intelligence; Psychometrics; Developmental psychology","authors":[{"name":"Guher Gorgun","is_ca":true},{"name":"Okan Bulut","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1967005621322491,"gpt":0.4278503410420334,"spread":0.2311497789097843,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03405164,0.001157849,0.0007550939,0.000753626,0.000533775,0.001884429,0.003070298,0.001326082,0.00270793],"category_scores_gemma":[0.1322777,0.000945216,0.0005778416,0.0007102212,0.000917786,0.00323792,0.001626935,0.002113918,0.0008768544],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001603922,"about_ca_system_score_gemma":0.001880966,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008011228,"about_ca_topic_score_gemma":0.005956651,"domain_scores_codex":[0.9849164,0.01145417,0.0009187471,0.00113738,0.001252314,0.0003209713],"domain_scores_gemma":[0.8163837,0.1545434,0.003561013,0.01226614,0.01161165,0.001634093],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01147016,0.02063601,0.08636843,0.001161271,0.0004352772,0.0009060684,0.007445504,0.1733389,0.06225172,0.005511854,0.0082977,0.6221771],"study_design_scores_gemma":[0.001451004,0.004709852,0.0165485,0.00009154419,0.000152971,0.0001976538,0.0005376074,0.9450889,0.02638712,0.002219969,0.002486016,0.0001288998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8504725,0.0001305987,0.1416808,0.000483351,0.00006182345,0.002063565,0.0002938132,0.003482544,0.001330942],"genre_scores_gemma":[0.8636357,0.00002986427,0.1345947,0.000114437,0.0000154556,0.0008314723,0.0002336638,0.0002096155,0.0003351135],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03405164,"threshold_uncertainty_score":0.1800845,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4319790235","doi":"10.1111/emip.12539","title":"Machine Learning Literacy for Measurement Professionals: A Practical Tutorial","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Medical Council of Canada","funders":"","keywords":"Toolbox; Computer science; Python (programming language); Data science; Context (archaeology); Artificial intelligence","authors":[{"name":"Rui Nie","is_ca":true},{"name":"Qi Guo","is_ca":true},{"name":"Maxim Morin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1935689270038954,"gpt":0.4464950114435416,"spread":0.2529260844396461,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002324205,0.001861665,0.0006857439,0.001729045,0.001208195,0.003490849,0.001354188,0.003310825,0.06550618],"category_scores_gemma":[0.009499392,0.0005344067,0.0008631657,0.001087887,0.0009767095,0.007632283,0.004278048,0.003927477,0.03056145],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0015992,"about_ca_system_score_gemma":0.002065645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007332377,"about_ca_topic_score_gemma":0.00149837,"domain_scores_codex":[0.9986584,0.0005536935,0.0001005281,0.0001800687,0.0003244852,0.0001827555],"domain_scores_gemma":[0.9959986,0.002464117,0.0001789596,0.0001376412,0.0007318705,0.0004888736],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003633596,0.0001816829,0.0002883685,0.0007271768,0.00001032431,0.0003938932,0.001831375,0.0007453235,0.001617124,0.03740408,0.67015,0.2866143],"study_design_scores_gemma":[0.00001252784,0.00005815257,0.0003958476,0.001211527,0.000004984559,0.0006642115,0.0004905313,0.001016559,0.0005079714,0.02584305,0.9697658,0.00002887728],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01316217,0.05995106,0.4524196,0.2111966,0.0222873,0.001604803,0.004019381,0.02269709,0.212662],"genre_scores_gemma":[0.07694711,0.07186833,0.3490481,0.06969782,0.01481138,0.003230477,0.008267229,0.00670344,0.3994262],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.06550618,"threshold_uncertainty_score":0.2191401,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4386482646","doi":"10.1111/emip.12572","title":"Digital Module 33: Fairness in Classroom Assessment: Dimensions and Tensions","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Legitimacy; Perception; Psychology; Disengagement theory; Critical reflection; Social psychology; Pedagogy; Political science","authors":[{"name":"Amirhossein Rasooli","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1105547684473078,"gpt":0.4277434972698937,"spread":0.3171887288225859,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002996639,0.0002959309,0.0003553841,0.0009530309,0.000688381,0.002366425,0.0005804882,0.0008413731,0.03363624],"category_scores_gemma":[0.008356256,0.0001635056,0.0002835893,0.001199334,0.0007191504,0.001269269,0.002010331,0.001336157,0.005133712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001322845,"about_ca_system_score_gemma":0.001888287,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009364573,"about_ca_topic_score_gemma":0.001960189,"domain_scores_codex":[0.9985655,0.0005285444,0.0001291683,0.00008990958,0.000504419,0.0001825791],"domain_scores_gemma":[0.9950497,0.001953311,0.0004819242,0.0003615623,0.001215348,0.0009381611],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002289899,0.001521808,0.02725884,0.0008043659,0.00001953324,0.0001727653,0.006041637,0.001582993,0.004564255,0.01720696,0.2458157,0.6947821],"study_design_scores_gemma":[0.00005398588,0.0009411409,0.2030651,0.001592934,0.00002835968,0.0009316682,0.007716163,0.004457063,0.008146683,0.02419714,0.7487369,0.0001330469],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5251467,0.002975557,0.04715491,0.03989238,0.002740375,0.003131433,0.004875981,0.002283016,0.3717996],"genre_scores_gemma":[0.7872729,0.004544395,0.0323427,0.004432544,0.001357504,0.002289751,0.003520265,0.0004672747,0.1637725],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03363624,"threshold_uncertainty_score":0.1125244,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4389347402","doi":"10.1111/emip.12585","title":"Digital Module 34: Introduction to Multilevel Measurement Modeling","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Mental Health Research Topics","field":"Psychology","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University","funders":"","keywords":"Structural equation modeling; Multilevel model; Computer science; Code (set theory); Latent variable; Process (computing); Data mining; Software engineering; Programming language; Artificial intelligence; Machine learning","authors":[{"name":"Mairead Shaw","is_ca":true},{"name":"Jessica Kay Flake","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3432883525820852,"gpt":0.4905485480493123,"spread":0.1472601954672271,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008192302,0.001032913,0.001049873,0.002366054,0.0005967552,0.002370733,0.002718792,0.001265862,0.2648159],"category_scores_gemma":[0.03672685,0.001646984,0.002291665,0.003219782,0.0005015482,0.00273433,0.002489928,0.003014383,0.125573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001316046,"about_ca_system_score_gemma":0.002214252,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002261269,"about_ca_topic_score_gemma":0.00249484,"domain_scores_codex":[0.9960145,0.002133652,0.0003573996,0.0003815401,0.0009902486,0.0001226781],"domain_scores_gemma":[0.9841192,0.01036156,0.0005312773,0.001830733,0.002772261,0.0003848687],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00005419242,0.0001184987,0.001147831,0.0009207024,0.00007866925,0.0000934648,0.0002877365,0.003589641,0.0008687341,0.04959067,0.641003,0.3022469],"study_design_scores_gemma":[0.000091799,0.0001133107,0.003803872,0.001138517,0.00004474549,0.0002422425,0.00008117612,0.01148427,0.001543509,0.07731537,0.9040373,0.0001039749],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001512139,0.001546722,0.8427812,0.006359137,0.00106826,0.003742112,0.02901872,0.03627136,0.07770034],"genre_scores_gemma":[0.01735487,0.003476891,0.8608521,0.003242705,0.001442832,0.01115076,0.02194619,0.02162906,0.05890455],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2648159,"threshold_uncertainty_score":0.8858973,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4233643233","doi":"10.1111/emip.12321","title":"Digital Module 12: Think‐aloud Interviews and Cognitive Labs https://ncme.elevate.commpartners.com","year":2020,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Cognitive Science and Mapping","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Credibility; Think aloud protocol; Cognitive interview; Glossary; Interview; Cognition; Data collection; Psychology; Computer science; Protocol analysis; Comprehension; Reliability (semiconductor); Test (biology); Multimedia; Human–computer interaction; Cognitive science; Usability; Linguistics","authors":[{"name":"Jacqueline P. Leighton","is_ca":true},{"name":"Blair Lehman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1497824668143513,"gpt":0.3536044543898638,"spread":0.2038219875755124,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004738505,0.0007436536,0.0005314983,0.002473379,0.0006535278,0.001436165,0.001199639,0.0008123282,0.3941737],"category_scores_gemma":[0.01440992,0.0004581227,0.0003886996,0.001915936,0.0003432275,0.001449476,0.002031577,0.0007937938,0.2320594],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008676577,"about_ca_system_score_gemma":0.001452662,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006069419,"about_ca_topic_score_gemma":0.001344305,"domain_scores_codex":[0.997419,0.00112435,0.000151191,0.000260575,0.0008374518,0.0002074159],"domain_scores_gemma":[0.9871149,0.005766004,0.0004464868,0.00134136,0.004207558,0.001123785],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001690915,0.0002810544,0.0005503812,0.00038547,0.000004772177,0.00006734956,0.0008996825,0.000183641,0.002413431,0.001130997,0.6672739,0.3266404],"study_design_scores_gemma":[0.0001043951,0.0003000737,0.005643428,0.0002851186,0.000005181516,0.0001862566,0.0006858766,0.0009531276,0.005370014,0.002340172,0.9840949,0.00003132412],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.04837463,0.0007957253,0.2489475,0.006778591,0.003066994,0.02091995,0.08242603,0.0817665,0.5069241],"genre_scores_gemma":[0.05459256,0.000855057,0.252921,0.002659929,0.001201568,0.0306477,0.04883272,0.01263692,0.5956525],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.3941737,"threshold_uncertainty_score":0.8641377,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2034878828","doi":"10.1111/j.1745-3992.2002.tb00087.x","title":"Scoring Examinee Responses for Multiple Inferences: Multiple Scoring in Assessments","year":2002,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Scoring system; Inference; Scale (ratio); Computer science; Psychology; Artificial intelligence; Medicine; Geography","authors":[{"name":"Kadriye Ercikan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8659566286900213,"gpt":0.5782094015970195,"spread":0.2877472270930018,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2417733,0.001823998,0.00271525,0.008021637,0.003583667,0.006277646,0.004437303,0.003681848,0.003011699],"category_scores_gemma":[0.6199554,0.001485733,0.001669702,0.009639783,0.005026605,0.008386918,0.009490683,0.005786226,0.0017098],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00234414,"about_ca_system_score_gemma":0.005235274,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00202344,"about_ca_topic_score_gemma":0.004366837,"domain_scores_codex":[0.4610018,0.4184369,0.03350605,0.01260563,0.0722926,0.002156997],"domain_scores_gemma":[0.3675581,0.4660519,0.04549431,0.05264957,0.06584612,0.002399975],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006182455,0.0005661629,0.05209879,0.003940347,0.001112608,0.0007368368,0.02457538,0.001912921,0.00455642,0.05679538,0.02009414,0.8329927],"study_design_scores_gemma":[0.0008950499,0.002880175,0.208109,0.01493108,0.002333569,0.009783708,0.02252549,0.068446,0.05486838,0.4212022,0.1920457,0.001979696],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0377364,0.003521816,0.9341734,0.004891808,0.001254472,0.003304689,0.0002142485,0.001044,0.0138592],"genre_scores_gemma":[0.2196508,0.002432112,0.7662473,0.002255481,0.0007000656,0.005131784,0.0002537149,0.0003984318,0.002930346],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2417733,"threshold_uncertainty_score":0.9350285,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4362673096","doi":"10.1111/emip.12553","title":"Validation as Evaluating Desired and Undesired Effects: Insights From Cross‐Classified Mixed Effects Model","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Reliability (semiconductor); Variance (accounting); Computer science; Reliability engineering; Variance components; Validity; External validity; Cross-validation; Random effects model; Statistics; Data mining; Econometrics; Psychology; Artificial intelligence; Psychometrics; Mathematics; Engineering; Medicine","authors":[{"name":"Xuejun Ryan Ji","is_ca":true},{"name":"Amery D. Wu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.7119997234981603,"gpt":0.5622263498475483,"spread":0.1497733736506121,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4465812,0.002467516,0.004040416,0.005289753,0.002344986,0.007238971,0.005535049,0.004590868,0.004024855],"category_scores_gemma":[0.6171535,0.001745145,0.008743202,0.003791886,0.006440164,0.006994923,0.005814273,0.006149193,0.0004122041],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003804737,"about_ca_system_score_gemma":0.005010772,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006114068,"about_ca_topic_score_gemma":0.004311057,"domain_scores_codex":[0.4743856,0.4918931,0.009345068,0.01003411,0.01331841,0.001023673],"domain_scores_gemma":[0.1053897,0.8527051,0.01225304,0.01803736,0.01099079,0.0006240343],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002086495,0.0009936129,0.1154624,0.002691276,0.01001178,0.001193763,0.01334507,0.1259694,0.001318283,0.5263015,0.003523879,0.1971025],"study_design_scores_gemma":[0.0002996612,0.001238022,0.01525,0.001679019,0.002436363,0.0003821402,0.001424269,0.712891,0.001833495,0.2571212,0.00520655,0.0002382346],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04507495,0.001123756,0.9484598,0.001217002,0.0001626336,0.0008421033,0.0002005038,0.0002107469,0.002708452],"genre_scores_gemma":[0.5017322,0.0004442841,0.4930753,0.0007310767,0.0001287713,0.00257916,0.0003633877,0.000151241,0.0007944524],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4465812,"threshold_uncertainty_score":0.682464,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4237928641","doi":"10.1111/emip.12098","title":"On This Issue's Cover","year":2015,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Education Practices and Evaluation","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Scrolling; Test (biology); Depiction; Set (abstract data type); Psychology; Session (web analytics); Think aloud protocol; Multiple choice; Computer science; Applied psychology; Human–computer interaction; World Wide Web; Artificial intelligence; Visual arts; Usability; Linguistics","authors":[{"name":"Katherine Furgol Castellano","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2812239541239853,"gpt":0.4813383893869291,"spread":0.2001144352629438,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0007960064,0.001438583,0.001378336,0.001734576,0.001676392,0.006681609,0.002176344,0.003945583,0.8323197],"category_scores_gemma":[0.007495862,0.0005594381,0.001081677,0.001335749,0.0006597284,0.003906981,0.003356014,0.003928324,0.6997379],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001118508,"about_ca_system_score_gemma":0.001445647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00126657,"about_ca_topic_score_gemma":0.002456242,"domain_scores_codex":[0.9989302,0.00009282821,0.00005848513,0.0001757202,0.0005614083,0.0001814824],"domain_scores_gemma":[0.9954626,0.0006239059,0.000233439,0.0004732574,0.001944718,0.001262072],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000009194225,0.00001476251,0.00002341346,0.00004455696,8.647663e-7,0.00001655333,0.000003907657,0.00001055511,0.00004535009,0.0002318934,0.9859541,0.01364499],"study_design_scores_gemma":[0.000005925925,0.00001067936,0.0001702276,0.00006480882,0.000001036611,0.00002326763,0.00001637508,0.00002311451,0.00003532639,0.0002013558,0.9994442,0.000003648635],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"editorial","genre_scores_codex":[0.0005013495,0.002958621,0.0009023096,0.01463337,0.2941839,0.0004595345,0.003428284,0.004180555,0.6787521],"genre_scores_gemma":[0.001084598,0.001335835,0.0003315326,0.00599985,0.03900959,0.0001215536,0.001832641,0.001071002,0.9492133],"genre_candidate":"editorial","genre_consensus":null,"teacher_disagreement_score":0.8323197,"threshold_uncertainty_score":0.2391755,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4246593371","doi":"10.1111/emip.12090","title":"Issue Information ‐ TOC &amp; Editorial board","year":2016,"lang":"en","type":"paratext","venue":"Educational Measurement Issues and Practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Editorial board; Computer science; Business; Environmental science; Library science","authors":[{"name":"André A. Rupp","is_ca":true},{"name":"Mark J. Gierl","is_ca":false},{"name":"Hollis Lai","is_ca":false},{"name":"David Andrich","is_ca":false},{"name":"Martin C. Yu","is_ca":false},{"name":"Paul R. Sackett","is_ca":false},{"name":"Nathan R. Kuncel","is_ca":false},{"name":"Howard T. Everson","is_ca":false},{"name":"Sarah M. Bonner","is_ca":false},{"name":"Katherine E. Castellano","is_ca":false},{"name":"Jamaal Abedi","is_ca":false},{"name":"Kyle J. Barton","is_ca":false},{"name":"Ctb-Mcgraw-Hill Bennett","is_ca":false},{"name":"Susan M. Brookhart","is_ca":false},{"name":"Wayne J. Camara","is_ca":false},{"name":"Neil J. Dorans","is_ca":false},{"name":"Kadriye Ercikan","is_ca":false},{"name":"Matthew Gaertner","is_ca":false},{"name":"Joana Pearson","is_ca":false},{"name":"John Hattie","is_ca":false},{"name":"Michael T. Kane","is_ca":false},{"name":"Suzanne Lane","is_ca":false},{"name":"Roy Levy","is_ca":false},{"name":"Scott F. Marion","is_ca":false},{"name":"Joseph A. Martineau","is_ca":false},{"name":"Marianne Perie","is_ca":false},{"name":"Jonathan Templin","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1602917520950539,"gpt":0.4656371866650795,"spread":0.3053454345700256,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.006387936,0.002308324,0.002618073,0.004842978,0.003479362,0.0137021,0.003188154,0.006936585,0.6071768],"category_scores_gemma":[0.03711121,0.001118493,0.001323623,0.00299895,0.001456916,0.005595725,0.00272363,0.005466798,0.6013224],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002407719,"about_ca_system_score_gemma":0.0055227,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002196689,"about_ca_topic_score_gemma":0.00473104,"domain_scores_codex":[0.9943495,0.0006775822,0.0004562396,0.0004512422,0.003581795,0.000483582],"domain_scores_gemma":[0.9642393,0.005511744,0.001734121,0.002026346,0.02050167,0.005986891],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001224638,0.000009700797,0.00001097416,0.0000421254,8.324103e-7,0.000009521917,0.000002734696,0.000006218203,0.00001856611,0.0001140116,0.9943968,0.005376267],"study_design_scores_gemma":[0.00004369307,0.00001836232,0.0001480948,0.0001276075,0.000003119613,0.00001401316,0.00002517606,0.00007845015,0.0001044791,0.0005509796,0.9988769,0.000009130536],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"other","genre_scores_codex":[0.0002362915,0.001313493,0.0007188396,0.04800826,0.6745228,0.0008099844,0.00403157,0.001866807,0.2684919],"genre_scores_gemma":[0.001091646,0.001133964,0.0004884404,0.01066741,0.1147462,0.0003320794,0.00179504,0.001096388,0.8686488],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.3928232,"threshold_uncertainty_score":0.5603147,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4388223267","doi":"10.1111/emip.12582","title":"Comparing Large‐Scale Assessments in Two Proctoring Modalities with Interactive Log Data Analysis","year":2023,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Medical Council of Canada","funders":"","keywords":"Modalities; Comparability; Modality (human–computer interaction); Test (biology); Scale (ratio); Medicine; Computer science; Human–computer interaction; Mathematics","authors":[{"name":"Jinnie Shin","is_ca":false},{"name":"Qi Guo","is_ca":true},{"name":"Maxim Morin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.5444063599822654,"gpt":0.5751607052320205,"spread":0.03075434524975507,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006801779,0.0004476993,0.0004103289,0.002322901,0.0004736032,0.001458186,0.0008274757,0.0005962356,0.001948237],"category_scores_gemma":[0.0578754,0.0001772092,0.0006006366,0.002267651,0.0007188527,0.0008613405,0.001401093,0.0007935964,0.0004936264],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0013621,"about_ca_system_score_gemma":0.001587935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02024706,"about_ca_topic_score_gemma":0.03498317,"domain_scores_codex":[0.9933679,0.003581427,0.0005128057,0.0009035917,0.001344151,0.0002902535],"domain_scores_gemma":[0.9198603,0.0574148,0.009402795,0.004648533,0.006930978,0.00174263],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00257426,0.002281828,0.8246668,0.0005706557,0.0007343006,0.000229991,0.00415935,0.01430251,0.00631761,0.0008249817,0.002362978,0.1409748],"study_design_scores_gemma":[0.00007520902,0.001263998,0.9486916,0.00008317943,0.0001431554,0.0001381984,0.002362645,0.03958053,0.0050226,0.0007554396,0.001811662,0.00007188016],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9886491,0.0001019832,0.00830463,0.0001084675,0.00001754152,0.0003147816,0.001042132,0.0001483019,0.001312908],"genre_scores_gemma":[0.9924785,0.00004076877,0.006231245,0.00003320073,0.0000132665,0.0002129788,0.0006349151,0.00001361715,0.0003416032],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02024706,"threshold_uncertainty_score":0.04025841,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}