{"meta":{"query_hash":"8b186b870421","filters":{"venue":"Educational Assessment Evaluation and Accountability"},"cohort_total":15,"direct_labels_cover":0,"predictions_cover":15,"exported":15,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/8b186b870421","api":"https://metacan.xera.ac/api/v1/cohort?venue=Educational+Assessment+Evaluation+and+Accountability"},"results":[{"id":"W1566806182","doi":"10.1007/s11092-011-9132-4","title":"Learning as identity and practice through involvement in online moderation","year":2011,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Online and Blended Learning","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Queen's University","keywords":"Moderation; Context (archaeology); Psychology; Identity (music); Social psychology; Qualitative research; Element (criminal law); Pedagogy; Sociology; Political science; Social science","score_opus":0.15529069055994224,"score_gpt":0.507281857070541,"score_spread":0.3519911665105987,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1566806182","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80659187,0.00014856418,0.08066887,0.0027417657,0.00015314786,0.0003716443,0.000038518097,0.00041954633,0.10886606],"genre_scores_gemma":[0.98470944,0.000021663149,0.011559604,0.000054498974,0.000012224339,0.00010104238,0.000008719985,0.000025584868,0.003507202],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97863424,0.016450161,0.00053353567,0.0010664616,0.002544239,0.0007713341],"domain_scores_gemma":[0.9337298,0.048157495,0.0041624466,0.006951428,0.0032677634,0.003731015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019917862,0.00036985328,0.00029521988,0.0009310902,0.002038174,0.0063160076,0.00097457203,0.0013492543,0.0069815083],"category_scores_gemma":[0.059321087,0.00023058057,0.00021989047,0.0005040805,0.0034821716,0.0072583575,0.008080579,0.0017689157,0.00066296145],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010071726,0.003298445,0.07494214,0.00042964626,0.000072054136,0.00039923083,0.20169607,0.0020694793,0.02343442,0.15182684,0.0026779594,0.5381466],"study_design_scores_gemma":[0.0007076857,0.005162575,0.15132771,0.0011283652,0.00034659906,0.0017224428,0.16873936,0.06775006,0.084487826,0.37415,0.14413029,0.00034700552],"about_ca_topic_score_codex":0.00049067073,"about_ca_topic_score_gemma":0.0008574149,"teacher_disagreement_score":0.019917862,"about_ca_system_score_codex":0.0013064251,"about_ca_system_score_gemma":0.0023263907,"threshold_uncertainty_score":0.105336964},"labels":[],"label_agreement":null},{"id":"W1948718713","doi":"10.1007/s11092-015-9231-8","title":"Towards a framework for the validation of early childhood assessment systems","year":2015,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Fok Ying Tong Education Foundation; U.S. Department of Education","keywords":"Early childhood education; Early childhood; Accountability; Educational assessment; Conceptual framework; Argument (complex analysis); Field (mathematics); Work (physics); Psychology; Political science; Pedagogy; Developmental psychology; Sociology; Medicine; Social science; Engineering","score_opus":0.09283552875712486,"score_gpt":0.4581947012178403,"score_spread":0.3653591724607154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1948718713","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03009197,0.0011264385,0.92606336,0.014249884,0.00027980132,0.0044520292,0.0005261705,0.0010029047,0.022207433],"genre_scores_gemma":[0.20816661,0.00019230419,0.7869204,0.00064147566,0.000043876888,0.0023985996,0.0006888188,0.000110029745,0.00083786994],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.52707946,0.3621851,0.041031174,0.014786943,0.04909934,0.005817988],"domain_scores_gemma":[0.29853973,0.3679558,0.032285076,0.092106745,0.2023206,0.0067920885],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4722507,0.0019047391,0.0025962195,0.012644228,0.006790054,0.026677597,0.011354395,0.008099965,0.0020045405],"category_scores_gemma":[0.54499376,0.0018917207,0.0026736022,0.0063680555,0.017558852,0.017559089,0.015786318,0.010718798,0.00075057324],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023013669,0.0009874721,0.046974037,0.001352689,0.00045728942,0.0002677286,0.025264021,0.017607605,0.0025426955,0.72067267,0.0062877783,0.17735592],"study_design_scores_gemma":[0.00036305428,0.0011589534,0.04934523,0.011006335,0.0005562817,0.00047532681,0.025136884,0.13980833,0.013190852,0.67129487,0.08713847,0.00052544125],"about_ca_topic_score_codex":0.04656231,"about_ca_topic_score_gemma":0.027734244,"teacher_disagreement_score":0.4722507,"about_ca_system_score_codex":0.024858389,"about_ca_system_score_gemma":0.06657612,"threshold_uncertainty_score":0.6508089},"labels":[],"label_agreement":null},{"id":"W1983949565","doi":"10.1007/s11092-008-9063-x","title":"Using data to support educational improvement","year":2008,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":127,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Ministry of Education and Child Care","funders":"","keywords":"Data collection; Context (archaeology); Computer science; Scale (ratio); Data science; Knowledge management; Sociology","score_opus":0.5729116068459882,"score_gpt":0.5914743349389096,"score_spread":0.018562728092921366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1983949565","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15726013,0.0065271393,0.3848421,0.11587481,0.004841809,0.005095218,0.08617444,0.006458748,0.2329256],"genre_scores_gemma":[0.6995129,0.0025443512,0.24673763,0.003679458,0.0005401695,0.001826631,0.038110685,0.00037034092,0.006677733],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.8743509,0.0764405,0.017380342,0.0050997343,0.025032409,0.0016961892],"domain_scores_gemma":[0.35269272,0.41511264,0.04712729,0.079839125,0.10048305,0.0047451267],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1265374,0.0012774221,0.0014800598,0.02060845,0.0015610838,0.012063716,0.0030932012,0.0024997569,0.00590054],"category_scores_gemma":[0.3885965,0.0007616498,0.00090134755,0.016906574,0.0016974784,0.015576547,0.004697712,0.0036426866,0.003100257],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001007604,0.0010242618,0.25061905,0.0026493121,0.0007679301,0.00040911886,0.0041377787,0.009264724,0.0019594678,0.09824534,0.086378634,0.5435368],"study_design_scores_gemma":[0.000541212,0.0010818178,0.09669526,0.009935837,0.0012045334,0.000494106,0.009884036,0.06922976,0.024252918,0.24015984,0.54603225,0.00048852013],"about_ca_topic_score_codex":0.012531377,"about_ca_topic_score_gemma":0.008790365,"teacher_disagreement_score":0.1265374,"about_ca_system_score_codex":0.005038702,"about_ca_system_score_gemma":0.013022959,"threshold_uncertainty_score":0.66920173},"labels":[],"label_agreement":null},{"id":"W2000242929","doi":"10.1007/s11092-014-9195-0","title":"Navigating dilemmas in transforming assessment practices: experiences of mathematics teachers in Ontario, Canada","year":2014,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Manitoba; University of Ottawa","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Situated; Accountability; Judgement; Pedagogy; Assessment for learning; Curriculum; Best practice; Mathematics education; Faculty development; Professional development; Psychology; Formative assessment; Political science; Computer science","score_opus":0.0639113308409666,"score_gpt":0.4511814897350099,"score_spread":0.38727015889404326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2000242929","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96347135,0.0006880422,0.00056971825,0.016314968,0.00012096397,0.00015913216,0.00012218926,0.000033529417,0.018520137],"genre_scores_gemma":[0.981501,0.00046888995,0.0007866142,0.0014511197,0.000017132876,0.00003706638,0.000045527035,0.000038961363,0.015653748],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.986188,0.004560778,0.0004456162,0.00081982894,0.0033544935,0.0046312725],"domain_scores_gemma":[0.9657706,0.006701896,0.0022727014,0.00054160575,0.010554418,0.01415869],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007283135,0.00038643565,0.00059162657,0.0011619146,0.04574479,0.011248453,0.0031051964,0.0030321274,0.0036721984],"category_scores_gemma":[0.019882271,0.00079895335,0.00041944935,0.0031912997,0.012098179,0.0027212366,0.007363375,0.004539494,0.0004474468],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000072787705,0.00010285064,0.026113959,0.000086024054,0.000011776628,0.0012046669,0.9481727,0.0001998083,0.0005432122,0.001508758,0.006378561,0.015604891],"study_design_scores_gemma":[0.0000116285255,0.00004664168,0.022897903,0.000120703655,0.00001039913,0.00011146795,0.9361263,0.00017533194,0.00015424634,0.0003128931,0.039988443,0.000044045264],"about_ca_topic_score_codex":0.9924852,"about_ca_topic_score_gemma":0.99853444,"teacher_disagreement_score":0.14255129,"about_ca_system_score_codex":0.14255129,"about_ca_system_score_gemma":0.29116982,"threshold_uncertainty_score":0.994519},"labels":[],"label_agreement":null},{"id":"W2040697633","doi":"10.1007/s11092-011-9117-3","title":"Consistency of report card grades and external assessments in a Canadian province","year":2011,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":21,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Trent University; University of Toronto","funders":"Ministère de l’Éducation, Gouvernement de l’Ontario","keywords":"Report card; Psychology; Consistency (knowledge bases); Reading (process); Medical education; Grade level; Variance (accounting); Mathematics education; Medicine; Pedagogy; Mathematics; Political science","score_opus":0.2160058372944961,"score_gpt":0.5025141811864667,"score_spread":0.28650834389197066,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2040697633","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9832392,0.0010155961,0.0007582012,0.0015426704,0.000085403124,0.00012462348,0.004506433,0.00006131162,0.0086666215],"genre_scores_gemma":[0.99607193,0.00025857703,0.0006597526,0.00009862695,0.000007314749,0.000031899883,0.0011389353,0.000020887042,0.0017119633],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9836595,0.0027391354,0.0012976436,0.0020018967,0.0075309128,0.0027709317],"domain_scores_gemma":[0.93075633,0.009845869,0.0064142817,0.0021887398,0.046455078,0.0043396773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012289216,0.00046598396,0.0009505978,0.005393516,0.0062556313,0.0036513344,0.004582487,0.0006719204,0.0018050263],"category_scores_gemma":[0.0520658,0.0008194342,0.00079404603,0.012950483,0.0023439887,0.0008072828,0.0020938085,0.0012525569,0.00022774757],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000245407,0.000077056764,0.9651389,0.00011619567,0.00014080448,0.00012243833,0.007284437,0.0009902563,0.00026373487,0.001016723,0.0056087296,0.018995322],"study_design_scores_gemma":[0.000012878048,0.000026349251,0.9929721,0.000071367234,0.000032576012,0.000029787338,0.0034395037,0.0007620092,0.00017194444,0.000062272484,0.0023906387,0.000028547694],"about_ca_topic_score_codex":0.9981608,"about_ca_topic_score_gemma":0.9990024,"teacher_disagreement_score":0.09960012,"about_ca_system_score_codex":0.09960012,"about_ca_system_score_gemma":0.1474368,"threshold_uncertainty_score":0.72265285},"labels":[],"label_agreement":null},{"id":"W2043024755","doi":"10.1007/s11092-013-9184-8","title":"The journey from rhetoric to reality: participatory evaluation in a development context","year":2014,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":34,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Citizen journalism; Context (archaeology); Situated; Stakeholder; Sociology; Warrant; Rhetoric; Participatory action research; International development; Participatory evaluation; Engineering ethics; Conceptual framework; Epistemology; Political science; Social science; Public relations; Computer science; Engineering; Geography; Business","score_opus":0.3829268271012413,"score_gpt":0.5696211298539817,"score_spread":0.1866943027527404,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2043024755","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.060630843,0.013727752,0.2608605,0.42299217,0.0018762929,0.00079429446,0.00012581714,0.00026521372,0.23872715],"genre_scores_gemma":[0.90266097,0.002981714,0.07620222,0.007636959,0.0003636835,0.0010677914,0.00006596203,0.00029154256,0.008729178],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.68190396,0.2908859,0.0031592331,0.005220538,0.014929382,0.003900976],"domain_scores_gemma":[0.5927063,0.36731043,0.0067076352,0.012628485,0.013773298,0.0068738535],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.21285865,0.0013379654,0.0019579476,0.005378884,0.024142364,0.041200057,0.00622024,0.012537497,0.0061710766],"category_scores_gemma":[0.23478156,0.0014232628,0.0007971605,0.0040734927,0.10409184,0.038177695,0.028695235,0.014432869,0.0008267542],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005363447,0.0001348157,0.0008712578,0.00046829515,0.00002368906,0.00027338704,0.14245221,0.00056946045,0.00034144614,0.8071725,0.005611359,0.042027943],"study_design_scores_gemma":[0.000055050583,0.0001221692,0.0005106848,0.0019545422,0.000021775284,0.00024260936,0.09190573,0.0014484435,0.000873724,0.8034388,0.09935386,0.00007253571],"about_ca_topic_score_codex":0.004843929,"about_ca_topic_score_gemma":0.0060217804,"teacher_disagreement_score":0.21285865,"about_ca_system_score_codex":0.01323755,"about_ca_system_score_gemma":0.034811582,"threshold_uncertainty_score":0.9706854},"labels":[],"label_agreement":null},{"id":"W2076487748","doi":"10.1007/s11092-014-9200-7","title":"Personality and student performance on evaluation methods used in business administration courses","year":2014,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Management and Marketing Education","field":"Business, Management and Accounting","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval; Université de Sherbrooke","funders":"","keywords":"Personality; Preference; Psychology; Sample (material); Multilevel model; Latent variable; Control (management); Applied psychology; Medical education; Social psychology; Computer science; Machine learning; Medicine; Artificial intelligence; Statistics; Mathematics","score_opus":0.06513401286280288,"score_gpt":0.446581531766921,"score_spread":0.3814475189041181,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2076487748","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9979905,0.00006427718,0.00015410033,0.000054661163,0.000026734655,0.0000194966,0.000026761592,0.0000087714625,0.0016546984],"genre_scores_gemma":[0.99863833,0.000044506665,0.00017051291,0.000029899631,0.0000104813735,0.000015796682,0.00004783537,0.0000074897002,0.0010351313],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99275595,0.003243392,0.0006458294,0.0002791351,0.0024125644,0.00066315226],"domain_scores_gemma":[0.91887045,0.035099193,0.015347066,0.0032442675,0.015480501,0.011958577],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010946452,0.00042130114,0.0004970817,0.0017796787,0.0008424236,0.0023255078,0.00034220048,0.0006738655,0.0016175283],"category_scores_gemma":[0.06828575,0.00023231418,0.00083193916,0.0011125067,0.00057051197,0.0009391606,0.0010830949,0.001485537,0.00092680944],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011077943,0.0022725088,0.96500844,0.00003067543,0.00015054748,0.00006301156,0.0018618298,0.0003730993,0.0009071567,0.00013659676,0.00055755454,0.027530832],"study_design_scores_gemma":[0.000024177394,0.0017631432,0.9949964,0.00002566217,0.00002915683,0.000050911065,0.0011630239,0.00083086814,0.00061694876,0.00011315901,0.00036982383,0.000016713078],"about_ca_topic_score_codex":0.0026578433,"about_ca_topic_score_gemma":0.004063396,"teacher_disagreement_score":0.010946452,"about_ca_system_score_codex":0.0010081375,"about_ca_system_score_gemma":0.0009520851,"threshold_uncertainty_score":0.05789107},"labels":[],"label_agreement":null},{"id":"W2088210129","doi":"10.1007/s11092-013-9181-y","title":"Teacher supervision practices and characteristics of in-school supervisors in Uganda","year":2014,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Teacher Education and Leadership Studies","field":"Social Sciences","cited_by":14,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Supervisor; Perception; Professional development; Psychology; Directive; Descriptive statistics; Medical education; Pedagogy; Medicine; Political science","score_opus":0.13688318478546668,"score_gpt":0.4800151790265387,"score_spread":0.343131994241072,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2088210129","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9996437,0.00005264014,0.00000955391,0.000022913588,0.0000016279909,0.0000027440599,0.000015048609,5.49345e-7,0.00025121256],"genre_scores_gemma":[0.9997011,0.000065407854,0.000019700905,0.000010342848,8.824424e-7,0.000004630238,0.000014757602,6.507361e-7,0.0001825912],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99915266,0.00028946888,0.00010151423,0.00006568907,0.0001077892,0.0002829446],"domain_scores_gemma":[0.99578285,0.00062661414,0.0017191689,0.0001070392,0.00062388135,0.001140406],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00075606065,0.00016832806,0.00024505876,0.00083449605,0.001735137,0.0010149915,0.00046762908,0.0003865899,0.0014749679],"category_scores_gemma":[0.005033321,0.00042160504,0.00013808874,0.0010780565,0.0006006208,0.0004950624,0.0009167971,0.0006391199,0.00022149156],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004909866,0.00009358252,0.98469156,0.000019912444,0.000007719557,0.00023290167,0.011764102,0.000027077489,0.00025750353,0.00003330391,0.00013537778,0.0026878433],"study_design_scores_gemma":[0.000002880435,0.00013802356,0.9689626,0.00003960041,0.0000055105425,0.00042911625,0.029737314,0.00007101626,0.00010980557,0.000019049348,0.0004795891,0.000005542598],"about_ca_topic_score_codex":0.013407205,"about_ca_topic_score_gemma":0.028541468,"teacher_disagreement_score":0.013407205,"about_ca_system_score_codex":0.0009705395,"about_ca_system_score_gemma":0.0010089043,"threshold_uncertainty_score":0.026658297},"labels":[],"label_agreement":null},{"id":"W2183435816","doi":"10.1007/s11092-015-9233-6","title":"Teacher assessment literacy: a review of international standards and measures","year":2015,"lang":"en","type":"review","venue":"Educational Assessment Evaluation and Accountability","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":253,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Literacy; Educational assessment; Mainland; Psychology; Pedagogy; Geography","score_opus":0.13742489297286617,"score_gpt":0.5795117938372106,"score_spread":0.44208690086434443,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2183435816","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00024926834,0.9972396,0.0005561354,0.0008709863,0.00016866504,0.00007580891,0.00021362817,0.000011542816,0.0006142973],"genre_scores_gemma":[0.0030696886,0.99308175,0.002676064,0.00050985155,0.00011715362,0.0001618511,0.00025237413,0.000008908911,0.00012238573],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9841541,0.003658684,0.006487186,0.0011018958,0.004334907,0.00026318192],"domain_scores_gemma":[0.9449297,0.036479093,0.007558456,0.00096644164,0.009522762,0.0005435169],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025188955,0.0012994491,0.0053110137,0.017954974,0.00084608834,0.003551908,0.0033269657,0.002146365,0.0025024132],"category_scores_gemma":[0.06414508,0.0008883267,0.0025402955,0.015681155,0.0023816095,0.0041477755,0.0028436505,0.0024254657,0.00062873133],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001248835,0.000075461816,0.0021587333,0.15369926,0.0005238173,0.00005114418,0.00046485334,0.00024195107,0.00025422088,0.0044345963,0.013288813,0.8246823],"study_design_scores_gemma":[0.000109902474,0.00023915106,0.014817286,0.45674878,0.005636623,0.00071982417,0.0010867497,0.00033010874,0.0010776788,0.0052996716,0.5137883,0.00014591766],"about_ca_topic_score_codex":0.010991415,"about_ca_topic_score_gemma":0.01972708,"teacher_disagreement_score":0.025188955,"about_ca_system_score_codex":0.0047452627,"about_ca_system_score_gemma":0.02158045,"threshold_uncertainty_score":0.13321352},"labels":[],"label_agreement":null},{"id":"W2531903003","doi":"10.1007/s11092-016-9247-8","title":"Two teacher quality measures and the role of context: evidence from Chile","year":2016,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"School Choice and Performance","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Institute for Christian Studies","funders":"","keywords":"Context (archaeology); Quality (philosophy); Teacher quality; Psychology; Geography; Political science; Business; Archaeology; Epistemology; Philosophy; Marketing","score_opus":0.09023241200854629,"score_gpt":0.47158391345562656,"score_spread":0.3813515014470803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2531903003","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98156,0.0049096458,0.00084375485,0.0018544153,0.000026903588,0.00008761167,0.00081782026,0.00001010799,0.009889748],"genre_scores_gemma":[0.9973686,0.0012128651,0.0005327332,0.00011521064,0.000011426158,0.00008438618,0.00028274095,0.000008449889,0.00038354978],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9832908,0.009827089,0.0012563118,0.0009768307,0.003689797,0.0009592554],"domain_scores_gemma":[0.8753725,0.07442695,0.028162634,0.005662463,0.014110297,0.002265082],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020633128,0.00033806983,0.0010588539,0.0042963396,0.0014724237,0.0031875717,0.0013166016,0.00076042704,0.0019085021],"category_scores_gemma":[0.071642525,0.00037553522,0.00053033716,0.0104229115,0.0029943117,0.0022865497,0.0042687757,0.0013671715,0.00012738293],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005809279,0.00052906695,0.92725366,0.0011811387,0.00038582392,0.00014265062,0.009435355,0.0004818239,0.0002022862,0.0053730793,0.0010915463,0.053342674],"study_design_scores_gemma":[0.00007384067,0.00012334797,0.9858066,0.0009863616,0.0001989677,0.000044436147,0.007475338,0.00034812352,0.00033133832,0.0009609185,0.0036201656,0.000030539788],"about_ca_topic_score_codex":0.13084954,"about_ca_topic_score_gemma":0.16133438,"teacher_disagreement_score":0.13084954,"about_ca_system_score_codex":0.009879809,"about_ca_system_score_gemma":0.01061742,"threshold_uncertainty_score":0.26017582},"labels":[],"label_agreement":null},{"id":"W3044087203","doi":"10.1007/s11092-020-09329-5","title":"Is Canada really an education superpower? The impact of non-participation on results from PISA 2015","year":2020,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Youth Substance Use and School Attendance","field":"Social Sciences","cited_by":31,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"H2020 European Research Council","keywords":"Superpower; Political science; Inclusion (mineral); Scale (ratio); Test (biology); Sample (material); Population; Economic growth; Psychology; Geography; Mathematics education; Demography; Sociology; China; Social psychology; Economics","score_opus":0.07773513248884605,"score_gpt":0.4788059318311023,"score_spread":0.40107079934225626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3044087203","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7292872,0.0050936243,0.004773048,0.08468408,0.0021268283,0.0014341052,0.03120109,0.0003933156,0.14100657],"genre_scores_gemma":[0.98347557,0.00079308683,0.0012639671,0.0042928937,0.00010981551,0.00034881345,0.003580794,0.00012837804,0.006006598],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.8601153,0.030797893,0.007887066,0.0048123733,0.07451155,0.021875802],"domain_scores_gemma":[0.7968855,0.047635302,0.024315618,0.013007878,0.09978079,0.018374942],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06781377,0.0005168097,0.0011972716,0.0053133243,0.0076478818,0.008260973,0.0040860414,0.0010103317,0.0062615867],"category_scores_gemma":[0.2042917,0.0004859334,0.0013011321,0.01166339,0.004322933,0.00198788,0.005768524,0.0029196092,0.00080017967],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007667745,0.00015089978,0.82223034,0.00067723915,0.00037789275,0.0003022551,0.012807755,0.0007516939,0.00026124666,0.010090867,0.058583733,0.09299927],"study_design_scores_gemma":[0.000028417766,0.000121063276,0.9497539,0.00107187,0.00012607669,0.00010495584,0.014154812,0.0007296338,0.00075226376,0.0006354003,0.032434966,0.00008657062],"about_ca_topic_score_codex":0.9778021,"about_ca_topic_score_gemma":0.9831933,"teacher_disagreement_score":0.9396161,"about_ca_system_score_codex":0.06038393,"about_ca_system_score_gemma":0.13473672,"threshold_uncertainty_score":0.43811816},"labels":[],"label_agreement":null},{"id":"W3188740232","doi":"10.1007/s11092-021-09368-6","title":"“I am here for the students”: principals’ perception of accountability amid work intensification","year":2021,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Teacher Education and Leadership Studies","field":"Social Sciences","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University; University of Manitoba; University of British Columbia","funders":"","keywords":"Accountability; Perception; Work (physics); Context (archaeology); Public relations; Psychology; Political science; Focus group; Task (project management); Pedagogy; Sociology; Management; Law","score_opus":0.2359750747103962,"score_gpt":0.5289763041093049,"score_spread":0.29300122939890866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3188740232","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99533045,0.00013875098,0.00043793677,0.0013517333,0.000046989266,0.00001741834,0.000010844978,0.000012033525,0.002653856],"genre_scores_gemma":[0.9994326,0.00007083415,0.000102890146,0.00010517844,0.00001305737,0.0000070529295,0.0000063355087,0.0000024544697,0.00025953187],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.98746175,0.007975736,0.0007153755,0.00046372824,0.0021062281,0.0012771332],"domain_scores_gemma":[0.97091204,0.007332267,0.01003209,0.0010634564,0.0043737227,0.006286333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010554666,0.00019964782,0.00040146802,0.0009356149,0.0034663577,0.0042780014,0.0005307032,0.0009026879,0.0037682967],"category_scores_gemma":[0.025284484,0.0003725196,0.0004653601,0.00062747026,0.0034017637,0.001998652,0.003105927,0.0022949642,0.000428852],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023580654,0.00039689857,0.62041235,0.00023890717,0.000079053876,0.0009971369,0.33766812,0.00012372895,0.0021887175,0.0020832224,0.0038560976,0.031719938],"study_design_scores_gemma":[0.000023361425,0.0004391745,0.31177604,0.00018647824,0.000027075883,0.0006420399,0.67641485,0.00042293963,0.0003791617,0.0008862457,0.008745939,0.000056774734],"about_ca_topic_score_codex":0.004152632,"about_ca_topic_score_gemma":0.005234713,"teacher_disagreement_score":0.010554666,"about_ca_system_score_codex":0.0015827761,"about_ca_system_score_gemma":0.0026308568,"threshold_uncertainty_score":0.055819094},"labels":[],"label_agreement":null},{"id":"W4283698056","doi":"10.1007/s11092-022-09389-9","title":"Mapping the constellation of assessment discourses: a scoping review study on assessment competence, literacy, capability, and identity","year":2022,"lang":"en","type":"review","venue":"Educational Assessment Evaluation and Accountability","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Memorial University of Newfoundland","funders":"","keywords":"Competence (human resources); Operationalization; Construct (python library); Authentic assessment; Literacy; Rubric; Pedagogy; Identity (music); Psychology; Engineering ethics; Sociology; Epistemology; Social psychology; Computer science; Engineering","score_opus":0.19602477827829456,"score_gpt":0.5663204934521253,"score_spread":0.37029571517383075,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283698056","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0054192077,0.9897786,0.0010793258,0.0012496868,0.00012762459,0.00029833784,0.00018248161,0.000008221212,0.0018565763],"genre_scores_gemma":[0.06338696,0.9300707,0.0045888233,0.0006039523,0.000064603264,0.00074389996,0.00026453758,0.000014243493,0.00026230505],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.9669871,0.017885856,0.0073160096,0.0019631896,0.005394739,0.00045322478],"domain_scores_gemma":[0.8583534,0.11983683,0.009114513,0.002287988,0.009898981,0.0005082685],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043165293,0.0009562565,0.0039463607,0.023390202,0.0016378141,0.00789762,0.0015738702,0.0019117075,0.0018921107],"category_scores_gemma":[0.14486739,0.0011266582,0.0029445621,0.025607869,0.0032300702,0.008579675,0.005092536,0.0025264625,0.00027812933],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019800241,0.000075085605,0.0056491857,0.38395298,0.0037502032,0.00027378026,0.024586918,0.00026657633,0.0005545319,0.0111287525,0.0037590456,0.565805],"study_design_scores_gemma":[0.00005391262,0.00010628905,0.012244143,0.8786181,0.009775531,0.00064417144,0.020021072,0.00017900727,0.00052959507,0.004332902,0.073411845,0.00008352069],"about_ca_topic_score_codex":0.01214686,"about_ca_topic_score_gemma":0.03306799,"teacher_disagreement_score":0.043165293,"about_ca_system_score_codex":0.0053409,"about_ca_system_score_gemma":0.025122361,"threshold_uncertainty_score":0.22828263},"labels":[],"label_agreement":null},{"id":"W4417293613","doi":"10.1007/s11092-025-09471-y","title":"Investigating the relationship between educational inequity and teacher participation in professional development: A cross-national and quasi-experimental approach using TIMSS","year":2025,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"Göteborgs Universitet; Vetenskapsrådet","keywords":"Socioeconomic status; Student achievement; Academic achievement; Professional development; Higher education; Social class","score_opus":0.18038049535966852,"score_gpt":0.5374982171340111,"score_spread":0.35711772177434264,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417293613","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9873881,0.000067035806,0.0042344327,0.000120287026,0.00008941411,0.005728747,0.00044310733,0.000025507723,0.0019034179],"genre_scores_gemma":[0.9415782,0.000058633927,0.0105419215,0.00032658654,0.00008249011,0.04401357,0.00044477716,0.0000146047605,0.0029391237],"study_design_codex":"nonrandomized_trial","study_design_gemma":"nonrandomized_trial","domain_scores_codex":[0.97583234,0.01841467,0.0011084973,0.0019579283,0.0014412177,0.0012454075],"domain_scores_gemma":[0.9679355,0.01914509,0.005304058,0.004882557,0.0018735154,0.00085926434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025495518,0.0008111792,0.0011213799,0.0010014389,0.002435577,0.0014749476,0.0014779803,0.0015734709,0.008882154],"category_scores_gemma":[0.02370867,0.0010177366,0.001602925,0.0008689546,0.003313287,0.0015722257,0.0017967632,0.0023885043,0.00087998854],"study_design_candidate":"nonrandomized_trial","study_design_consensus":"nonrandomized_trial","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.04677794,0.38082,0.37896594,0.003503022,0.0030804202,0.0007593093,0.039041527,0.005187574,0.018670745,0.022536997,0.0063930466,0.09426348],"study_design_scores_gemma":[0.010024149,0.2496843,0.6666707,0.000569504,0.0015344656,0.0001919093,0.019583954,0.012931661,0.012322541,0.008862029,0.017412513,0.00021218533],"about_ca_topic_score_codex":0.0054497016,"about_ca_topic_score_gemma":0.0068652406,"teacher_disagreement_score":0.025495518,"about_ca_system_score_codex":0.0017789988,"about_ca_system_score_gemma":0.00280144,"threshold_uncertainty_score":0.13483477},"labels":[],"label_agreement":null},{"id":"W944434266","doi":"10.1007/s11092-015-9224-7","title":"Juggling multiple accountability systems: how three principals manage these tensions in Ontario, Canada","year":2015,"lang":"en","type":"article","venue":"Educational Assessment Evaluation and Accountability","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":31,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"York University; Western University","funders":"","keywords":"Accountability; Negotiation; Mandate; Variety (cybernetics); Public relations; Work (physics); Public administration; Political science; Sociology; Engineering; Law","score_opus":0.2865166200978165,"score_gpt":0.45404375441143446,"score_spread":0.167527134313618,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W944434266","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8804682,0.001075571,0.0024579517,0.071768016,0.0004692662,0.0006283903,0.00025063189,0.00017772021,0.042704187],"genre_scores_gemma":[0.95228964,0.00047306696,0.0031846715,0.004134033,0.000049266175,0.000115510455,0.00008629143,0.00007522511,0.039592225],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.97991985,0.004774053,0.00068323454,0.0011514424,0.004873044,0.008598349],"domain_scores_gemma":[0.9412913,0.005984603,0.0022764958,0.0011207238,0.02197374,0.027353115],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010790241,0.00037760372,0.0004936897,0.0012435787,0.052788075,0.010438528,0.004351851,0.003860218,0.0060812617],"category_scores_gemma":[0.025304519,0.0012365356,0.00054020464,0.0026386708,0.008097548,0.003014042,0.0064134584,0.005285263,0.0006234656],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00069957797,0.0010335376,0.24573016,0.00042585965,0.000138556,0.0053577847,0.41045418,0.003851916,0.0027150223,0.03103493,0.1396137,0.15894477],"study_design_scores_gemma":[0.00017568073,0.0003831287,0.2243507,0.0003962144,0.000116437506,0.0003578246,0.519703,0.003717955,0.0008901486,0.0033567986,0.2462268,0.0003252937],"about_ca_topic_score_codex":0.9958521,"about_ca_topic_score_gemma":0.9991947,"teacher_disagreement_score":0.24705316,"about_ca_system_score_codex":0.24705316,"about_ca_system_score_gemma":0.4548379,"threshold_uncertainty_score":0.87331164},"labels":[],"label_agreement":null}]}