{"meta":{"query_hash":"9bc908a09f7e","filters":{"venue":"ITC 2016 Conference"},"cohort_total":23,"direct_labels_cover":0,"predictions_cover":23,"exported":23,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/9bc908a09f7e","api":"https://metacan.xera.ac/api/v1/cohort?venue=ITC+2016+Conference"},"results":[{"id":"W2467740042","doi":"","title":"PAPER: The Power and Potential of Psychological Assessment Around the World -- Rethinking Traditional Paradigms","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Presentation (obstetrics); Sine qua non; Transformative learning; Psychology; Process (computing); Public relations; Engineering ethics; Power (physics); Political science; Computer science; Engineering; Pedagogy; Medicine","score_opus":0.10710727184053166,"score_gpt":0.38994936403258673,"score_spread":0.2828420921920551,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2467740042","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005213079,0.064959995,0.019815302,0.6243642,0.049119398,0.00011759261,0.00017682294,0.00016576133,0.23606789],"genre_scores_gemma":[0.36696023,0.12791796,0.035215463,0.18991387,0.09729302,0.00085174013,0.00033032903,0.0012032184,0.18031426],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98012656,0.01376995,0.000484695,0.0016189322,0.0036703984,0.0003294606],"domain_scores_gemma":[0.94771934,0.041532617,0.0012430293,0.0029234237,0.00467558,0.0019059355],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02185955,0.0005626863,0.00062299985,0.0017204997,0.0050058933,0.023981938,0.0023183068,0.004554963,0.011520806],"category_scores_gemma":[0.05146719,0.00031528928,0.0005862617,0.002060898,0.02525589,0.028292935,0.006628664,0.011360482,0.003223978],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000060087346,0.000052981104,0.0009079922,0.00062652596,0.000019152229,0.00012554083,0.019102681,0.00020438593,0.00014291726,0.5953453,0.26173693,0.12167542],"study_design_scores_gemma":[0.000020974468,0.00004943521,0.0009533268,0.0020065696,0.000014944557,0.0003212551,0.012173746,0.00032026396,0.00016477588,0.32725048,0.656694,0.000030274752],"about_ca_topic_score_codex":0.0016561742,"about_ca_topic_score_gemma":0.0017906869,"teacher_disagreement_score":0.9781405,"about_ca_system_score_codex":0.004542945,"about_ca_system_score_gemma":0.0066289906,"threshold_uncertainty_score":0.11560577},"labels":[],"label_agreement":null},{"id":"W2477167192","doi":"","title":"POSTER: Bayesian Estimation of the Polychoric Correlation Coefficient with Skewed and Sparse Data","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Advanced Statistical Modeling Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Polychoric correlation; Statistics; Skewness; Mathematics; Econometrics; Bayesian probability; Contingency table; Correlation; Sample size determination; Population; Data set","score_opus":0.0324205745307385,"score_gpt":0.2770568380063151,"score_spread":0.2446362634755766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2477167192","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.013183382,0.00043796186,0.9792291,0.0012267106,0.00016539915,0.00018436377,0.0006599922,0.00057610637,0.0043370263],"genre_scores_gemma":[0.3127859,0.0007533518,0.672699,0.0008224392,0.00035016288,0.00091120414,0.00260893,0.0005155388,0.008553504],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99060446,0.006525742,0.00028682078,0.0013555129,0.0010057584,0.00022173092],"domain_scores_gemma":[0.94830155,0.03994247,0.0021828369,0.005155506,0.0038128106,0.00060483796],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018371986,0.0009788073,0.0014949114,0.0019886636,0.0013246213,0.0026491813,0.0019912159,0.002281104,0.01750462],"category_scores_gemma":[0.11614123,0.0011393391,0.0015255669,0.002167234,0.0017043345,0.0026193038,0.0025474448,0.004040886,0.005240586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009276302,0.0002782961,0.029943962,0.0007944622,0.0005990376,0.0006475967,0.0013968332,0.23365349,0.0023020797,0.20376055,0.06781609,0.45788005],"study_design_scores_gemma":[0.0001621877,0.00016664303,0.009756869,0.00046878145,0.00009463817,0.0006199711,0.00022693425,0.7047886,0.0016096924,0.2667263,0.015241195,0.00013814085],"about_ca_topic_score_codex":0.006123138,"about_ca_topic_score_gemma":0.007107009,"teacher_disagreement_score":0.018371986,"about_ca_system_score_codex":0.001105287,"about_ca_system_score_gemma":0.0021203381,"threshold_uncertainty_score":0.09716147},"labels":[],"label_agreement":null},{"id":"W2490078899","doi":"","title":"SYMPOSIUM: The Validity of Testing: Perceptions of Stakeholders","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Perception; Psychology; Computer science","score_opus":0.24014471523629496,"score_gpt":0.3671960779230274,"score_spread":0.12705136268673242,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2490078899","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.047273368,0.008075241,0.0064153215,0.84519565,0.027440961,0.00042701827,0.0003108583,0.00045391705,0.064407766],"genre_scores_gemma":[0.6906788,0.009202149,0.0066901455,0.19435886,0.008932988,0.0007973705,0.0005360506,0.0007772507,0.08802649],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.98293394,0.008517014,0.0005338627,0.000969908,0.004974648,0.0020706689],"domain_scores_gemma":[0.9577552,0.01369398,0.0016627708,0.0008492707,0.014486316,0.0115524065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.024163878,0.000670896,0.0006387438,0.00087911024,0.0068294182,0.010972998,0.0018298911,0.0065302844,0.026931148],"category_scores_gemma":[0.05266505,0.00065192865,0.0006659314,0.0009826032,0.0050886893,0.009455554,0.007535564,0.011558564,0.004320327],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00011296433,0.00020392479,0.010509415,0.00059539406,0.000020038515,0.0006390756,0.119962186,0.00024660595,0.0019997777,0.016946625,0.7788451,0.06991887],"study_design_scores_gemma":[0.000016889115,0.00022543572,0.008976918,0.0010443963,0.0000138498235,0.0004788177,0.16776992,0.00054534303,0.0006864922,0.008341879,0.81178784,0.000112171125],"about_ca_topic_score_codex":0.0052990746,"about_ca_topic_score_gemma":0.0028802825,"teacher_disagreement_score":0.026931148,"about_ca_system_score_codex":0.0042429003,"about_ca_system_score_gemma":0.010116631,"threshold_uncertainty_score":0.1277923},"labels":[],"label_agreement":null},{"id":"W2498349618","doi":"","title":"POSTER: Detecting Differential Item Functioning: A Comparison of Two Effect Size Measures in Logistic Regression Analysis","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Mental Health Research Topics","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Differential item functioning; Juvenile delinquency; Psychology; Context (archaeology); Logistic regression; Neighbourhood (mathematics); Odds; Scale (ratio); Statistics; Social psychology; Item response theory; Developmental psychology; Psychometrics; Mathematics; Geography","score_opus":0.16116164982751477,"score_gpt":0.4685125216579524,"score_spread":0.30735087183043763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2498349618","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6346529,0.009071171,0.31586796,0.0065994584,0.0041925395,0.0044042105,0.003868962,0.0022035097,0.019139377],"genre_scores_gemma":[0.87009794,0.0008138188,0.12052878,0.0008217131,0.0005348427,0.0040606824,0.001322872,0.00037161043,0.0014478021],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.7628817,0.20332843,0.00939176,0.006725649,0.016764602,0.00090790674],"domain_scores_gemma":[0.35339746,0.59196305,0.02156185,0.0149120195,0.016680602,0.0014849814],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1560573,0.0012484726,0.0020502107,0.005649393,0.0011672621,0.0037890864,0.0020308488,0.002747984,0.0077755824],"category_scores_gemma":[0.45622602,0.0006812136,0.0043123453,0.004049312,0.0025744555,0.0046065864,0.0034678255,0.0025733623,0.0015025216],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017077032,0.0010013062,0.40566805,0.0042474344,0.008240874,0.00036700122,0.005268249,0.0050632716,0.0053957547,0.010269444,0.018478503,0.51892304],"study_design_scores_gemma":[0.00237159,0.019106198,0.8332058,0.0035134898,0.0053674504,0.002469784,0.0048510716,0.058220606,0.013593487,0.029485406,0.026972292,0.00084279577],"about_ca_topic_score_codex":0.00041727876,"about_ca_topic_score_gemma":0.00046797533,"teacher_disagreement_score":0.1560573,"about_ca_system_score_codex":0.0010169772,"about_ca_system_score_gemma":0.00081042945,"threshold_uncertainty_score":0.82531977},"labels":[],"label_agreement":null},{"id":"W2546985950","doi":"","title":"PAPER: Making Sense of the Gender Gap in Reading on the PISA 2009","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"School Choice and Performance","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Gender gap; Reading (process); Psychology; Socioeconomic status; Set (abstract data type); Latent class model; Developmental psychology; Scale (ratio); Demography; Statistics; Geography; Mathematics; Population; Computer science; Political science","score_opus":0.1094491655318233,"score_gpt":0.3346620453961246,"score_spread":0.2252128798643013,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2546985950","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39532667,0.01414818,0.0047105965,0.35189116,0.039604764,0.001180588,0.016408809,0.00042967312,0.17629956],"genre_scores_gemma":[0.87603855,0.0070214323,0.006960748,0.022541525,0.006054187,0.0009161068,0.0073904186,0.0005121412,0.07256487],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.9945695,0.0013088904,0.000263493,0.0010239178,0.0024126424,0.00042164256],"domain_scores_gemma":[0.95842665,0.017623395,0.003990651,0.002036226,0.015689237,0.0022338673],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011222553,0.0003613428,0.0007327816,0.0022352736,0.0047609983,0.005682864,0.0013451831,0.0012595514,0.017387334],"category_scores_gemma":[0.08946128,0.00025926105,0.00041843634,0.002627988,0.0027425704,0.0047586807,0.0042846464,0.003083292,0.0029609797],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005402204,0.0002441808,0.114049435,0.0007017306,0.000090227084,0.0003392647,0.037557192,0.00029928828,0.00045830666,0.008295843,0.6038965,0.23352778],"study_design_scores_gemma":[0.000066087785,0.0002082873,0.5790413,0.0022206118,0.00012682327,0.0002327897,0.051506616,0.0006490584,0.0008784763,0.0073159942,0.35765943,0.000094554955],"about_ca_topic_score_codex":0.12636146,"about_ca_topic_score_gemma":0.14283623,"teacher_disagreement_score":0.12636146,"about_ca_system_score_codex":0.0068237353,"about_ca_system_score_gemma":0.009269804,"threshold_uncertainty_score":0.25125194},"labels":[],"label_agreement":null},{"id":"W2547516676","doi":"","title":"POSTER: Enhancements of Simulated Science Laboratory Assessments","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; University of Calgary","funders":"","keywords":"Intervention (counseling); Computer science; Mathematics education; Psychology","score_opus":0.06923889265197267,"score_gpt":0.45057729212756237,"score_spread":0.3813383994755897,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2547516676","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21819869,0.0018138284,0.0982035,0.036436267,0.041651722,0.018273111,0.015134622,0.016268855,0.55401933],"genre_scores_gemma":[0.4957989,0.0019661796,0.14716181,0.0072800918,0.0059624715,0.011144572,0.009703956,0.0018609111,0.31912106],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99729675,0.0009104456,0.00017485814,0.00036535403,0.0010495037,0.0002030546],"domain_scores_gemma":[0.98393655,0.0052517154,0.0005827001,0.0015628977,0.0061536827,0.002512379],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006433879,0.0007823428,0.00058130577,0.00075464114,0.0007297373,0.0019449847,0.001749832,0.0013853195,0.17743582],"category_scores_gemma":[0.020138158,0.0003737247,0.0008721236,0.00041759893,0.00040973263,0.0013390967,0.002577016,0.0024626625,0.02695577],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0073289326,0.01046449,0.006214627,0.0016964636,0.00008325664,0.0005321014,0.0012621405,0.0056735445,0.021491334,0.005749531,0.4186021,0.5209015],"study_design_scores_gemma":[0.0022024629,0.016612304,0.03869738,0.00096063985,0.00019685464,0.0006793878,0.0012552338,0.014263231,0.022833306,0.011501784,0.89048856,0.000308814],"about_ca_topic_score_codex":0.00061590056,"about_ca_topic_score_gemma":0.002031596,"teacher_disagreement_score":0.17743582,"about_ca_system_score_codex":0.0009092508,"about_ca_system_score_gemma":0.0018232127,"threshold_uncertainty_score":0.5935819},"labels":[],"label_agreement":null},{"id":"W2555581813","doi":"","title":"POSTER: Factor Structure, Wording Effects, and Reliability Estimates in Health and Housing Specific Self-efficacy Scales with Individuals who are Homeless or Vulnerably Housed","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Homelessness and Social Issues","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Polychoric correlation; Exploratory factor analysis; Confirmatory factor analysis; Likert scale; Psychology; Ordinal Scale; Structural equation modeling; Ordinal data; Scale (ratio); Factor analysis; Internal consistency; Statistics; Psychometrics; Competence (human resources); Clinical psychology; Econometrics; Mathematics; Social psychology; Developmental psychology; Correlation","score_opus":0.04124889507717,"score_gpt":0.35342926676075886,"score_spread":0.31218037168358886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2555581813","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98759395,0.00020336319,0.0021856683,0.00061320944,0.00028389704,0.00062080886,0.0009025672,0.00007299999,0.007523636],"genre_scores_gemma":[0.9916835,0.0001282197,0.0034203092,0.00009628173,0.000088816465,0.00087223493,0.0007979148,0.000028495264,0.0028842918],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99648297,0.0014706437,0.0003702976,0.00039385937,0.0010826631,0.00019957472],"domain_scores_gemma":[0.96984655,0.020651674,0.0019890782,0.0013661298,0.005462349,0.00068434875],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.012081966,0.00037795072,0.0005184154,0.0014779458,0.0010431653,0.0010435351,0.00048310825,0.00042848062,0.019823574],"category_scores_gemma":[0.032208025,0.00025288595,0.0013123632,0.0009289742,0.0007499526,0.00083686627,0.0011343964,0.0009332039,0.0017022801],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009805687,0.0008611655,0.8265109,0.00041541678,0.0003428498,0.00014368941,0.011939823,0.00077655277,0.0025502536,0.001607297,0.017174039,0.13669734],"study_design_scores_gemma":[0.000091867485,0.00055778376,0.985571,0.0002448625,0.00010559947,0.00014284569,0.005507744,0.0020044812,0.001031383,0.00071442325,0.003997402,0.000030496116],"about_ca_topic_score_codex":0.001021343,"about_ca_topic_score_gemma":0.0013627548,"teacher_disagreement_score":0.987918,"about_ca_system_score_codex":0.0004254865,"about_ca_system_score_gemma":0.0006664047,"threshold_uncertainty_score":0.066316485},"labels":[],"label_agreement":null},{"id":"W2566375536","doi":"","title":"SYMPOSIUM: Testing in Linguistic Diversity Contexts","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Linguistics, Language Diversity, and Identity","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; University of British Columbia","funders":"","keywords":"Linguistics; Linguistic diversity; Diversity (politics); Computer science; Natural language processing; Sociology; Anthropology; Philosophy","score_opus":0.061896280785448295,"score_gpt":0.23108295730309403,"score_spread":0.16918667651764574,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2566375536","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3168009,0.0075676036,0.04360266,0.35879847,0.031105507,0.0009876381,0.005042277,0.0012195098,0.23487546],"genre_scores_gemma":[0.8402866,0.0019744635,0.02886776,0.020791272,0.008568098,0.0008960739,0.0041163755,0.0009749795,0.093524285],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9848347,0.009694339,0.00070941495,0.0017359158,0.0019698776,0.0010556955],"domain_scores_gemma":[0.9131736,0.044965185,0.0017413144,0.007035453,0.01635121,0.016733238],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03943099,0.000778682,0.00093979744,0.0019459865,0.0050143795,0.008571815,0.0025024335,0.004334874,0.027729804],"category_scores_gemma":[0.07455981,0.00046083645,0.0010961806,0.0017427308,0.006014912,0.008082036,0.009740174,0.007115012,0.0043469467],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013229334,0.0019519281,0.057870988,0.0005247049,0.00017111583,0.0010597032,0.026716027,0.002120598,0.006542656,0.105639376,0.5569902,0.23908974],"study_design_scores_gemma":[0.0004334208,0.0020096286,0.09489293,0.0017057172,0.00028497766,0.0011461356,0.052681983,0.008142968,0.011220696,0.19970627,0.6274544,0.0003208838],"about_ca_topic_score_codex":0.014533537,"about_ca_topic_score_gemma":0.021180522,"teacher_disagreement_score":0.03943099,"about_ca_system_score_codex":0.0038368185,"about_ca_system_score_gemma":0.009948829,"threshold_uncertainty_score":0.20853353},"labels":[],"label_agreement":null},{"id":"W2599662706","doi":"","title":"PAPER: Examining Validity Evidence for Multidimensional Forced Choice Measures using Four Scoring Approaches","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Advanced Statistical Modeling Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Two-alternative forced choice; Item response theory; Psychology; Test (biology); Normative; Likert scale; Classical test theory; Rank (graph theory); Statistics; Psychometrics; Social psychology; Mathematics; Cognitive psychology; Clinical psychology; Developmental psychology","score_opus":0.690729672794958,"score_gpt":0.3971126026713275,"score_spread":0.29361707012363053,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2599662706","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85926217,0.0026096278,0.0879347,0.0027296017,0.0014986024,0.0028312197,0.0020104586,0.0002451356,0.04087846],"genre_scores_gemma":[0.9522648,0.0004962943,0.041328862,0.00067616056,0.00028102982,0.0026581525,0.0008145085,0.00011119603,0.0013689046],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.8281459,0.1040701,0.014251789,0.012170204,0.03992104,0.0014409267],"domain_scores_gemma":[0.26730737,0.5939977,0.048340898,0.03926019,0.04881157,0.002282363],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1427534,0.001387674,0.0012204923,0.006759272,0.0028299005,0.004894155,0.003156346,0.0019371851,0.009720724],"category_scores_gemma":[0.4304318,0.0006317914,0.003368416,0.005737255,0.005482672,0.006451876,0.005244147,0.0024748612,0.0014234452],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024472428,0.000872754,0.74498177,0.0011932086,0.0019648066,0.00015697374,0.010612359,0.001399077,0.0010404397,0.027039569,0.0031258161,0.20516591],"study_design_scores_gemma":[0.0011397256,0.004707618,0.8607566,0.003337367,0.0018470932,0.0012677803,0.01267448,0.03741439,0.0067551434,0.048224546,0.02133365,0.00054162985],"about_ca_topic_score_codex":0.001443085,"about_ca_topic_score_gemma":0.0014179925,"teacher_disagreement_score":0.8572466,"about_ca_system_score_codex":0.002862982,"about_ca_system_score_gemma":0.00300552,"threshold_uncertainty_score":0.7549612},"labels":[],"label_agreement":null},{"id":"W2602631024","doi":"","title":"WORKSHOP: Testing Multigroup and Multilevel Assessment Scale Equivalence across Nations and Cultures","year":2015,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Advanced Statistical Modeling Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Scale (ratio); Mainstream; Equivalence (formal languages); Multilevel model; Set (abstract data type); Level of measurement; Extant taxon; Psychology; Social science; Geography; Political science; Sociology; Mathematics; Statistics; Computer science; Cartography","score_opus":0.14311314071787276,"score_gpt":0.41575796760203926,"score_spread":0.27264482688416647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2602631024","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.100763515,0.0035297226,0.61695224,0.15054114,0.028592559,0.022357784,0.0036480036,0.0016440065,0.07197112],"genre_scores_gemma":[0.29619318,0.002337584,0.58396465,0.025836024,0.004267473,0.043788686,0.003110877,0.0011095569,0.039391875],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9478544,0.040243488,0.0014297486,0.0043624,0.0043133968,0.0017966412],"domain_scores_gemma":[0.83118904,0.11845863,0.0025645809,0.0092571825,0.03238131,0.006149241],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.15239555,0.001772335,0.0014097026,0.0010576622,0.0044283085,0.007051268,0.006417093,0.0043102307,0.013284722],"category_scores_gemma":[0.17064436,0.0014145903,0.0036957047,0.0010642197,0.0037512004,0.0060217516,0.015684608,0.012777689,0.0037797345],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015227183,0.00253959,0.009777384,0.001711263,0.0006727268,0.0012470741,0.05640339,0.005993399,0.00729971,0.16783758,0.4236219,0.3213733],"study_design_scores_gemma":[0.0013835311,0.0040063,0.019609481,0.0054177996,0.00045595615,0.0011410948,0.059722964,0.027817968,0.010691513,0.24160182,0.6273193,0.00083227153],"about_ca_topic_score_codex":0.0026282906,"about_ca_topic_score_gemma":0.0034113424,"teacher_disagreement_score":0.15239555,"about_ca_system_score_codex":0.0034279893,"about_ca_system_score_gemma":0.007375367,"threshold_uncertainty_score":0.80595434},"labels":[],"label_agreement":null},{"id":"W2605338221","doi":"","title":"SYMPOSIUM: Continuous Testing: Psychometric Challenges and Opportunities","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Fungal Infections and Studies","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Psychometric testing; Psychology; Computer science; Psychometrics; Data science; Clinical psychology; Cronbach's alpha","score_opus":0.20318485710955597,"score_gpt":0.29992223470879903,"score_spread":0.09673737759924306,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2605338221","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004797237,0.032144383,0.014542489,0.87591606,0.06249231,0.00022735461,0.00035687504,0.00019711754,0.009326126],"genre_scores_gemma":[0.22605534,0.09262039,0.10144099,0.29527852,0.23751038,0.0018790282,0.001821889,0.0010661574,0.042327408],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9622191,0.02574158,0.002990562,0.0014837239,0.006026209,0.001538812],"domain_scores_gemma":[0.68672734,0.14246489,0.0072788415,0.014354956,0.11758954,0.031584404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.14930488,0.0011685543,0.0017523309,0.0027247637,0.0025183032,0.012920759,0.0031340432,0.009012695,0.016449371],"category_scores_gemma":[0.2192174,0.0006710188,0.0016333207,0.0027581903,0.0047717947,0.008759661,0.005830011,0.015322295,0.0037233427],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004129258,0.0004152914,0.008811626,0.00071839866,0.00009863189,0.00023171068,0.0013962361,0.0004419991,0.00096486846,0.012517583,0.68353826,0.29045248],"study_design_scores_gemma":[0.00030646188,0.0013780892,0.035798766,0.008681869,0.00021194557,0.0015313496,0.008859371,0.004752998,0.001322987,0.07882966,0.8579963,0.0003302475],"about_ca_topic_score_codex":0.0037377686,"about_ca_topic_score_gemma":0.007269303,"teacher_disagreement_score":0.14930488,"about_ca_system_score_codex":0.0036196744,"about_ca_system_score_gemma":0.016076794,"threshold_uncertainty_score":0.78960913},"labels":[],"label_agreement":null},{"id":"W2611906781","doi":"","title":"POSTER: The Impact of Group Imbalance on Logistic Regression Analyses with Assessment Data","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Advanced Statistical Modeling Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Statistics; Covariate; Skewness; Type I and type II errors; Logistic regression; Mathematics; Sample size determination; Econometrics; Variables; Wald test; Regression analysis; Analysis of covariance; Variable (mathematics); Statistical hypothesis testing","score_opus":0.22700990350077827,"score_gpt":0.4714656255143577,"score_spread":0.24445572201357946,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2611906781","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12725808,0.008617878,0.5607301,0.12223809,0.0357801,0.006362526,0.008549851,0.010975791,0.11948765],"genre_scores_gemma":[0.573124,0.00529798,0.286192,0.02211738,0.013129298,0.009683566,0.0069756657,0.0033144697,0.08016567],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9455085,0.043256585,0.0017943801,0.0021484634,0.0065329825,0.0007590898],"domain_scores_gemma":[0.69230276,0.26057476,0.00977013,0.0141493,0.019826664,0.0033764308],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.060055047,0.0010517321,0.0014359732,0.0018474928,0.0015446447,0.0039974293,0.0019828933,0.0022080264,0.088581204],"category_scores_gemma":[0.25822794,0.00065341283,0.0022104187,0.0017572206,0.001623773,0.002769115,0.0034570473,0.003291191,0.02127858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004977048,0.0009940962,0.02595871,0.002165349,0.0005387695,0.00076947425,0.0014954066,0.0125172,0.0022897553,0.023531973,0.39879102,0.52597123],"study_design_scores_gemma":[0.003223838,0.009100606,0.09287244,0.007407756,0.0009843756,0.004409558,0.0021999618,0.13375287,0.014117531,0.23115924,0.49999595,0.0007758169],"about_ca_topic_score_codex":0.0004567095,"about_ca_topic_score_gemma":0.0005565794,"teacher_disagreement_score":0.088581204,"about_ca_system_score_codex":0.0014615844,"about_ca_system_score_gemma":0.0018307894,"threshold_uncertainty_score":0.31760526},"labels":[],"label_agreement":null},{"id":"W2612009109","doi":"","title":"WORKSHOP: Keeping up with The Standards: How to Design and Evaluate Reliability and Validity Studies","year":2015,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Facilitator; Context (archaeology); Psychology; Reliability (semiconductor); Internal validity; Validity; Standards for Educational and Psychological Testing; External validity; Construct validity; Applied psychology; Presentation (obstetrics); Quality (philosophy); Psychometrics; Social psychology; Clinical psychology; Medicine; Higher education; Education theory; Political science","score_opus":0.8179560533686604,"score_gpt":0.5321420139146238,"score_spread":0.2858140394540366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2612009109","genre_codex":"commentary","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0032313413,0.017215014,0.20917045,0.6382353,0.07596797,0.028696436,0.0018276426,0.003110501,0.022545453],"genre_scores_gemma":[0.031601343,0.020962514,0.5628442,0.23754187,0.023248862,0.08756224,0.0025174108,0.004439138,0.029282423],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7099651,0.22738501,0.020225253,0.0073912772,0.029993055,0.005040185],"domain_scores_gemma":[0.37363747,0.2894179,0.025946872,0.03682421,0.25032663,0.023846932],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3926879,0.0025379984,0.003455858,0.005316621,0.004524206,0.013792829,0.009743472,0.012532551,0.020416887],"category_scores_gemma":[0.5513165,0.003276187,0.0050137346,0.0031885977,0.011029693,0.015809754,0.014720645,0.024500579,0.030726375],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027055148,0.00018640111,0.0006626252,0.004320999,0.00018881053,0.00026636382,0.009276407,0.0004641039,0.0012026518,0.011140332,0.7714138,0.2006069],"study_design_scores_gemma":[0.000404625,0.00043577023,0.0022955616,0.017139439,0.00017381042,0.0004623724,0.0068448163,0.00079516094,0.0014916337,0.04751332,0.92209387,0.00034950481],"about_ca_topic_score_codex":0.00471622,"about_ca_topic_score_gemma":0.007482599,"teacher_disagreement_score":0.6073121,"about_ca_system_score_codex":0.009510858,"about_ca_system_score_gemma":0.03631271,"threshold_uncertainty_score":0.74892396},"labels":[],"label_agreement":null},{"id":"W2612770846","doi":"","title":"PAPER: Examining Position Effects in Large-Scale Assessments Using an SEM Approach","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Rasch model; Position (finance); Reading (process); Test (biology); Structural equation modeling; Scale (ratio); Psychology; Item response theory; Sample (material); Computer science; Statistics; Econometrics; Psychometrics; Social psychology; Cognitive psychology; Artificial intelligence; Natural language processing; Mathematics; Developmental psychology; Linguistics","score_opus":0.5208722029508508,"score_gpt":0.4857174030503449,"score_spread":0.035154799900505906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2612770846","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41033256,0.00040400334,0.5778269,0.0010007445,0.00025691738,0.0009960153,0.0011521014,0.0012599065,0.0067708436],"genre_scores_gemma":[0.6869673,0.00014888769,0.30994773,0.00021006505,0.00006582098,0.00095971394,0.00068779377,0.00016368528,0.00084899337],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9657129,0.026666816,0.0014240583,0.0028588804,0.003139896,0.00019746396],"domain_scores_gemma":[0.7443389,0.22522138,0.012148992,0.010340909,0.0073648333,0.0005849114],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.029248111,0.0012363773,0.0009132183,0.0022958007,0.00069031876,0.0020852864,0.0012246666,0.0006038253,0.004705667],"category_scores_gemma":[0.14053418,0.00053136377,0.0014863783,0.0035570457,0.0014368917,0.001988184,0.002048442,0.001152768,0.0006297427],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004695309,0.00063977507,0.45071492,0.0016082056,0.0035808014,0.000578784,0.006704047,0.040666852,0.0055709225,0.024058383,0.008141628,0.45726615],"study_design_scores_gemma":[0.0002053251,0.0024297978,0.5385437,0.001243727,0.0013784546,0.0009050974,0.00661851,0.35490656,0.010330195,0.06607521,0.01715819,0.00020515344],"about_ca_topic_score_codex":0.0018322713,"about_ca_topic_score_gemma":0.003214291,"teacher_disagreement_score":0.9707519,"about_ca_system_score_codex":0.0009956965,"about_ca_system_score_gemma":0.0022125398,"threshold_uncertainty_score":0.15468061},"labels":[],"label_agreement":null},{"id":"W2615639425","doi":"","title":"PAPER: Measuring Teachers’ Assessment Literacy and Conceptions of Assessment in a High Performative Canadian Context: A Construct Validity Study","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Performative utterance; Psychology; Literacy; Context (archaeology); Construct (python library); Construct validity; Mathematics education; Pedagogy; Psychometrics; Computer science; Developmental psychology","score_opus":0.06347086988825551,"score_gpt":0.3642921981591032,"score_spread":0.3008213282708477,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2615639425","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.987973,0.00021119589,0.00086995313,0.000552625,0.00004133345,0.000422158,0.0010208272,0.0000135946375,0.00889538],"genre_scores_gemma":[0.99342585,0.00033414637,0.0023735235,0.00024792276,0.000019121178,0.0004206114,0.00084025424,0.000015799445,0.002322847],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9903909,0.0011472224,0.0006851619,0.0009826417,0.0059632384,0.0008308022],"domain_scores_gemma":[0.9310805,0.013382258,0.007772631,0.0032195463,0.04096781,0.0035771953],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011739544,0.000460096,0.0005688453,0.0032212562,0.009140628,0.0039823437,0.0016607074,0.0006011861,0.0023829455],"category_scores_gemma":[0.059126217,0.00046491603,0.00064715307,0.0058397558,0.0033079218,0.0016706819,0.0033354214,0.0015312553,0.00031943558],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000121456775,0.00024282312,0.87417257,0.00022403366,0.000056833265,0.00012444251,0.085386366,0.00023452636,0.000652279,0.0015329429,0.0032174913,0.03403427],"study_design_scores_gemma":[0.000018489258,0.00009493572,0.94158775,0.00016853496,0.000043861193,0.00008241986,0.04828421,0.0004848715,0.0005320926,0.00022154706,0.008422924,0.00005842097],"about_ca_topic_score_codex":0.94443506,"about_ca_topic_score_gemma":0.95358855,"teacher_disagreement_score":0.05556494,"about_ca_system_score_codex":0.026048573,"about_ca_system_score_gemma":0.06963117,"threshold_uncertainty_score":0.1889965},"labels":[],"label_agreement":null},{"id":"W2616059813","doi":"","title":"SYMPOSIUM: The ITC Guidelines for Test Adaptation, version 2.0","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Mathematics, Computing, and Information Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Test (biology); Adaptation (eye); Computer science; Psychology; Geology","score_opus":0.09135627759384762,"score_gpt":0.3022145433295347,"score_spread":0.21085826573568706,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2616059813","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0066236537,0.010952053,0.7052139,0.023480715,0.012332127,0.006324668,0.032139394,0.08410673,0.11882676],"genre_scores_gemma":[0.040436424,0.009119187,0.67233884,0.017215252,0.004610493,0.011171394,0.09793942,0.03688259,0.110286415],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.95223886,0.017675089,0.0093346415,0.001534773,0.017068183,0.0021485046],"domain_scores_gemma":[0.8325523,0.038863283,0.0067438786,0.02326479,0.094107494,0.004468195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04580239,0.0036215875,0.0027912227,0.009924487,0.0021189344,0.011496184,0.0125832185,0.012905989,0.038905296],"category_scores_gemma":[0.15499511,0.0035093736,0.0037553012,0.006174177,0.0031725396,0.0057951673,0.006935744,0.009880452,0.07267366],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066210853,0.0005563952,0.0020744773,0.0017514249,0.00013579142,0.00033165514,0.0008888919,0.008514868,0.006589413,0.014130113,0.7653895,0.19897541],"study_design_scores_gemma":[0.00036487626,0.0005009612,0.006403788,0.0061280415,0.00024293903,0.0014766066,0.00042092334,0.01270175,0.013609645,0.028225116,0.9295216,0.0004036945],"about_ca_topic_score_codex":0.016103787,"about_ca_topic_score_gemma":0.010970334,"teacher_disagreement_score":0.04580239,"about_ca_system_score_codex":0.0032551712,"about_ca_system_score_gemma":0.013910095,"threshold_uncertainty_score":0.24222904},"labels":[],"label_agreement":null},{"id":"W2617029349","doi":"","title":"POSTER: Common Misconceptions and the Misuses of Standardized Assessments","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Teacher Education and Assessments","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Calgary; University of Alberta","funders":"","keywords":"Scrutiny; Argument (complex analysis); Perception; Psychology; Public relations; Sample (material); Public opinion; Test (biology); Public policy; Social psychology; Political science; Medicine; Law; Politics","score_opus":0.06636221969219422,"score_gpt":0.42757827779907426,"score_spread":0.36121605810688007,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2617029349","genre_codex":"commentary","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14694387,0.039637707,0.021959532,0.61349213,0.044943616,0.00091905356,0.0025740024,0.0013326791,0.12819749],"genre_scores_gemma":[0.74122006,0.028717875,0.00929242,0.15287758,0.016763976,0.0009033515,0.00092485157,0.00085307425,0.0484468],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9595657,0.024430927,0.0025124534,0.0015881769,0.010760418,0.0011423114],"domain_scores_gemma":[0.7807943,0.16385114,0.018719872,0.007293446,0.02608778,0.0032535077],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04417986,0.00080060906,0.0006597407,0.0046466426,0.0054321834,0.0063333944,0.001803963,0.0045630718,0.022107448],"category_scores_gemma":[0.16938122,0.0005532475,0.0008582653,0.0027705666,0.008248201,0.0070290407,0.0060395077,0.006818634,0.005207908],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002527554,0.00014267904,0.012753126,0.0054216976,0.00007450051,0.0021448832,0.29483736,0.000270343,0.0016449923,0.030510968,0.4559384,0.19600835],"study_design_scores_gemma":[0.00003253535,0.0001906482,0.0068657445,0.011261312,0.000061110804,0.0023011153,0.10826475,0.0005619254,0.00122672,0.011071554,0.8580365,0.00012600019],"about_ca_topic_score_codex":0.0037734967,"about_ca_topic_score_gemma":0.0040464266,"teacher_disagreement_score":0.04417986,"about_ca_system_score_codex":0.0052890046,"about_ca_system_score_gemma":0.00633812,"threshold_uncertainty_score":0.23364818},"labels":[],"label_agreement":null},{"id":"W2619373123","doi":"","title":"INVITED SYPOSIUM: The Life and Work of Dr. Thomas David Oakland","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Tribute; Management; Honor; Library science; Test (biology); Political science; Sociology; Psychology; Law","score_opus":0.03728799600135992,"score_gpt":0.28968563677974063,"score_spread":0.2523976407783807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2619373123","genre_codex":"editorial","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013571332,0.018106936,0.0018576544,0.45954064,0.46505612,0.00017533531,0.00025438768,0.00042441286,0.053227365],"genre_scores_gemma":[0.024995051,0.018151082,0.002140358,0.2407967,0.17460655,0.0005088137,0.00038483992,0.0009969755,0.5374196],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963792,0.001312134,0.000118757715,0.00054452353,0.0012294096,0.00041591423],"domain_scores_gemma":[0.98818886,0.0010795317,0.00033139862,0.0003264531,0.0037927178,0.0062810224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004148229,0.0018334831,0.0012671689,0.001242076,0.006816597,0.0067122034,0.0016253988,0.0038271025,0.077213],"category_scores_gemma":[0.014732213,0.0007604053,0.00079605484,0.0005908483,0.002091553,0.0043557286,0.0049848496,0.010790134,0.05104073],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000012860831,0.000013025231,0.00007421719,0.000029199113,0.0000025258992,0.000078878744,0.00027134398,0.000010362162,0.00009719394,0.0007201164,0.99144113,0.007249146],"study_design_scores_gemma":[0.000005263993,0.000020797648,0.00022972311,0.00011342809,0.0000027940305,0.00021120379,0.0014119626,0.00004200239,0.000062534455,0.00058807596,0.99729866,0.000013597649],"about_ca_topic_score_codex":0.005218556,"about_ca_topic_score_gemma":0.008392541,"teacher_disagreement_score":0.077213,"about_ca_system_score_codex":0.004553554,"about_ca_system_score_gemma":0.0057291794,"threshold_uncertainty_score":0.25830317},"labels":[],"label_agreement":null},{"id":"W2619791510","doi":"","title":"PAPER: Adjusting for Cumulative French Language Exposure in the WISC V French Canadian Norms","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Language Development and Disorders","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"French; Psychology; Demography; Population; Neighbourhood (mathematics); Geography; Developmental psychology; Sociology; Mathematics","score_opus":0.027126692773121748,"score_gpt":0.29363266078983663,"score_spread":0.2665059680167149,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2619791510","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97407687,0.0012383074,0.0072844885,0.0008860771,0.0001711806,0.00016590615,0.009954152,0.00028880677,0.005934094],"genre_scores_gemma":[0.9902754,0.00018543735,0.0036369392,0.00012065775,0.000023409986,0.00010738915,0.003969301,0.000041544157,0.0016399684],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9948101,0.0009002023,0.00041511256,0.0015809322,0.0016922295,0.0006014039],"domain_scores_gemma":[0.99191475,0.0015535952,0.002564601,0.00090161926,0.0025193624,0.0005460501],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004504506,0.0010852313,0.00046459242,0.0015783934,0.0011882654,0.0013544074,0.0020947591,0.00053039676,0.0025081425],"category_scores_gemma":[0.018444024,0.00025862508,0.0013483448,0.0022130662,0.00057242106,0.0005488095,0.0009940705,0.00074400625,0.0002816995],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007659664,0.000012136282,0.9889466,0.000020015454,0.00035490302,0.000054257896,0.00015417549,0.00038884557,0.00011950492,0.00021281432,0.0017374834,0.007922629],"study_design_scores_gemma":[0.0000065081604,0.00004685741,0.9960859,0.000028193857,0.00013076451,0.00007067206,0.00016955528,0.001116127,0.00024279846,0.00007378176,0.002018429,0.000010473066],"about_ca_topic_score_codex":0.8954008,"about_ca_topic_score_gemma":0.81084573,"teacher_disagreement_score":0.10459918,"about_ca_system_score_codex":0.0059072115,"about_ca_system_score_gemma":0.011891932,"threshold_uncertainty_score":0.21043032},"labels":[],"label_agreement":null},{"id":"W2623451806","doi":"","title":"POSTER: A Comprehensive Review of the Historical Contributions into Language Acquisition and Development Over the Last Century are Formulated Here to Provide an Understanding of how Normal Language Develops and Difficulties Occur","year":2016,"lang":"en","type":"review","venue":"ITC 2016 Conference","topic":"Language Development and Disorders","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Reading (process); Cognitive science; Perception; Cognition; Psychology; Modalities; Process (computing); Language development; Language acquisition; Cognitive psychology; Linguistics; Computer science; Mathematics education; Developmental psychology; Sociology; Neuroscience","score_opus":0.04355742069082458,"score_gpt":0.32360935669070734,"score_spread":0.28005193599988276,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2623451806","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0007353304,0.97728544,0.00058798055,0.0038042953,0.0032837545,0.000009469315,0.00009334869,0.00002091037,0.014179449],"genre_scores_gemma":[0.0065217135,0.9789161,0.0007013723,0.0018113599,0.0040744795,0.000022566765,0.00018409245,0.000028542518,0.0077397455],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9992397,0.00020288173,0.00010993847,0.0001860528,0.0001995372,0.0000619838],"domain_scores_gemma":[0.99799466,0.001221276,0.00022577013,0.00006182733,0.00037624588,0.00012019091],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013338254,0.0008161363,0.0007485457,0.0053089783,0.0013767384,0.0029706853,0.0008478606,0.0014266091,0.017028645],"category_scores_gemma":[0.0038131948,0.00032584346,0.00045780442,0.0059773517,0.0019027792,0.004405357,0.0016511179,0.00212876,0.00616943],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107100466,0.00006467537,0.00186857,0.013667596,0.00004839565,0.00041561335,0.003039605,0.00030055293,0.00084873085,0.045167044,0.12449023,0.8099819],"study_design_scores_gemma":[0.0000019976637,0.000036795205,0.0029925848,0.007712441,0.000020568126,0.0010951223,0.0009760946,0.000043745753,0.00016694449,0.0046977485,0.9822318,0.00002422004],"about_ca_topic_score_codex":0.0019704509,"about_ca_topic_score_gemma":0.002748763,"teacher_disagreement_score":0.017028645,"about_ca_system_score_codex":0.0015965414,"about_ca_system_score_gemma":0.002095093,"threshold_uncertainty_score":0.056966543},"labels":[],"label_agreement":null},{"id":"W2736595548","doi":"","title":"POSTER: A Validity Argument Approach to Collaborative Development of the Colleges Mathematics Assessment Program","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"York University; University of Toronto","funders":"","keywords":"Argument (complex analysis); Test (biology); Mathematics education; Computer science; Psychology","score_opus":0.08059059746095346,"score_gpt":0.3733509050348911,"score_spread":0.2927603075739377,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2736595548","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01154037,0.0012037137,0.18149103,0.21376558,0.0026790036,0.00300185,0.00043748427,0.0004887584,0.58539224],"genre_scores_gemma":[0.5770336,0.0013494467,0.2186732,0.018836545,0.0015748566,0.004435054,0.0004980541,0.00054771407,0.17705165],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9545036,0.030741936,0.0013319404,0.0023609793,0.009532104,0.0015294413],"domain_scores_gemma":[0.91050273,0.058058538,0.0033451098,0.006080434,0.019155474,0.002857829],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.059498023,0.0006169098,0.0004832618,0.0029983986,0.0098161185,0.012015584,0.0036847568,0.006987207,0.049228773],"category_scores_gemma":[0.093991175,0.0006226134,0.0011163906,0.0016287202,0.013335868,0.0076285666,0.009948881,0.006297727,0.0048281797],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007726312,0.00015629028,0.0020987538,0.00049791834,0.0000194711,0.00046104818,0.023993907,0.0013539392,0.0004153223,0.8290516,0.069941185,0.07193331],"study_design_scores_gemma":[0.00014502011,0.0001693059,0.003465975,0.0022015502,0.000065537904,0.00034646873,0.013730791,0.0062458036,0.0018451267,0.33820826,0.63350016,0.00007593182],"about_ca_topic_score_codex":0.017234894,"about_ca_topic_score_gemma":0.02107902,"teacher_disagreement_score":0.059498023,"about_ca_system_score_codex":0.02031814,"about_ca_system_score_gemma":0.023629924,"threshold_uncertainty_score":0.31465942},"labels":[],"label_agreement":null},{"id":"W2742582834","doi":"","title":"POSTER: Overclaiming as Convenient Proxy for Confidence and Overconfidence","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Statistics Education and Methodologies","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Overconfidence effect; Confidence interval; Proxy (statistics); Vocabulary; Respondent; Psychology; Test (biology); Self-confidence; Social psychology; Statistics; Mathematics; Linguistics","score_opus":0.21118577158893947,"score_gpt":0.42812657072379057,"score_spread":0.2169407991348511,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2742582834","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.59123796,0.011262744,0.022516083,0.06339071,0.02984338,0.0019507781,0.0141081475,0.0032750538,0.2624152],"genre_scores_gemma":[0.8660729,0.0037143703,0.0133483475,0.009561128,0.008692863,0.001093097,0.0034648608,0.00044506605,0.093607426],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99746764,0.00091647665,0.00019594218,0.00033365202,0.00087378407,0.00021253285],"domain_scores_gemma":[0.9761619,0.009907311,0.005153048,0.0016070139,0.004365666,0.0028050027],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038664239,0.0006393492,0.0004975449,0.0009921181,0.0007712142,0.001913028,0.0006577448,0.0016161483,0.11327003],"category_scores_gemma":[0.02746315,0.0001921586,0.00052342383,0.00054832635,0.00074680184,0.0010870704,0.0014510591,0.0020962688,0.019955423],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0037883625,0.0013954968,0.1345286,0.0024711438,0.00021432989,0.0006935933,0.0020929147,0.00079569523,0.020996476,0.0054315417,0.4815259,0.34606594],"study_design_scores_gemma":[0.00045489933,0.0060576242,0.51058185,0.0028557337,0.0004360512,0.00975594,0.0042596087,0.0033321127,0.033232,0.016332477,0.4122937,0.00040800092],"about_ca_topic_score_codex":0.00030105,"about_ca_topic_score_gemma":0.00043339096,"teacher_disagreement_score":0.11327003,"about_ca_system_score_codex":0.0004931944,"about_ca_system_score_gemma":0.00060616754,"threshold_uncertainty_score":0.37892598},"labels":[],"label_agreement":null},{"id":"W2761268105","doi":"","title":"POSTER: What work habits are being assessed across Canada?","year":2016,"lang":"en","type":"article","venue":"ITC 2016 Conference","topic":"Agricultural and Financial Auditing","field":"Agricultural and Biological Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Context (archaeology); Work (physics); Psychology; Medical education; Geography; Medicine","score_opus":0.020165917419981368,"score_gpt":0.20148767723297017,"score_spread":0.18132175981298881,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2761268105","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7562915,0.008313637,0.0014988193,0.033056322,0.0012411909,0.0014599519,0.07301524,0.0004024236,0.124720804],"genre_scores_gemma":[0.93637407,0.0043098163,0.0020658963,0.0037898903,0.00013571251,0.00052267447,0.008312967,0.00008954587,0.0443994],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99598014,0.00033773575,0.00021373066,0.00043492907,0.0017788371,0.001254685],"domain_scores_gemma":[0.98317844,0.0012157899,0.0015217962,0.00033699008,0.009992698,0.0037543715],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037189869,0.00027456146,0.00044175543,0.0031655456,0.007451447,0.003970165,0.0011730179,0.0006953961,0.018511869],"category_scores_gemma":[0.0103651825,0.00030618822,0.0004958192,0.0053937486,0.0020362253,0.00075072167,0.0017235867,0.0009865327,0.0026250829],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002138511,0.00014887896,0.7270203,0.0011534262,0.000053921154,0.00027803986,0.026445616,0.00023969014,0.00066119124,0.0011746006,0.1127908,0.12981954],"study_design_scores_gemma":[0.000010011944,0.00006881355,0.89855146,0.0008390705,0.000023626815,0.00008234993,0.04186979,0.00010098156,0.00036367768,0.00017736614,0.05783804,0.00007475633],"about_ca_topic_score_codex":0.9753912,"about_ca_topic_score_gemma":0.9883847,"teacher_disagreement_score":0.037439544,"about_ca_system_score_codex":0.037439544,"about_ca_system_score_gemma":0.09772246,"threshold_uncertainty_score":0.27164418},"labels":[],"label_agreement":null}]}