{"id":"W2885169382","doi":"10.1111/emip.12211","title":"How Robust Are Cross‐Country Comparisons of PISA Scores to the Scaling Model Used?","year":2018,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Criticism; Robustness (evolution); Underpinning; Item response theory; Psychology; Test (biology); Cross country; Psychometrics; Political science; Developmental psychology; Economics; Demographic economics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1752474,0.001186957,0.001751624,0.003552754,0.001817807,0.007863597,0.004104985,0.002014757,0.008193031],"category_scores_gemma":[0.5214518,0.0007287199,0.003588361,0.006480979,0.004739623,0.005871437,0.005877248,0.004241674,0.002977046],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001329993,"about_ca_system_score_gemma":0.001455731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006513461,"about_ca_topic_score_gemma":0.003431921,"domain_scores_codex":[0.7958846,0.1635188,0.008506343,0.01714711,0.01210095,0.002842152],"domain_scores_gemma":[0.4642074,0.388118,0.04102074,0.07958438,0.02438313,0.00268637],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0021362,0.0005403552,0.7238457,0.001343145,0.02169016,0.00049515,0.009539147,0.02101484,0.001695386,0.04135913,0.01777097,0.1585699],"study_design_scores_gemma":[0.0005092201,0.002588385,0.7896217,0.003260493,0.00595405,0.0007606614,0.02936522,0.03650263,0.008088131,0.07724637,0.04556357,0.0005395883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7514302,0.004523666,0.1609038,0.01039313,0.004449642,0.00100455,0.006122195,0.0006920865,0.06048077],"genre_scores_gemma":[0.9857512,0.0002454584,0.009946746,0.0006397519,0.0002115768,0.0003323913,0.001891143,0.0002320943,0.000749607],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8247526,"threshold_uncertainty_score":0.9268077,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1301498662001438,"score_gpt":0.3787550552457036,"score_spread":0.2486051890455598,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}