{"id":"W3109000035","doi":"10.3386/w25460","title":"Addressing Cross-National Generalizability in Educational Impact Evaluation","year":2019,"lang":"en","type":"preprint","venue":"National Bureau of Economic Research","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generalizability theory; Variety (cybernetics); Cross country; External validity; Quarter (Canadian coin); Political science; Cross-cultural; Econometrics; Psychology; Regional science; Computer science; Economics; Geography; Demographic economics; Artificial intelligence; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.6268005,0.001821939,0.004073977,0.009383632,0.004258936,0.01162054,0.006141243,0.005134783,0.008104191],"category_scores_gemma":[0.8211632,0.001362231,0.005018861,0.0117357,0.01279032,0.01279817,0.01492199,0.00843762,0.001329814],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009376002,"about_ca_system_score_gemma":0.0104346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01315624,"about_ca_topic_score_gemma":0.01161852,"domain_scores_codex":[0.3531944,0.5056979,0.04303892,0.03331342,0.06112357,0.003631793],"domain_scores_gemma":[0.1382063,0.6047602,0.04150554,0.1457969,0.06793674,0.001794417],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001990625,0.000665704,0.2831659,0.01109855,0.01435856,0.0008617697,0.03851868,0.00734566,0.001200846,0.2334537,0.03810445,0.3692356],"study_design_scores_gemma":[0.0008974468,0.002066415,0.2759279,0.02334396,0.005755748,0.001028895,0.03138445,0.01641425,0.006583712,0.4906228,0.145397,0.000577501],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2095925,0.03819514,0.4814333,0.08405615,0.01393821,0.0116471,0.005756306,0.001152027,0.1542293],"genre_scores_gemma":[0.8611698,0.003345395,0.09448066,0.0186995,0.002149242,0.01326541,0.002544597,0.0006578345,0.003687439],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6268005,"threshold_uncertainty_score":0.4602215,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8923654787697515,"score_gpt":0.7649700817209754,"score_spread":0.1273953970487761,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}