{"id":"W1985629392","doi":"10.1198/073500102288618702","title":"Was There a Riverside Miracle? A Hierarchical Framework for Evaluating Programs With Grouped Data","year":2003,"lang":"en","type":"article","venue":"Journal of Business and Economic Statistics","topic":"Advanced Causal Inference Techniques","field":"Mathematics","cited_by":41,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Connaught Fund","keywords":"Pooling; Computer science; Econometrics; Independence (probability theory); Statistics; Mathematics; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007225352,0.000147646,0.0003819267,0.0000587032,0.00009010376,0.00008262895,0.0002151034,0.00007274614,0.00004172069],"category_scores_gemma":[0.001698561,0.0001122937,0.0000220031,0.00005644591,0.000111924,0.0002911728,0.00005263204,0.0002090888,8.661637e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006560383,"about_ca_system_score_gemma":0.0001884578,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000009647721,"about_ca_topic_score_gemma":0.00003421968,"domain_scores_codex":[0.9990024,0.00004536578,0.0004689542,0.0001734249,0.0001097055,0.0002001533],"domain_scores_gemma":[0.9974527,0.001334439,0.0006062245,0.0002806711,0.0002426047,0.00008341404],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004024948,0.0001986979,0.004272362,0.0003656333,0.0002132976,0.00004516476,0.0004174169,0.0001111601,0.00003926756,0.9330869,0.001270548,0.05957703],"study_design_scores_gemma":[0.0008363834,0.0004845832,0.001140605,0.0003133944,0.0001636652,0.0001954164,0.0003049833,0.005135118,0.00003253983,0.9903508,0.0008518135,0.0001906771],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09141849,0.00007989345,0.9078508,0.0001062514,0.00008674039,0.0002551498,0.000113279,0.00001606871,0.00007334376],"genre_scores_gemma":[0.1459329,0.0001016289,0.8537987,0.00002728969,0.00007614698,0.000008139862,0.00001171098,0.00002691279,0.00001659739],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05938635,"threshold_uncertainty_score":0.4579205,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2549804226516588,"score_gpt":0.4273379998675451,"score_spread":0.1723575772158863,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}