{"id":"W4398147653","doi":"10.3386/w32456","title":"A Sharp Test for the Judge Leniency Design","year":2024,"lang":"en","type":"report","venue":"National Bureau of Economic Research","topic":"Law, Economics, and Judicial Systems","field":"Economics, Econometrics and Finance","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; HEC Montréal","funders":"","keywords":"Test (biology); Computer science; Law; Political science; Geology; Paleontology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.01734781,0.0003202435,0.001162393,0.001154579,0.0003233788,0.0003543936,0.00134004,0.0006218371,0.0009783099],"category_scores_gemma":[0.003857275,0.0003567832,0.0006417515,0.000275052,0.0005515894,0.0002517045,0.0002641479,0.0008030574,0.002880103],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003082671,"about_ca_system_score_gemma":0.002907295,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004112941,"about_ca_topic_score_gemma":0.0003189977,"domain_scores_codex":[0.9954539,0.00005828571,0.002238936,0.001160051,0.0003415819,0.0007472293],"domain_scores_gemma":[0.9921045,0.005169735,0.0009808741,0.0006861713,0.0009182174,0.0001405062],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002390118,0.00006019541,0.000421183,0.0002915481,0.0003250676,0.000001372385,0.00008702249,0.0002747101,0.000004704448,0.7070612,0.2909596,0.0004895159],"study_design_scores_gemma":[0.0003036978,0.0001436649,0.0001214323,0.00008956423,0.00001757762,0.00001072638,0.00003589832,0.003455908,0.0000349017,0.7030024,0.292492,0.0002922396],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.0001103386,0.02214783,0.0007510387,0.0037048,0.005006007,0.00286385,0.003902517,0.00004585048,0.9614677],"genre_scores_gemma":[0.9202049,0.01001318,0.0005057799,0.0001458713,0.008221786,0.002293504,0.0005889489,0.0003101304,0.05771587],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9200946,"threshold_uncertainty_score":0.9999349,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4988921677658905,"score_gpt":0.4627913122661106,"score_spread":0.03610085549977987,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}