{"id":"W4415372161","doi":"10.21203/rs.3.rs-7830468/v1","title":"Explaining Black-Box Models Through Statistical Inference","year":2025,"lang":"","type":"preprint","venue":"Research Square","topic":"Data Analysis with R","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University","funders":"","keywords":"Spurious relationship; Heuristics; Pairwise comparison; Statistical inference; Heuristic; Statistical model; Inference; Statistical hypothesis testing; Stability (learning theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02076879,0.001692684,0.002453893,0.002007609,0.000988564,0.003893193,0.003119437,0.002682887,0.0093408],"category_scores_gemma":[0.1313173,0.001750927,0.002434383,0.002177726,0.003695673,0.008106691,0.003156436,0.006092778,0.001917438],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001400416,"about_ca_system_score_gemma":0.002474862,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003938266,"about_ca_topic_score_gemma":0.003507899,"domain_scores_codex":[0.9899566,0.007447828,0.0002785796,0.001424819,0.0006536667,0.0002384795],"domain_scores_gemma":[0.8552967,0.1306056,0.003444845,0.008562487,0.001425489,0.000664952],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000322099,0.0001187798,0.005703494,0.0004479527,0.0005703146,0.0002579184,0.0005625596,0.3042793,0.001108607,0.6057066,0.00753346,0.07338906],"study_design_scores_gemma":[0.00002197076,0.00001071,0.0002514425,0.00003290031,0.00003414774,0.00002204958,0.00001527744,0.5002138,0.0002259482,0.4983073,0.000850264,0.00001419212],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00510629,0.0002439982,0.9928901,0.0005310286,0.00003675249,0.00001981841,0.0001823861,0.0005027834,0.0004868902],"genre_scores_gemma":[0.3628207,0.00119744,0.6275363,0.0007752305,0.0003583629,0.0004079153,0.001482113,0.001183685,0.004238448],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02076879,"threshold_uncertainty_score":0.1098372,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1905112513852044,"score_gpt":0.4710558457389411,"score_spread":0.2805445943537367,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}