{"id":"W4399880449","doi":"10.1093/rfs/hhae029","title":"Computational Reproducibility in Finance: Evidence from 1,000 Tests","year":2024,"lang":"en","type":"article","venue":"Review of Financial Studies","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"Cascades (Canada)","funders":"Riksbankens Jubileumsfond; Vermont Agency of Natural Resources; Knut och Alice Wallenbergs Stiftelse; Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Agence Nationale de la Recherche","keywords":"Reproducibility; Overconfidence effect; Coding (social sciences); Computer science; Code (set theory); Psychology; Statistics; Mathematics; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3418146,0.0009515914,0.002447617,0.006116671,0.00270662,0.006176298,0.005980386,0.004064117,0.003849707],"category_scores_gemma":[0.853677,0.001028127,0.00382625,0.01052414,0.0144367,0.006499283,0.00511692,0.003872081,0.0009094155],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002439482,"about_ca_system_score_gemma":0.003102352,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004212947,"about_ca_topic_score_gemma":0.002346529,"domain_scores_codex":[0.5752018,0.3072483,0.02960905,0.0297424,0.0542815,0.003916905],"domain_scores_gemma":[0.01955312,0.8770543,0.03720465,0.05094814,0.01413738,0.001102379],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00471179,0.0007653057,0.8007309,0.002552219,0.0134066,0.001104137,0.006304642,0.02552523,0.0006061663,0.02953086,0.01732619,0.09743591],"study_design_scores_gemma":[0.001922311,0.002837196,0.6838293,0.002628232,0.006088748,0.001899288,0.003993723,0.1208598,0.007213885,0.1351553,0.03299556,0.0005766261],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8986114,0.01639752,0.05486988,0.01221498,0.001239883,0.0003872695,0.003183185,0.0006323699,0.01246362],"genre_scores_gemma":[0.9943321,0.0004390869,0.003184618,0.0004563516,0.000288403,0.0001040831,0.0008677569,0.0001144065,0.0002131376],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6581854,"threshold_uncertainty_score":0.8116598,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09613543887438811,"score_gpt":0.3888556883459117,"score_spread":0.2927202494715235,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}