{"id":"W4396646664","doi":"10.31234/osf.io/z4yqe","title":"Clarifying the reliability paradox: poor test-retest reliability attenuates group differences","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Advanced Causal Inference Techniques","field":"Mathematics","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Centre for Addiction and Mental Health","funders":"","keywords":"Reliability (semiconductor); Reliability engineering; Test (biology); Group (periodic table); Psychology; Computer science; Engineering; Chemistry; Physics; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1542116,0.001346135,0.001822049,0.001978542,0.001397402,0.002841345,0.002589373,0.002309734,0.002475026],"category_scores_gemma":[0.4543578,0.0006903767,0.00159213,0.00181388,0.01099587,0.006487648,0.004902225,0.005486865,0.0006371441],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001298241,"about_ca_system_score_gemma":0.002805051,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001479192,"about_ca_topic_score_gemma":0.0008715786,"domain_scores_codex":[0.8796202,0.100516,0.002775669,0.007513119,0.008738284,0.0008367047],"domain_scores_gemma":[0.4233659,0.5001431,0.01763921,0.04681451,0.0109164,0.001120885],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001766088,0.0003938819,0.07814535,0.001777517,0.002585786,0.00113695,0.0182436,0.01991178,0.01590385,0.608431,0.01113897,0.2405652],"study_design_scores_gemma":[0.0003670054,0.001266749,0.03071095,0.0004345982,0.0008575144,0.000856768,0.001119999,0.04804928,0.01378787,0.890303,0.01210903,0.0001370518],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1414652,0.002899897,0.8190151,0.02379554,0.0007461538,0.000361295,0.0002712304,0.0007932811,0.01065226],"genre_scores_gemma":[0.8039874,0.0008039601,0.1879797,0.004224103,0.0006027081,0.0007719753,0.0001728346,0.0003514473,0.001105961],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8457884,"threshold_uncertainty_score":0.8155584,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1347329622286347,"score_gpt":0.3928317611641184,"score_spread":0.2580987989354837,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}