{"id":"W4243378373","doi":"10.31234/osf.io/v6crg","title":"Validation as Evaluating Desired and Undesired Effects: Insights from Cross-Classified Mixed Effects Model","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Reliability (semiconductor); Variance (accounting); Computer science; Reliability engineering; Variance components; Validity; Data mining; Statistics; Psychometrics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.006405668,0.0006938181,0.001202187,0.0003089846,0.0004860483,0.003703459,0.00119878,0.0007478073,0.0002297921],"category_scores_gemma":[0.01779376,0.0005188284,0.000387079,0.0004677318,0.0002368773,0.0005744494,0.00175779,0.0007208706,0.0001088395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003216127,"about_ca_system_score_gemma":0.000575717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003121279,"about_ca_topic_score_gemma":0.00006780762,"domain_scores_codex":[0.987538,0.002739894,0.00174397,0.002551541,0.004969537,0.0004570559],"domain_scores_gemma":[0.9872434,0.008111195,0.0008720161,0.002099257,0.001359992,0.0003141905],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008093573,0.001545465,0.01924853,0.0019235,0.001249011,0.0001283975,0.007352286,0.2617334,0.5199486,0.006191446,0.002202421,0.1776676],"study_design_scores_gemma":[0.00163165,0.0001679602,0.03279775,0.000685298,0.0002038729,0.000001102474,0.0003315657,0.4299174,0.1388443,0.3947174,0.00001854673,0.0006832229],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8908841,0.00094009,0.101074,0.0002164777,0.001916068,0.001846279,0.00001135175,0.0001346187,0.002977048],"genre_scores_gemma":[0.9690508,0.00006946437,0.02874959,0.0003626942,0.0001524638,0.0002439912,0.0001502843,0.00003644781,0.001184224],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3885259,"threshold_uncertainty_score":0.9997264,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2892285741810964,"score_gpt":0.4363073635794985,"score_spread":0.1470787893984021,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}