{"id":"W4323652773","doi":"10.1002/bsl.2609","title":"From “below chance” to “a single error is one too many”: Evaluating various thresholds for invalid performance on two forced choice recognition tests","year":2023,"lang":"en","type":"article","venue":"Behavioral Sciences & the Law","topic":"Traumatic Brain Injury Research","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Windsor","funders":"","keywords":"Malingering; Psychology; Binomial distribution; Statistics; Audiology; Two-alternative forced choice; Test (biology); Cognitive psychology; Clinical psychology; Mathematics; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002103726,0.0002084181,0.0003024465,0.0001474828,0.0008337898,0.000183125,0.0005626291,0.00008268499,0.0003163027],"category_scores_gemma":[0.0002422992,0.0001489516,0.00009828821,0.001192017,0.0003800761,0.0002942515,0.0001587996,0.000248978,0.0004785026],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001697445,"about_ca_system_score_gemma":0.0001321462,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001689611,"about_ca_topic_score_gemma":0.0006313419,"domain_scores_codex":[0.9968387,0.00008764353,0.0003664974,0.0006153455,0.001339536,0.0007522399],"domain_scores_gemma":[0.998447,0.0005382099,0.000116566,0.0005021544,0.0002033975,0.0001926784],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001416107,0.001686308,0.02131693,0.0002776791,0.00008331228,0.00002318022,0.01545344,0.0003614626,0.4304464,0.001036958,0.008962665,0.5189356],"study_design_scores_gemma":[0.01574974,0.05076759,0.3820555,0.004869882,0.001252442,0.00008713235,0.005597056,0.08787006,0.4194332,0.02188439,0.007690106,0.002742854],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9840224,0.00001669982,0.00003635649,0.01121491,0.0003466689,0.002062412,0.00009342858,0.0001720038,0.002035107],"genre_scores_gemma":[0.990631,0.000003493862,0.004470242,0.002667102,0.0003889447,0.0005743764,0.00006539555,0.0000318382,0.00116759],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5161927,"threshold_uncertainty_score":0.6412921,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5522029002402319,"score_gpt":0.4986658389475107,"score_spread":0.05353706129272118,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}