{"id":"W2921135034","doi":"10.1111/bjep.12271","title":"The risk–return trade‐off: Performance assessments and cognitive validation of inferences","year":2019,"lang":"en","type":"article","venue":"British Journal of Educational Psychology","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Alberta Advanced Education; University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada; American Institute of Certified Public Accountants","keywords":"Cognition; Psychology; Process (computing); Cognitive interview; Argument (complex analysis); Test (biology); Cognitive psychology; Task (project management); Applied psychology; Think aloud protocol; Cognitive test; Empirical evidence; Social psychology; Computer science; Usability","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3799913,0.001913052,0.001723535,0.006689948,0.002390994,0.01390014,0.004398022,0.007724536,0.007269175],"category_scores_gemma":[0.815451,0.001120366,0.002254234,0.005057889,0.02423928,0.02928206,0.01050552,0.008421773,0.001325479],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007648852,"about_ca_system_score_gemma":0.005616746,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002414776,"about_ca_topic_score_gemma":0.001334153,"domain_scores_codex":[0.545226,0.344378,0.01781561,0.02074754,0.06912557,0.002707239],"domain_scores_gemma":[0.05546158,0.857598,0.03227645,0.03574528,0.0176134,0.00130523],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004318959,0.0005805285,0.117031,0.002365814,0.001960447,0.0005452661,0.01688533,0.01387851,0.00141251,0.4510982,0.005827832,0.3840956],"study_design_scores_gemma":[0.0003226995,0.001620579,0.08089332,0.004340892,0.0006506527,0.001080327,0.004308699,0.04668389,0.004901867,0.8381841,0.01643579,0.0005771358],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2413943,0.01360562,0.5352396,0.04701701,0.001625808,0.001572191,0.000829995,0.0009785413,0.157737],"genre_scores_gemma":[0.9206465,0.000922841,0.07197694,0.002675099,0.0005592116,0.000922562,0.0001719692,0.0002078872,0.001917143],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6200087,"threshold_uncertainty_score":0.7645811,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2487111179184398,"score_gpt":0.5083836971060273,"score_spread":0.2596725791875875,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}