{"id":"W2062618525","doi":"10.1186/1472-6947-8-18","title":"Script Concordance Tests: Guidelines for Construction","year":2008,"lang":"en","type":"article","venue":"BMC Medical Informatics and Decision Making","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":253,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Université du Québec à Montréal","funders":"","keywords":"Concordance; Context (archaeology); Computer science; Test (biology); Task (project management); Process (computing); Health informatics; Quality (philosophy); Natural language processing; Information retrieval; Data science; Artificial intelligence; Medicine; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08508883,0.004353638,0.003329171,0.01275918,0.002522883,0.005774057,0.00979643,0.003977681,0.09092414],"category_scores_gemma":[0.324451,0.003969053,0.004244547,0.01151907,0.003384081,0.004450014,0.0068941,0.006692898,0.04096746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002858268,"about_ca_system_score_gemma":0.01195887,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004869482,"about_ca_topic_score_gemma":0.007068465,"domain_scores_codex":[0.8756766,0.0864459,0.01758954,0.003574696,0.01562801,0.001085286],"domain_scores_gemma":[0.6716565,0.2308432,0.01219466,0.02954268,0.05329864,0.002464339],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00067837,0.0007996438,0.003516268,0.005040874,0.0003847802,0.0006662523,0.002591187,0.003506123,0.0009501748,0.04422956,0.4606302,0.4770065],"study_design_scores_gemma":[0.001057382,0.0007067614,0.007020592,0.01052116,0.0003737683,0.002147609,0.001721008,0.02945364,0.005684817,0.1730547,0.7678096,0.0004489465],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002270638,0.002388247,0.9088609,0.002174008,0.001717312,0.01883403,0.02147745,0.02056717,0.0217102],"genre_scores_gemma":[0.007278075,0.0008882354,0.9454032,0.0004619668,0.0002214768,0.02924158,0.007413739,0.004066512,0.005025213],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.09092414,"threshold_uncertainty_score":0.4499981,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1298384051838863,"score_gpt":0.4334338565504915,"score_spread":0.3035954513666053,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}