{"id":"W2150519824","doi":"10.1080/01421590601032427","title":"Composition of the panel of reference for concordance tests: Do teaching functions have an impact on examinees’ ranks and absolute scores?","year":2007,"lang":"en","type":"article","venue":"Medical Teacher","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Concordance; Ranking (information retrieval); Test (biology); Medicine; Ambiguity; Family medicine; Psychology; Statistics; Mathematics; Computer science; Internal medicine; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00176959,0.0001192717,0.0003355978,0.0000441753,0.00006611992,0.000005785667,0.0001063656,0.000173038,0.0001621518],"category_scores_gemma":[0.02572992,0.00006830018,0.00009939649,0.00006610834,0.0003327128,0.00002875353,0.00003180441,0.0005202383,0.000001485718],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004072539,"about_ca_system_score_gemma":0.00009247763,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002861737,"about_ca_topic_score_gemma":0.00003060589,"domain_scores_codex":[0.9986538,0.00009474638,0.0004117964,0.0002082951,0.0004359829,0.0001953448],"domain_scores_gemma":[0.9937347,0.005404695,0.0001613771,0.0003268599,0.00008874968,0.0002836812],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002397685,0.003212128,0.817408,0.0001658938,0.0002085075,0.00002418772,0.001511054,0.00002597747,0.003252229,0.001362156,0.005004468,0.1654277],"study_design_scores_gemma":[0.003157874,0.001776235,0.9885949,0.002809691,0.0001443962,0.00003915202,0.0003096516,0.002233794,0.0002844093,0.0002409642,0.0003056717,0.0001032102],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9937978,0.0001608894,0.002392847,0.000499575,0.0001168353,0.0003056863,0.00001841533,0.00002132897,0.002686569],"genre_scores_gemma":[0.9987511,0.00001945751,0.0003112769,0.0002554558,0.0001944236,0.00000945838,0.00003302728,0.0000135181,0.0004122858],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.171187,"threshold_uncertainty_score":0.9824768,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0828956857860335,"score_gpt":0.4049771565153758,"score_spread":0.3220814707293423,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}