{"id":"W4388524631","doi":"10.1136/bmjresp-2023-001942","title":"Performance evaluation of human cough annotators: optimal metrics and sex differences","year":2023,"lang":"en","type":"article","venue":"BMJ Open Respiratory Research","topic":"Respiratory and Cough-Related Research","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Generalitat de Catalunya; Ministerio de Ciencia e Innovación; Centres de Recerca de Catalunya; Patrick J. McGovern Foundation","keywords":"Medicine; Gold standard (test); Limits of agreement; Annotation; Correlation; Linear correlation; Pearson product-moment correlation coefficient; Audiology; Artificial intelligence; Statistics; Computer science; Nuclear medicine; Mathematics; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0338243,0.0009606339,0.001167922,0.001661255,0.001148344,0.002459908,0.001162916,0.001519588,0.001910522],"category_scores_gemma":[0.08224229,0.0003849433,0.000831422,0.001109296,0.001381949,0.001513017,0.002172634,0.0005383013,0.001019654],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001050477,"about_ca_system_score_gemma":0.0009799827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002100978,"about_ca_topic_score_gemma":0.002945021,"domain_scores_codex":[0.9764783,0.01156779,0.002487794,0.005098229,0.003607745,0.0007601419],"domain_scores_gemma":[0.9215601,0.04732367,0.008128637,0.005227113,0.0161704,0.001590054],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01128497,0.0004194088,0.5330808,0.002515727,0.001396207,0.0005257169,0.01086948,0.01014076,0.0428999,0.001788999,0.007594331,0.3774838],"study_design_scores_gemma":[0.0003496738,0.005082932,0.7254437,0.0008738666,0.001062204,0.003305845,0.006442846,0.1799962,0.05325058,0.009723436,0.01398898,0.0004798121],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8193157,0.006662012,0.1614958,0.0006300332,0.0004319678,0.001084474,0.001887359,0.0018187,0.00667403],"genre_scores_gemma":[0.9398295,0.0003923192,0.05589299,0.0001401035,0.00009823857,0.0007884389,0.00121945,0.0003961745,0.001242713],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9661757,"threshold_uncertainty_score":0.1788822,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5294249097745025,"score_gpt":0.5562476579354084,"score_spread":0.02682274816090591,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}