{"id":"W6885830477","doi":"10.1371/journal.pone.0233732.s001","title":"Agreement test for Newcastle–Ottawa scale scores evaluated by reviewers.","year":2020,"lang":"en","type":"article","venue":"Figshare","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Scale (ratio); Measure (data warehouse); Test (biology); Level of measurement; Reliability (semiconductor)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001604886,0.0002312544,0.0003921621,0.00004844717,0.0002072931,0.0003127371,0.001122964,0.00008376381,0.2617677],"category_scores_gemma":[0.04561981,0.0001667661,0.0002558292,0.0006206058,0.00001790935,0.0002413347,0.0002040796,0.000138112,0.01293312],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006238071,"about_ca_system_score_gemma":0.0001003017,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004453811,"about_ca_topic_score_gemma":0.0000153218,"domain_scores_codex":[0.9955154,0.0001685417,0.0008833101,0.0007906252,0.002262848,0.0003792928],"domain_scores_gemma":[0.9961936,0.001453869,0.0003382157,0.000641193,0.001059358,0.0003137125],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001335817,0.00007421797,0.0008448287,0.0001113655,0.000009283917,9.04386e-7,0.0001142353,0.00006107258,0.0006436136,0.0000020528,0.9782669,0.01985819],"study_design_scores_gemma":[0.0005434988,0.0003573732,0.001263916,0.0005259683,0.00001703082,4.994363e-7,0.0001057324,0.004175024,0.002252494,0.0003588353,0.9901652,0.0002344254],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.003825694,0.01196306,0.001134593,0.08101697,0.0008916362,0.01147677,0.8653879,0.0004383253,0.02386507],"genre_scores_gemma":[0.6844279,0.0002348679,0.007540367,0.07237231,0.003796844,0.00741099,0.2024261,0.0002531777,0.0215375],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.6806021,"threshold_uncertainty_score":0.9878354,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3904297146738943,"score_gpt":0.4112590352549984,"score_spread":0.02082932058110404,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}