{"id":"W4318150379","doi":"10.1371/journal.pone.0274429","title":"Predicting reliability through structured expert elicitation with the repliCATS (Collaborative Assessments for Trustworthy Science) process","year":2023,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Delphi Technique in Research","field":"Social Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University; University of British Columbia","funders":"Advanced Research Projects Agency; Defense Advanced Research Projects Agency","keywords":"Expert elicitation; Generalizability theory; Computer science; Process (computing); Resource (disambiguation); Replication (statistics); Data science; Reliability (semiconductor); Delphi method; Scalability; Crowdsourcing; Trustworthiness; Data mining; Management science; Artificial intelligence; Psychology; Statistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4615444,0.002604137,0.002336968,0.009786537,0.003048517,0.006111353,0.003965945,0.003373155,0.005147827],"category_scores_gemma":[0.7287483,0.002104858,0.003731759,0.006371185,0.005691163,0.007623158,0.009862159,0.004425904,0.002450558],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003879981,"about_ca_system_score_gemma":0.01104464,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002537321,"about_ca_topic_score_gemma":0.003347926,"domain_scores_codex":[0.4553745,0.4382662,0.03242749,0.01779827,0.0542876,0.001845941],"domain_scores_gemma":[0.1207424,0.6918972,0.0541628,0.07250965,0.05937554,0.001312418],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00253452,0.001438444,0.06390792,0.01033658,0.002370753,0.0009125473,0.08081846,0.03063185,0.01284429,0.06612707,0.01217379,0.7159038],"study_design_scores_gemma":[0.002566786,0.006548217,0.07187266,0.008662969,0.002100842,0.001563297,0.01852969,0.277784,0.04445871,0.4871411,0.07680473,0.00196712],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05479161,0.0006587339,0.9228221,0.001058702,0.0001901503,0.01289009,0.0005048212,0.0008308218,0.006252921],"genre_scores_gemma":[0.1881755,0.0003948537,0.7928195,0.000385857,0.0001130531,0.01683713,0.0004094651,0.0001770453,0.0006875884],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.5384556,"threshold_uncertainty_score":0.6640117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1889548488658979,"score_gpt":0.4845837112621013,"score_spread":0.2956288623962035,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}