{"id":"W2050611670","doi":"10.1371/journal.pcbi.1000391","title":"How to Get the Most out of Your Curation Effort","year":2009,"lang":"en","type":"article","venue":"PLoS Computational Biology","topic":"Topic Modeling","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"U.S. National Library of Medicine; National Institute of General Medical Sciences; National Institutes of Health; National Science Foundation","keywords":"Annotation; Computer science; Data curation; Sentence; Probabilistic logic; Set (abstract data type); Natural language processing; Task (project management); Information retrieval; Artificial intelligence; Machine learning; Data science; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02874864,0.001963978,0.002270555,0.004093243,0.003057362,0.008295806,0.002908372,0.00366965,0.006020083],"category_scores_gemma":[0.1267374,0.001396306,0.001725226,0.004115978,0.002161376,0.01313078,0.003584188,0.002648334,0.009386887],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001617573,"about_ca_system_score_gemma":0.004412586,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00499262,"about_ca_topic_score_gemma":0.008634796,"domain_scores_codex":[0.9749274,0.01193487,0.001587917,0.00570954,0.004987231,0.0008529749],"domain_scores_gemma":[0.9275143,0.03456102,0.005719194,0.01575757,0.01418416,0.002263872],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006369033,0.0002527375,0.01811993,0.002175858,0.0005438358,0.0004102197,0.006839093,0.01250706,0.01939756,0.02343614,0.166114,0.7495665],"study_design_scores_gemma":[0.0003319077,0.0005378872,0.02202657,0.001945483,0.0007662181,0.002161119,0.0110093,0.1532643,0.04529071,0.2997809,0.4616258,0.001259844],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03903454,0.00425347,0.8849812,0.03717446,0.001169435,0.0005760849,0.002726512,0.01233059,0.01775365],"genre_scores_gemma":[0.2028225,0.002144511,0.7778029,0.003500672,0.000564202,0.0005299975,0.002208591,0.003861611,0.006565021],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02874864,"threshold_uncertainty_score":0.1520391,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04654170247765851,"score_gpt":0.2859339761740192,"score_spread":0.2393922736963607,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}