{"id":"W2597363143","doi":"10.1101/108555","title":"Unmet Needs for Analyzing Biological Big Data: A Survey of 704 NSF Principal Investigators","year":2017,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genetics, Bioinformatics, and Biomedical Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Science Foundation","keywords":"Principal (computer security); Workflow; Data science; Data archive; Big data; Data management; Computer science; Publication; Biological data; Political science; Bioinformatics; Data mining; Database; Biology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07909369,0.0005189587,0.0007856449,0.007255699,0.00323348,0.006568579,0.003269082,0.001954755,0.007271959],"category_scores_gemma":[0.1268747,0.0009228697,0.0008260714,0.01019122,0.002508524,0.006198219,0.01094464,0.00373077,0.001835382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003540223,"about_ca_system_score_gemma":0.01807446,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004970127,"about_ca_topic_score_gemma":0.005989098,"domain_scores_codex":[0.9653884,0.01015896,0.003789188,0.002569459,0.01447591,0.003618121],"domain_scores_gemma":[0.8246778,0.04947282,0.02043382,0.00987864,0.05154174,0.04399524],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0004882104,0.0002918704,0.2613643,0.001493533,0.0002134708,0.0004346945,0.004578615,0.0003967579,0.001115183,0.01246583,0.2573084,0.4598491],"study_design_scores_gemma":[0.0002133946,0.0003750153,0.2780737,0.004031745,0.0002898762,0.001846639,0.01934111,0.001798881,0.001803395,0.02288527,0.6691123,0.0002288354],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.307449,0.07732001,0.01579263,0.5446845,0.003212596,0.0005735355,0.006016901,0.001591406,0.04335947],"genre_scores_gemma":[0.7429408,0.09197721,0.03620927,0.1015511,0.004365452,0.001490125,0.01229798,0.0009820135,0.008186041],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9209063,"threshold_uncertainty_score":0.4182924,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1014270074696426,"score_gpt":0.3087143420382056,"score_spread":0.207287334568563,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}