{"id":"W4225286402","doi":"10.21203/rs.3.rs-1465079/v1","title":"Features of a FAIR vocabulary","year":2022,"lang":"en","type":"preprint","venue":"Research Square","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Institute for Cancer Research","funders":"Engineering and Physical Sciences Research Council; European Commission; EOSC-Life; European Molecular Biology Laboratory","keywords":"Vocabulary; Computer science; Controlled vocabulary; Fair use; Information retrieval; Artificial intelligence; Natural language processing; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.03449977,0.0008133267,0.001039302,0.008024396,0.004180719,0.009679961,0.00209922,0.002182794,0.00590373],"category_scores_gemma":[0.1335908,0.0006464901,0.001812505,0.004483757,0.01014613,0.02121856,0.009446479,0.003048123,0.0008986411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005139065,"about_ca_system_score_gemma":0.007514837,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009943457,"about_ca_topic_score_gemma":0.006692245,"domain_scores_codex":[0.9614372,0.01354957,0.006517471,0.003953723,0.01273323,0.001808907],"domain_scores_gemma":[0.8960821,0.04701124,0.009343602,0.02130982,0.02334339,0.002909947],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001269022,0.00007910821,0.009816595,0.000309585,0.00006762962,0.0001531343,0.002838875,0.004623941,0.001787463,0.906235,0.005489882,0.06847189],"study_design_scores_gemma":[0.00003194327,0.0001000082,0.003536976,0.000524038,0.00007164761,0.0002381681,0.001867113,0.01543677,0.003660739,0.9315339,0.04291977,0.00007894515],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.084485,0.0008430777,0.8390388,0.006786508,0.0002867207,0.001402077,0.00220905,0.00157403,0.06337466],"genre_scores_gemma":[0.6620792,0.0002537569,0.3279557,0.0009159701,0.0001492648,0.001234603,0.002211745,0.0003543372,0.004845415],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9979008,"threshold_uncertainty_score":0.1824545,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05636369689694107,"score_gpt":0.4119136454336711,"score_spread":0.35554994853673,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}