{"id":"W3017715546","doi":"10.1162/dint_a_00058","title":"The Semantic Data Dictionary – An Approach for Describing and Annotating Data","year":2020,"lang":"en","type":"article","venue":"Data Intelligence","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"CARE Canada; Privacy Analytics (Canada)","funders":"National Institute of Environmental Health Sciences","keywords":"Computer science; Natural language processing; Data dictionary; Information retrieval; Artificial intelligence; World Wide Web; Metadata","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03443132,0.001823797,0.002418643,0.02072649,0.005183016,0.01395074,0.006920692,0.004153549,0.005078309],"category_scores_gemma":[0.06554477,0.002299914,0.004308238,0.02462066,0.008127064,0.02814,0.0157383,0.009992349,0.004106305],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004709535,"about_ca_system_score_gemma":0.01622023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01372815,"about_ca_topic_score_gemma":0.01337762,"domain_scores_codex":[0.9608474,0.01607397,0.01031572,0.00479681,0.007190414,0.0007756839],"domain_scores_gemma":[0.9289941,0.02610219,0.004969495,0.02778645,0.01047864,0.001669238],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001643563,0.0001310327,0.002398757,0.001849735,0.0002370539,0.0005915393,0.008423929,0.004739582,0.004845076,0.7862682,0.0401779,0.1501727],"study_design_scores_gemma":[0.00004139676,0.00006212531,0.0007049228,0.001605872,0.0001116309,0.0007696553,0.003358992,0.01374531,0.004673659,0.3228967,0.6518505,0.0001792178],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001125762,0.0005026071,0.9827316,0.002434977,0.0003308161,0.0007745255,0.004640388,0.001840637,0.005618762],"genre_scores_gemma":[0.01327961,0.0009341579,0.9684631,0.001407623,0.0001489803,0.001328205,0.01138343,0.0009005755,0.002154369],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03443132,"threshold_uncertainty_score":0.1820924,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3725294609797752,"score_gpt":0.3740642515334556,"score_spread":0.001534790553680387,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}