{"id":"W2166648275","doi":"10.2196/medinform.4959","title":"NHash: Randomized N-Gram Hashing for Distributed Generation of Validatable Unique Study Identifiers in Multicenter Research","year":2015,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; National Institute of Neurological Disorders and Stroke; National Heart, Lung, and Blood Institute","keywords":"Identifier; Computer science; Hash function; Unique identifier; n-gram; Cryptography; Universal hashing; Theoretical computer science; Hash table; Data mining; Computer security; Computer network; Artificial intelligence; Double hashing","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005910803,0.0001099037,0.0003558423,0.0001104621,0.00005098607,0.00003901441,0.0003192043,0.0002844473,0.000008576993],"category_scores_gemma":[0.005743663,0.0000835556,0.00006230376,0.0002000089,0.0003608852,0.00001393182,0.0001941295,0.0002450828,0.000003228327],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000325057,"about_ca_system_score_gemma":0.0002547964,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006776887,"about_ca_topic_score_gemma":0.0001015038,"domain_scores_codex":[0.9977931,0.0003385289,0.0007839852,0.0001332452,0.0006473096,0.0003038795],"domain_scores_gemma":[0.9988503,0.0002124452,0.000145252,0.000277402,0.0003091758,0.0002053852],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"randomized_trial","study_design_scores_codex":[0.08551158,0.01304833,0.02889465,0.003121885,0.001996396,0.00008622646,0.1108386,0.00110285,0.03453716,0.002039295,0.5676213,0.1512018],"study_design_scores_gemma":[0.4761064,0.004962511,0.0006790765,0.0006434196,0.000151777,0.00002539402,0.09142779,0.280107,0.05766478,0.001033735,0.08603762,0.001160471],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9408719,0.00009105694,0.05701419,0.0002297078,0.000218697,0.001380164,0.0000389862,0.00001610144,0.0001392224],"genre_scores_gemma":[0.9933845,0.00003505432,0.005227125,0.0001039267,0.0001244527,0.0003523406,0.0006660371,0.000009367694,0.00009722116],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4815837,"threshold_uncertainty_score":0.6876116,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1453992460823107,"score_gpt":0.4347469399101895,"score_spread":0.2893476938278788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}