{"id":"W4200412035","doi":"10.1186/s40537-021-00547-2","title":"Dynamic order Markov model for categorical sequence clustering","year":2021,"lang":"en","type":"article","venue":"Journal Of Big Data","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Sherbrooke","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Hidden Markov model; Cluster analysis; Pattern recognition (psychology); Markov chain; Markov model; Sequence (biology); Categorical variable; Suffix tree; Data mining; Artificial intelligence; Algorithm; Data structure; Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001415644,0.0007647495,0.001465483,0.001699751,0.0009260456,0.001266636,0.003340684,0.001735629,0.004077865],"category_scores_gemma":[0.004908535,0.0005168018,0.001503816,0.002558939,0.001105,0.0023328,0.00127704,0.002098307,0.001400483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002209621,"about_ca_system_score_gemma":0.002009471,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01621412,"about_ca_topic_score_gemma":0.01636552,"domain_scores_codex":[0.9985197,0.0003150997,0.00008644565,0.0005578768,0.0003549667,0.0001659719],"domain_scores_gemma":[0.9977367,0.001265651,0.0002950247,0.000250079,0.0003510301,0.000101554],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001252697,0.00007909784,0.003942625,0.0001725525,0.00008611458,0.0002608136,0.0003930384,0.7449486,0.002738634,0.1809092,0.003733318,0.06261088],"study_design_scores_gemma":[0.00000511762,0.00001137073,0.0002146083,0.000005856948,0.000008022006,0.00004007436,0.00001300845,0.9643211,0.0001757384,0.03403002,0.001163073,0.00001200741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01255827,0.0004349147,0.9834715,0.0003634076,0.00005659505,0.00008828669,0.0006598192,0.000537543,0.001829646],"genre_scores_gemma":[0.6459919,0.001484591,0.3329063,0.0004724401,0.0002221645,0.0008841886,0.004150235,0.0002399829,0.01364826],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01621412,"threshold_uncertainty_score":0.0322395,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1343751196579093,"score_gpt":0.3358048966184071,"score_spread":0.2014297769604978,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}