{"id":"W1992795877","doi":"10.3115/1220835.1220896","title":"Language model-based document clustering using random walks","year":2006,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"McGill University; National Science Foundation","keywords":"Cluster analysis; Computer science; Random walk; Document clustering; Representation (politics); Graph; Artificial intelligence; Dimension (graph theory); Hierarchical clustering; Theoretical computer science; Mathematics; Combinatorics; Statistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001796068,0.00113679,0.001880928,0.003094718,0.0009306911,0.002039361,0.002475609,0.001554981,0.001581127],"category_scores_gemma":[0.006878821,0.0006427529,0.001729854,0.003699772,0.0007992412,0.003569765,0.001438707,0.001291863,0.002018031],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000973848,"about_ca_system_score_gemma":0.001198272,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005178489,"about_ca_topic_score_gemma":0.005574237,"domain_scores_codex":[0.9978404,0.0008473185,0.0001177671,0.0005855763,0.0005000533,0.0001088433],"domain_scores_gemma":[0.9972022,0.001437817,0.0002402997,0.0005257108,0.0005059087,0.00008817549],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002114758,0.0002568606,0.001492016,0.0003157033,0.0002883954,0.0001975815,0.0003880122,0.4425511,0.01264858,0.05579229,0.008877199,0.4769809],"study_design_scores_gemma":[0.00001734529,0.00002460741,0.0001072065,0.000008974756,0.00001900509,0.00005749223,0.00001287076,0.9767671,0.001645962,0.01983228,0.001483928,0.00002323707],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004585294,0.0002251377,0.9933783,0.00009354556,0.0000321657,0.00005462994,0.00008084367,0.001136084,0.0004140153],"genre_scores_gemma":[0.143656,0.0005705598,0.8502968,0.0001897361,0.0001399664,0.00035802,0.001321993,0.0005875891,0.002879336],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005178489,"threshold_uncertainty_score":0.0102967,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01630013396356433,"score_gpt":0.2558534587018628,"score_spread":0.2395533247382985,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}