{"id":"W4399424565","doi":"10.48550/arxiv.2406.02465","title":"An Empirical Study into Clustering of Unseen Datasets with Self-Supervised Encoders","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Clustering Algorithms Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Danmarks Grundforskningsfond; Government of Canada; Canadian Institute for Advanced Research","keywords":"Cluster analysis; Computer science; Artificial intelligence; Encoder; Empirical research; Pattern recognition (psychology); Data mining; Machine learning; Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01973509,0.002298759,0.001353446,0.002620887,0.001392199,0.00215769,0.00340418,0.002555543,0.001257826],"category_scores_gemma":[0.08171555,0.0009478864,0.001259522,0.002705017,0.002929694,0.006098903,0.002625305,0.003394342,0.001292528],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003236486,"about_ca_system_score_gemma":0.001321088,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01170332,"about_ca_topic_score_gemma":0.01742047,"domain_scores_codex":[0.9859793,0.007405377,0.0007045968,0.003657325,0.001692422,0.0005609497],"domain_scores_gemma":[0.9410829,0.03094395,0.002861971,0.01776887,0.006339062,0.001003209],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002009691,0.001411906,0.09163068,0.001420681,0.00125333,0.0002750869,0.001308211,0.5405238,0.005779489,0.006847972,0.03500847,0.3125306],"study_design_scores_gemma":[0.00008715437,0.0005652547,0.01278921,0.0001658323,0.00009580601,0.0003278274,0.0004046987,0.9657071,0.008047716,0.008616279,0.003118852,0.00007426042],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8247591,0.007224103,0.1476217,0.002018458,0.0004511243,0.0006084208,0.004730116,0.006295461,0.006291596],"genre_scores_gemma":[0.9057577,0.0007884055,0.07702891,0.0004452385,0.0001136099,0.0002263268,0.01283887,0.0007792099,0.002021693],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01973509,"threshold_uncertainty_score":0.1043704,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07352471772403338,"score_gpt":0.27488410284662,"score_spread":0.2013593851225866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}