{"id":"W4393159548","doi":"10.1609/aaai.v38i12.29263","title":"Narrowing the Gap between Supervised and Unsupervised Sentence Representation Learning with Large Language Model","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Fundamental Research Funds for the Central Universities; State Key Laboratory of Software Development Environment; National Natural Science Foundation of China","keywords":"Sentence; Natural language processing; Artificial intelligence; Computer science; Representation (politics); Unsupervised learning; Language model; Linguistics; Machine learning; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007178865,0.0001980795,0.0001999022,0.0001022559,0.0003411267,0.0005854502,0.001171464,0.00006254389,0.00001136837],"category_scores_gemma":[0.0002003097,0.0001184727,0.00006820224,0.0006086101,0.0001702585,0.0006558074,0.0003957646,0.0004661266,0.00001344988],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002832251,"about_ca_system_score_gemma":0.00008695923,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006070148,"about_ca_topic_score_gemma":0.000009511055,"domain_scores_codex":[0.99819,0.00003398072,0.000362399,0.0005789823,0.0005068863,0.0003276902],"domain_scores_gemma":[0.9991052,0.0001731584,0.0001126776,0.0002986866,0.0002430264,0.00006731533],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003618537,0.00003995929,0.002264208,0.0001551692,0.00004738494,0.000003655721,0.03477602,0.003816571,0.05886333,0.8051948,0.00002837988,0.09477435],"study_design_scores_gemma":[0.00002645406,0.00006102756,0.00009235911,0.0003344757,0.0000191308,0.000005776004,0.003979229,0.89633,0.07347809,0.02551466,0.00001122451,0.0001475674],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6701395,0.0001153013,0.3223327,0.005217551,0.0001124044,0.0003468823,0.000003574565,0.0001795305,0.001552562],"genre_scores_gemma":[0.9952193,0.00003123888,0.00429735,0.000125112,0.00007058599,0.00002125258,7.645667e-7,0.00001540678,0.0002189728],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8925135,"threshold_uncertainty_score":0.5645509,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09381631991397431,"score_gpt":0.3127379495608163,"score_spread":0.218921629646842,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}