{"id":"W4390542458","doi":"10.1007/s13369-023-08567-1","title":"WASM: A Dataset for Hashtag Recommendation for Arabic Tweets","year":2024,"lang":"en","type":"article","venue":"Arabian Journal for Science and Engineering","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Social media; Information retrieval; Arabic; Task (project management); Categorization; Benchmark (surveying); Microblogging; Natural language processing; Artificial intelligence; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004866338,0.001856107,0.0008049557,0.00516275,0.001073304,0.001019413,0.0009365621,0.001649803,0.01009168],"category_scores_gemma":[0.003263282,0.0002924672,0.0008355738,0.004240854,0.0002574847,0.001131947,0.001099408,0.0009606846,0.01887662],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009820696,"about_ca_system_score_gemma":0.001555711,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0239096,"about_ca_topic_score_gemma":0.04118613,"domain_scores_codex":[0.9993303,0.000118733,0.0000929595,0.0001453604,0.0001967784,0.0001159402],"domain_scores_gemma":[0.998747,0.000290427,0.0001210057,0.000209943,0.0004451514,0.0001865438],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008276544,0.0004306757,0.01440071,0.002472275,0.0002081613,0.0005086823,0.0005626592,0.002389079,0.00936232,0.001282369,0.9037172,0.06383822],"study_design_scores_gemma":[0.0003750595,0.0003595023,0.06866156,0.0005587113,0.0002137659,0.000894155,0.002125083,0.02426442,0.01218354,0.002059301,0.8880404,0.0002645793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.02343009,0.0008886257,0.001512878,0.0003726974,0.0002984562,0.000283867,0.9662493,0.003321159,0.003643],"genre_scores_gemma":[0.01684982,0.0002818675,0.005102272,0.0001238273,0.00007361764,0.0003133191,0.974096,0.0001034289,0.003055857],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0239096,"threshold_uncertainty_score":0.04754084,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0375174136116534,"score_gpt":0.2986276063840521,"score_spread":0.2611101927723987,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}