{"id":"W4382893966","doi":"10.5121/ijnlc.2023.12301","title":"Testing Different Log Bases for Vector Model Weighting Technique","year":2023,"lang":"en","type":"article","venue":"International Journal on Natural Language Computing","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Workers Compensation Board of Alberta; Alberta Biodiversity Monitoring Institute","funders":"","keywords":"Weighting; tf–idf; Computer science; Term (time); Logarithm; Information retrieval; Vector space model; Range (aeronautics); Data mining; Base (topology); Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004577123,0.0001873589,0.0001720122,0.000326517,0.0003259186,0.0004703484,0.001365805,0.00004731164,0.000004235636],"category_scores_gemma":[0.0005979783,0.0001397026,0.0001302922,0.0002561337,0.00001472083,0.0003911544,0.0005946038,0.0004555034,0.00001261534],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001337783,"about_ca_system_score_gemma":0.00004873713,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006492649,"about_ca_topic_score_gemma":7.465033e-7,"domain_scores_codex":[0.998347,0.00004887649,0.0003629001,0.000327436,0.0005713255,0.0003424586],"domain_scores_gemma":[0.9982783,0.000826042,0.0002726825,0.0002028418,0.0003250667,0.00009508151],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008288426,0.0001715109,0.0004032244,0.00003796338,0.0001512552,0.0009003307,0.001306114,0.0459772,0.09716363,0.02073966,0.00672875,0.8263375],"study_design_scores_gemma":[0.0003551221,0.00005901672,0.0009074157,0.0002991143,0.000003259207,0.000208731,0.00003185089,0.9909823,0.005642815,0.001022855,0.0003046134,0.0001829089],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09464051,0.0001608732,0.9005948,0.001289327,0.002482841,0.000196491,0.00002262736,0.0004395299,0.0001730043],"genre_scores_gemma":[0.8336968,0.00000360136,0.1645387,0.0005380218,0.001070927,0.000007209216,0.00002551754,0.00001629986,0.0001028877],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9450051,"threshold_uncertainty_score":0.5696908,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02888458705003525,"score_gpt":0.3181266725152209,"score_spread":0.2892420854651856,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}