{"id":"W2990321249","doi":"10.1111/coin.12248","title":"A topic‐based term frequency normalization framework to enhance probabilistic information retrieval","year":2019,"lang":"en","type":"article","venue":"Computational Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Wilfrid Laurier University; York University","funders":"Natural Sciences and Engineering Research Council of Canada; Yenepoya Research Centre","keywords":"Computer science; Divergence-from-randomness model; Normalization (sociology); Term Discrimination; Probabilistic logic; Artificial intelligence; Term (time); Language model; Natural language processing; Embedding; Sentence; Vector space model; Information retrieval; Visual Word","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003129125,0.001201277,0.001502881,0.00375075,0.0005063219,0.001395019,0.00197602,0.001368988,0.002128294],"category_scores_gemma":[0.007734466,0.0004426525,0.001627657,0.004365838,0.000689491,0.003783822,0.001096312,0.001501112,0.001494874],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001268878,"about_ca_system_score_gemma":0.001629982,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008849991,"about_ca_topic_score_gemma":0.007259883,"domain_scores_codex":[0.998185,0.000567945,0.0001462153,0.0004589811,0.0005108757,0.000130966],"domain_scores_gemma":[0.9981407,0.0007610609,0.0002106281,0.000274686,0.0005555068,0.00005743368],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003064956,0.0002571693,0.001895285,0.00043859,0.0002449008,0.0001509045,0.0002382496,0.1793318,0.02303608,0.02493711,0.01306291,0.7561005],"study_design_scores_gemma":[0.00002505921,0.00009189727,0.001165518,0.00003037524,0.00007247851,0.0001373831,0.00002860962,0.9766987,0.004714698,0.01107975,0.005898613,0.0000569082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009692645,0.002469305,0.984626,0.0002950042,0.0001234248,0.00007560066,0.0002897561,0.001444592,0.0009836319],"genre_scores_gemma":[0.3959742,0.003389741,0.5890187,0.0004965207,0.0008523762,0.0005979001,0.002178091,0.000478996,0.007013365],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008849991,"threshold_uncertainty_score":0.01759696,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01777396872445287,"score_gpt":0.2958569037642808,"score_spread":0.2780829350398279,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}