{"id":"W1571050549","doi":"10.2307/25148806","title":"Enhancing Information Retrieval Through Statistical Natural Language Processing: A Study of Collocation Indexing1","year":2007,"lang":"en","type":"article","venue":"MIS Quarterly","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; University of Alberta","funders":"","keywords":"Search engine indexing; Collocation (remote sensing); Information retrieval; Computer science; Automatic indexing; Natural language; Natural language processing; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01452833,0.0006724813,0.001365204,0.003170991,0.0009622978,0.002478163,0.001363554,0.001074368,0.001114117],"category_scores_gemma":[0.1105913,0.0005775856,0.0005544852,0.006212157,0.002742441,0.007378506,0.001299477,0.001242517,0.0004099669],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001759907,"about_ca_system_score_gemma":0.001443721,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004301393,"about_ca_topic_score_gemma":0.002902266,"domain_scores_codex":[0.9892809,0.008246376,0.0003208521,0.0005143136,0.00141117,0.0002263434],"domain_scores_gemma":[0.803671,0.1812515,0.004563876,0.004881975,0.005152463,0.0004791624],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001032438,0.001697687,0.02475939,0.00175218,0.0003233771,0.0003482071,0.003300948,0.1354739,0.01930867,0.1059539,0.005138175,0.7009112],"study_design_scores_gemma":[0.0001966718,0.0008273362,0.01178654,0.0000785225,0.0001618577,0.0003370398,0.0007560598,0.8928095,0.007208891,0.08114056,0.004593762,0.0001032493],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4718409,0.01218,0.4940012,0.004181066,0.0000875959,0.0006332773,0.0001542477,0.000752781,0.01616896],"genre_scores_gemma":[0.8253427,0.002984202,0.1698596,0.0002962325,0.0001617504,0.0002072937,0.0001222208,0.0001046534,0.0009213932],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01452833,"threshold_uncertainty_score":0.07683402,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01072236122967445,"score_gpt":0.2951325797756313,"score_spread":0.2844102185459569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}