{"id":"W2142905080","doi":"10.1145/2094072.2094073","title":"Word-based self-indexes for natural language text","year":2012,"lang":"en","type":"article","venue":"ACM Transactions on Information Systems","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Fondo Nacional de Desarrollo Científico y Tecnológico; Xunta de Galicia; Ministerio de Ciencia e Innovación","keywords":"Computer science; Word (group theory); Phrase; Search engine indexing; Inverted index; Natural language processing; Space (punctuation); Artificial intelligence; Natural language; Index (typography); Sequence (biology); Full text search; Information retrieval; Search engine; Linguistics; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008716438,0.000644444,0.0007355104,0.004422506,0.0008684902,0.002129955,0.001207211,0.0005546057,0.005454919],"category_scores_gemma":[0.007670279,0.0003854395,0.0005672356,0.005333535,0.0009042929,0.006394855,0.001937554,0.0006300167,0.004634276],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009085741,"about_ca_system_score_gemma":0.001184312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001195757,"about_ca_topic_score_gemma":0.001451055,"domain_scores_codex":[0.9985269,0.0002126498,0.0003198749,0.000175226,0.0006836397,0.00008170219],"domain_scores_gemma":[0.9955047,0.001355988,0.0004379973,0.001408597,0.001158451,0.0001343066],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004080376,0.0001128822,0.001960255,0.001033148,0.00007019085,0.0003387892,0.0008259866,0.004875425,0.05809882,0.08866899,0.01887656,0.8247309],"study_design_scores_gemma":[0.000163582,0.0009333633,0.004137134,0.0003690865,0.0001911528,0.002469921,0.0007588504,0.1887617,0.2900374,0.2139318,0.2979839,0.0002620909],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03738642,0.002924431,0.9331111,0.0003263292,0.0003749731,0.0005629993,0.002710214,0.01317322,0.009430313],"genre_scores_gemma":[0.1211208,0.00155913,0.8613623,0.0001848997,0.0002833638,0.0006134513,0.005630057,0.001295902,0.007950197],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005454919,"threshold_uncertainty_score":0.01824856,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01209488053201561,"score_gpt":0.2518274829218977,"score_spread":0.2397326023898821,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}