{"id":"W2147671191","doi":"10.1145/1148170.1148233","title":"Hybrid index maintenance for growing text collections","year":2006,"lang":"en","type":"article","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Search engine indexing; Computer science; Merge (version control); Index (typography); Monotonic function; Auxiliary memory; Information retrieval; Zipf's law; Data mining; World Wide Web; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001290673,0.0005998445,0.0009195072,0.00242704,0.0007567884,0.001451631,0.002321295,0.0006678817,0.00156679],"category_scores_gemma":[0.00839185,0.0004294964,0.0005607174,0.002734407,0.0006451537,0.003332622,0.00191464,0.0008038598,0.0009764322],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005295635,"about_ca_system_score_gemma":0.0006485364,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001093665,"about_ca_topic_score_gemma":0.001521412,"domain_scores_codex":[0.998774,0.0002337514,0.0001284667,0.0001819123,0.0005988489,0.00008298324],"domain_scores_gemma":[0.9934157,0.002108987,0.0006972824,0.002590446,0.001002075,0.0001854621],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003134245,0.0002035859,0.003165553,0.0003079222,0.0001033633,0.0002324295,0.0005229898,0.04330308,0.05924089,0.01657581,0.00646615,0.8695649],"study_design_scores_gemma":[0.0001105371,0.000448224,0.002508392,0.00004406884,0.0001375132,0.00135672,0.0002086947,0.8680118,0.08405605,0.02598113,0.01705479,0.00008212935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04547602,0.0005956145,0.9492392,0.00009293568,0.00002868301,0.0001129104,0.0001491606,0.002619492,0.001686012],"genre_scores_gemma":[0.242607,0.0003290691,0.7530462,0.00008941654,0.00008917724,0.000245169,0.000625082,0.0004301004,0.0025389],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00242704,"threshold_uncertainty_score":0.006825864,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008690065081308028,"score_gpt":0.22291227247964,"score_spread":0.2142222073983319,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}