{"id":"W4402293703","doi":"10.1002/9781119912965.ch9","title":"Data Indexing and Filtering Techniques for Big Data Systems","year":2024,"lang":"en","type":"other","venue":"","topic":"Advanced Database Systems and Queries","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Search engine indexing; Computer science; Big data; Information retrieval; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003455431,0.001277165,0.001538429,0.005424272,0.002072803,0.004674453,0.00225977,0.001435695,0.01124484],"category_scores_gemma":[0.008283274,0.0008842663,0.001577397,0.01031113,0.001304821,0.006602536,0.001968263,0.002324573,0.007521554],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001396793,"about_ca_system_score_gemma":0.00147629,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002220118,"about_ca_topic_score_gemma":0.002214684,"domain_scores_codex":[0.9955992,0.0005936243,0.0004539903,0.0004821108,0.002648665,0.0002224856],"domain_scores_gemma":[0.9950288,0.001671215,0.0003426675,0.001746067,0.001103978,0.0001072589],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001804092,0.00013157,0.0009368666,0.001109058,0.0001216011,0.000246608,0.0004686523,0.005486789,0.01327544,0.09883483,0.06592749,0.8132808],"study_design_scores_gemma":[0.0001144846,0.0002232763,0.001818673,0.0005103393,0.0001532621,0.001689494,0.0004724574,0.09317286,0.03119848,0.2456055,0.6248914,0.0001497512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002723854,0.01270852,0.9567968,0.002317909,0.0008137139,0.000501067,0.001093879,0.006939369,0.01610487],"genre_scores_gemma":[0.03772315,0.01584364,0.9230123,0.001176239,0.001474912,0.0005817793,0.002547523,0.0009894362,0.016651],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01124484,"threshold_uncertainty_score":0.03761774,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1531366665921439,"score_gpt":0.3376372985522201,"score_spread":0.1845006319600762,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}