{"id":"W4413360655","doi":"10.1109/icde65448.2025.00101","title":"A Length Enhanced B<sup>+</sup>-Tree Based Index for Efficient Set Similarity Query","year":2025,"lang":"en","type":"article","venue":"","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of New Brunswick","funders":"National Natural Science Foundation of China","keywords":"Computer science; Set (abstract data type); Index (typography); Similarity (geometry); Tree (set theory); Algorithm; Mathematics; Combinatorics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004954556,0.0001861387,0.0001980632,0.0002329801,0.0001710722,0.0003137605,0.001265426,0.00006376225,0.00003631918],"category_scores_gemma":[0.00008188175,0.0001638369,0.000112158,0.0006273542,0.00003852395,0.0003404332,0.0005484232,0.0001085114,0.00003137161],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005115542,"about_ca_system_score_gemma":0.0001120355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000401176,"about_ca_topic_score_gemma":0.00003050723,"domain_scores_codex":[0.9984354,0.00004375866,0.0002512237,0.0006069861,0.0002639636,0.0003986729],"domain_scores_gemma":[0.9986879,0.0002159783,0.00005147847,0.0008859559,0.00008958598,0.00006914631],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001091025,0.001032232,0.0009536871,0.0004249201,0.0002079406,0.00001943159,0.0005233105,0.05535498,0.0002768922,0.3113103,0.1547777,0.4750095],"study_design_scores_gemma":[0.0009382557,0.00004805247,0.0005091154,0.00002655934,0.00001189528,8.499235e-8,0.00005131785,0.9676665,0.003062085,0.001446343,0.02604054,0.0001992223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00311812,0.00001696128,0.9780753,0.002618756,0.0002843793,0.0005449821,0.00002877135,0.0003280231,0.01498475],"genre_scores_gemma":[0.8655779,0.00000389858,0.1211765,0.005159855,0.00007709046,0.0001537825,0.00008003292,0.00001244123,0.007758429],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9123116,"threshold_uncertainty_score":0.6681076,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01608985870247986,"score_gpt":0.2646229309902215,"score_spread":0.2485330722877417,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}