{"id":"W2786093037","doi":"10.1109/hpcc-smartcity-dss.2017.35","title":"An Efficient Type-Agnostic Approach for Finding Sub-sequences in Data","year":2017,"lang":"en","type":"article","venue":"","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Search engine indexing; Data mining; Data deduplication; Inverted index; Normalization (sociology); Semantics (computer science); Index (typography); Raw data; Data type; Data structure; Theoretical computer science; Information retrieval; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.007665099,0.00008677381,0.0001739018,0.0001458984,0.0003645035,0.001125248,0.00465385,0.00003763859,0.00008920904],"category_scores_gemma":[0.007428887,0.00005980523,0.00002235842,0.0001571463,0.0001201407,0.0008837686,0.000917403,0.00005486749,0.0001148097],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001600505,"about_ca_system_score_gemma":0.00004139783,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002638101,"about_ca_topic_score_gemma":0.0006231196,"domain_scores_codex":[0.998004,0.0000963063,0.0003651528,0.0006849489,0.0006108652,0.0002387462],"domain_scores_gemma":[0.9958396,0.0006719064,0.0001755591,0.003171635,0.00006964165,0.00007163994],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002902443,0.002151286,0.02898579,0.0001330628,0.00006460185,0.0000425795,0.002263976,0.03617534,0.001080064,0.3664667,0.2697701,0.2925763],"study_design_scores_gemma":[0.0008950809,0.0001575201,0.02966514,0.00002657684,0.00002326238,0.00000216291,0.004296276,0.882825,0.0005229975,0.01878163,0.06237472,0.000429588],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3018147,0.0000835773,0.6463284,0.002644656,0.00174057,0.001510717,0.0004706086,0.0000828486,0.04532397],"genre_scores_gemma":[0.9824299,0.000007665324,0.01596121,0.000157899,0.00005223883,0.00001385124,0.000165866,0.000004597688,0.001206725],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8466497,"threshold_uncertainty_score":0.9999117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5830582344166529,"score_gpt":0.5204564048528669,"score_spread":0.06260182956378602,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}