{"id":"W3208600296","doi":"10.1109/dcc52660.2022.00015","title":"Computing Matching Statistics on Repetitive Texts","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Algorithms and Data Compression","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Matching (statistics); String (physics); String searching algorithm; Attractor; Measure (data warehouse); Pattern matching; Computer science; Sequence (biology); Space (punctuation); Theoretical computer science; Algorithm; Mathematics; Statistics; Data mining; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001153336,0.0005091102,0.001364696,0.003303411,0.0006555395,0.002208224,0.001354148,0.001143684,0.003175021],"category_scores_gemma":[0.01911716,0.0003644895,0.0006465991,0.005782075,0.001079222,0.004869767,0.002037739,0.0009269582,0.002030925],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009125777,"about_ca_system_score_gemma":0.001474749,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001166141,"about_ca_topic_score_gemma":0.001396111,"domain_scores_codex":[0.997898,0.0002884345,0.0002927477,0.0005275897,0.0008081804,0.0001850332],"domain_scores_gemma":[0.9907432,0.004591005,0.001248458,0.001977735,0.00108024,0.0003593802],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002464812,0.0003124932,0.02004012,0.0006980835,0.0001877411,0.0008633662,0.0007483924,0.1768158,0.07797208,0.08171128,0.01339837,0.6247874],"study_design_scores_gemma":[0.00009418867,0.0003300951,0.0053843,0.00004262943,0.00005474833,0.0004548978,0.0003516333,0.7886965,0.05627364,0.1424514,0.005808111,0.00005775963],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4802894,0.0007928125,0.5038074,0.0008201854,0.000106001,0.000117224,0.003886664,0.006550864,0.003629439],"genre_scores_gemma":[0.7353642,0.0004295766,0.2489726,0.0002105647,0.0002452499,0.000202718,0.01076487,0.0007142329,0.003096056],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003303411,"threshold_uncertainty_score":0.01062149,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02070156763076163,"score_gpt":0.2928627847955807,"score_spread":0.2721612171648191,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}