{"id":"W4318305993","doi":"10.1101/2023.01.26.525723","title":"Building a Pangenome Alignment Index via Recursive Prefix-Free Parsing","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Parsing; Computer science; Suffix; Suffix array; Preprocessor; Prefix; Compressed suffix array; Index (typography); Graph; Genome; Algorithm; Artificial intelligence; Theoretical computer science; Data structure; String searching algorithm; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001168585,0.0008469889,0.001404963,0.002828573,0.001084851,0.002480971,0.00229028,0.0009911163,0.00497162],"category_scores_gemma":[0.00533324,0.0008343501,0.001265119,0.004428265,0.0009647626,0.004804553,0.003404429,0.002003554,0.003932739],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001177449,"about_ca_system_score_gemma":0.002293424,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002259041,"about_ca_topic_score_gemma":0.003656584,"domain_scores_codex":[0.9987673,0.0001798216,0.0001444975,0.0004374523,0.0003431542,0.0001278591],"domain_scores_gemma":[0.9973485,0.0007147389,0.0001855881,0.001046258,0.0005711796,0.0001336546],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004927536,0.0002849083,0.005413218,0.0007460419,0.0001369425,0.0004656097,0.0006795491,0.06043871,0.0510753,0.1084942,0.04809351,0.7236793],"study_design_scores_gemma":[0.00007780782,0.0002246976,0.001965922,0.0001210153,0.00009656302,0.0005464467,0.0002206622,0.6638991,0.07530835,0.1626513,0.09474058,0.0001475324],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01889343,0.0004593222,0.9436899,0.0003088618,0.0001202236,0.0001590891,0.00285661,0.02903197,0.004480558],"genre_scores_gemma":[0.0620826,0.0002828257,0.9214737,0.0002371077,0.00007841452,0.0002189325,0.01037115,0.002741658,0.002513694],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00497162,"threshold_uncertainty_score":0.01663172,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01710240932444647,"score_gpt":0.2282348393094716,"score_spread":0.2111324299850251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}