{"id":"W2063050588","doi":"10.3758/bf03195586","title":"Case-sensitive letter and bigram frequency counts from large-scale English corpora","year":2004,"lang":"en","type":"article","venue":"Behavior Research Methods, Instruments, & Computers","topic":"Reading and Literacy Development","field":"Psychology","cited_by":118,"is_retracted":false,"has_abstract":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bigram; Computer science; Speech recognition; Frequency; Natural language processing; Scale (ratio); Artificial intelligence; Statistics; Mathematics; Trigram","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002681079,0.0006725977,0.0006520809,0.006354725,0.001323539,0.002079188,0.001142087,0.001094776,0.01120458],"category_scores_gemma":[0.02376632,0.0005764829,0.000476886,0.004523447,0.0008310684,0.002785652,0.001660419,0.001336143,0.007909944],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005566218,"about_ca_system_score_gemma":0.0008918326,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003626586,"about_ca_topic_score_gemma":0.008514736,"domain_scores_codex":[0.9958643,0.001393615,0.0006006929,0.000981253,0.0009128276,0.0002473226],"domain_scores_gemma":[0.9691604,0.01853183,0.001373491,0.005511008,0.004572216,0.0008510317],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001853901,0.001190812,0.1026207,0.002991216,0.0004010539,0.00443129,0.01095984,0.004493594,0.1614413,0.01254576,0.09407514,0.6029955],"study_design_scores_gemma":[0.00038815,0.0006268222,0.5626566,0.0005311989,0.0006197281,0.0144883,0.009219024,0.07776669,0.1023045,0.02543674,0.2054128,0.0005495076],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7809904,0.001722023,0.09046816,0.0007757566,0.0004175986,0.0004455509,0.08680969,0.006364665,0.03200619],"genre_scores_gemma":[0.7727089,0.0004967924,0.07516953,0.0001730988,0.000238237,0.00042583,0.1436635,0.001314687,0.00580944],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01120458,"threshold_uncertainty_score":0.03748304,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1203234814297276,"score_gpt":0.4528343708533598,"score_spread":0.3325108894236322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}