{"id":"W3029889931","doi":"","title":"The Johns Hopkins University Bible Corpus: 1600+ Tongues for Typological Exploration","year":2020,"lang":"en","type":"article","venue":"Language Resources and Evaluation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":48,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Natural language processing; Variety (cybernetics); Representation (politics); Linguistics; Corpus linguistics; Parallel corpora; Pronoun; Artificial intelligence; Information retrieval; History; Machine translation; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001044035,0.0005028843,0.0004775097,0.009435523,0.002489982,0.001889485,0.0007299999,0.0006235684,0.05307864],"category_scores_gemma":[0.006478014,0.000280877,0.0001629539,0.01003085,0.0011042,0.00118633,0.002365574,0.0008277936,0.02594782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009811205,"about_ca_system_score_gemma":0.002650174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01462967,"about_ca_topic_score_gemma":0.03021956,"domain_scores_codex":[0.998987,0.000280591,0.0001522211,0.000135184,0.0003599264,0.00008511085],"domain_scores_gemma":[0.9962239,0.001292575,0.0002337776,0.0006562498,0.0012055,0.0003878548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0006354945,0.0002004904,0.0079096,0.002174381,0.00003442351,0.001020322,0.01150041,0.0004840266,0.01314717,0.01575643,0.7551908,0.1919464],"study_design_scores_gemma":[0.0001077273,0.00005769155,0.05234583,0.0005151813,0.00003638779,0.0006620885,0.005707761,0.0007425279,0.005066743,0.001955146,0.9327561,0.00004677055],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.2018405,0.00351904,0.007090948,0.001658021,0.001288684,0.00056106,0.6098959,0.001563343,0.1725826],"genre_scores_gemma":[0.2664446,0.00203197,0.01897835,0.0005498966,0.0005688617,0.001442448,0.6440079,0.00175504,0.06422094],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.05307864,"threshold_uncertainty_score":0.1775658,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04572153790965464,"score_gpt":0.2934511416866595,"score_spread":0.2477296037770049,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}