{"id":"W848342684","doi":"","title":"PaddyWaC: A Minimally-Supervised Web-Corpus of Hiberno-English","year":2011,"lang":"en","type":"article","venue":"Research Portal (Queen's University Belfast)","topic":"Digital Communication and Language","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Irish; Computer science; Variety (cybernetics); Bootstrapping (finance); Artificial intelligence; Domain (mathematical analysis); Varieties of English; Natural language processing; World Wide Web; Linguistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001798065,0.0009662244,0.0006656722,0.005562134,0.001734315,0.001787485,0.001352009,0.001052715,0.01469825],"category_scores_gemma":[0.006295746,0.0005373158,0.0006284566,0.003537607,0.001323777,0.001766526,0.00274131,0.001311973,0.0115244],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009570207,"about_ca_system_score_gemma":0.001948468,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009835176,"about_ca_topic_score_gemma":0.02349446,"domain_scores_codex":[0.9985115,0.0003601862,0.0002150251,0.0005608408,0.0002390908,0.0001133213],"domain_scores_gemma":[0.9937469,0.00314246,0.0004182387,0.001054333,0.00129115,0.000346896],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008859916,0.0008268775,0.02563275,0.01037801,0.0002893211,0.005089037,0.01385635,0.003243454,0.08096399,0.009963185,0.4628477,0.3860234],"study_design_scores_gemma":[0.00025453,0.0001606771,0.08950913,0.0007643148,0.0001243569,0.002814404,0.005779373,0.01181194,0.02324117,0.002344345,0.86303,0.0001657777],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.3181217,0.003856857,0.09554008,0.001423553,0.001128574,0.004050503,0.462406,0.02396745,0.0895054],"genre_scores_gemma":[0.1646444,0.0006557279,0.1096012,0.0003321281,0.0001688785,0.003728221,0.6950248,0.003757582,0.02208713],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01469825,"threshold_uncertainty_score":0.04917061,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03669335032176523,"score_gpt":0.2615932642133681,"score_spread":0.2248999138916029,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}