{"id":"W4393888259","doi":"10.5281/zenodo.4781769","title":"CTAB: Corpus of Tunisian Arabizi","year":2021,"lang":"en","type":"dataset","venue":"Figshare","topic":"Language, Linguistics, Cultural Analysis","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Moncton","funders":"","keywords":"Natural language processing; Linguistics; Computer science; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007664038,0.001848627,0.0007425563,0.005132431,0.00173223,0.001517795,0.00176296,0.001850587,0.02672834],"category_scores_gemma":[0.003279434,0.0003779637,0.0006284827,0.005998001,0.0006890686,0.001058589,0.001724472,0.00130358,0.03242121],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001821558,"about_ca_system_score_gemma":0.001766849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0441959,"about_ca_topic_score_gemma":0.0708016,"domain_scores_codex":[0.9992823,0.0001789953,0.00008450473,0.0001931582,0.0001511889,0.0001098875],"domain_scores_gemma":[0.999007,0.000297282,0.0000888997,0.0002015389,0.0003114975,0.0000937666],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002148417,0.000135201,0.002227145,0.001330605,0.00004116055,0.0003915288,0.0004349791,0.0004843146,0.001640565,0.0008619321,0.9701959,0.02204175],"study_design_scores_gemma":[0.0001264384,0.00003490047,0.01607167,0.0004572512,0.00003926717,0.0003896537,0.0009054554,0.001160424,0.001532628,0.0006199729,0.9786131,0.00004920165],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.008176882,0.0009364872,0.0004263826,0.0004012698,0.0002338485,0.0001384958,0.9818727,0.001043071,0.006770816],"genre_scores_gemma":[0.004889884,0.000188142,0.00108234,0.00008281159,0.00003072221,0.0002771338,0.9907972,0.00009260105,0.002559256],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.0441959,"threshold_uncertainty_score":0.08941519,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05305606617256659,"score_gpt":0.2468621852004857,"score_spread":0.1938061190279191,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}