{"id":"W4385570873","doi":"10.18653/v1/2023.cawl-1.11","title":"Learning the Character Inventories of Undeciphered Scripts Using Unsupervised Deep Clustering","year":2023,"lang":"en","type":"article","venue":"","topic":"Handwritten Text Recognition Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Scripting language; Computer science; Cluster analysis; Character (mathematics); Artificial intelligence; Exploit; Task (project management); Natural language processing; Set (abstract data type); Process (computing); Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004881834,0.0001073612,0.0001450497,0.0001406867,0.0001769135,0.0001486775,0.0006097225,0.00005196818,0.00007567128],"category_scores_gemma":[0.00008195469,0.00007819576,0.00006768367,0.000814727,0.00006048529,0.0004715643,0.0004481676,0.0001424811,0.00005382967],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002972045,"about_ca_system_score_gemma":0.00003393821,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007583745,"about_ca_topic_score_gemma":0.00002992322,"domain_scores_codex":[0.9989062,0.0001213191,0.0002566277,0.0002178755,0.00025738,0.0002405868],"domain_scores_gemma":[0.9993007,0.0001092846,0.00009134813,0.0003344897,0.0001199115,0.0000442258],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003769142,0.0001469036,0.01480857,0.0003134571,0.0002171405,0.00006730767,0.02648681,0.002053055,0.2508752,0.03037744,0.001628267,0.6729882],"study_design_scores_gemma":[0.0002992024,0.00009193498,0.004140423,0.0001066796,0.00001012855,0.00001840229,0.0009815263,0.9230104,0.0631175,0.005723071,0.002236048,0.0002646495],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2099356,0.00002518104,0.7856258,0.0005678195,0.0002952705,0.0002000831,6.3841e-7,0.001016479,0.002333184],"genre_scores_gemma":[0.9670555,0.00002685591,0.03179959,0.0002735715,0.00006078202,0.00001684005,0.000003518652,0.00001582001,0.0007475133],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9209574,"threshold_uncertainty_score":0.318873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05054510033694829,"score_gpt":0.2796570827119955,"score_spread":0.2291119823750472,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}