{"id":"W2774419307","doi":"10.12685/027.7-5-1-169","title":"Transfer Learning for OCRopus Model Training on Early Printed Books","year":2017,"lang":"en","type":"preprint","venue":"027 7 Zeitschrift für Bibliothekskultur","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Character (mathematics); Computer science; Scratch; Ground truth; Alphabet; Code (set theory); Training set; Test set; Set (abstract data type); Natural language processing; Artificial intelligence; Training (meteorology); Character encoding; Test (biology); Test data; Speech recognition; Linguistics; Mathematics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001294838,0.00191898,0.0009798761,0.001042728,0.0005410301,0.001157343,0.002082212,0.001350064,0.005948951],"category_scores_gemma":[0.004721934,0.0006748838,0.0008032175,0.0009663524,0.0005946057,0.001815328,0.002022537,0.00261208,0.00450159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009768505,"about_ca_system_score_gemma":0.0008456418,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006155325,"about_ca_topic_score_gemma":0.006068369,"domain_scores_codex":[0.9991902,0.0001910135,0.00005458676,0.0002881106,0.0001811687,0.00009504361],"domain_scores_gemma":[0.9979914,0.0009640037,0.0001060886,0.0004849969,0.0003727096,0.00008073167],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003875374,0.0002386688,0.001541067,0.0002315167,0.000151615,0.0003215652,0.0002201912,0.1480245,0.03036417,0.001189337,0.007463806,0.8098661],"study_design_scores_gemma":[0.00002236015,0.0001453115,0.0009454142,0.00002311325,0.00003381273,0.0001230716,0.0000848509,0.9617483,0.0318819,0.001788983,0.003174508,0.00002830699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2031226,0.001425138,0.7528527,0.000435244,0.0003281999,0.0003416956,0.001002118,0.03225701,0.008235369],"genre_scores_gemma":[0.7240758,0.0003757181,0.2526029,0.0002988882,0.0001205142,0.00053845,0.003914135,0.001373317,0.01670027],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006155325,"threshold_uncertainty_score":0.01990128,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05356677114805272,"score_gpt":0.3376364533032796,"score_spread":0.2840696821552269,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}