{"id":"W4287374337","doi":"10.5281/zenodo.4065271","title":"Combining Visual and Textual Features for Semantic Segmentation of Historical Newspapers","year":2021,"lang":"en","type":"article","venue":"Infoscience (Ecole Polytechnique Fédérale de Lausanne)","topic":"Digital Humanities and Scholarship","field":"Arts and Humanities","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Fredericton","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Newspaper; Segmentation; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Sociology; Media studies","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005058212,0.0009569023,0.0004651053,0.002554488,0.000297384,0.001105908,0.0004997086,0.0007216646,0.002398236],"category_scores_gemma":[0.001835327,0.000263161,0.0007222253,0.001309576,0.0003310535,0.001647835,0.0006603732,0.0007996297,0.001961733],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000508911,"about_ca_system_score_gemma":0.000304733,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005515202,"about_ca_topic_score_gemma":0.008410201,"domain_scores_codex":[0.9997239,0.00004971384,0.0000131125,0.0001234117,0.00004162543,0.00004825946],"domain_scores_gemma":[0.999314,0.00028773,0.00008059532,0.000120586,0.0001367601,0.00006033692],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001431687,0.0005852328,0.01513828,0.000462055,0.0003180037,0.0003740565,0.0004504348,0.06964686,0.06058485,0.001059102,0.01045391,0.8394955],"study_design_scores_gemma":[0.0000324487,0.0002692491,0.02706572,0.00006808501,0.0001790684,0.0002520591,0.0002801751,0.9363433,0.02768145,0.003108366,0.00466824,0.00005184547],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7835514,0.00254095,0.1841744,0.000657697,0.0002970363,0.0002329119,0.004757666,0.008671443,0.01511646],"genre_scores_gemma":[0.9370672,0.0004840929,0.05261977,0.00009913535,0.0001558949,0.00005389454,0.005130901,0.0002440794,0.004144873],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005515202,"threshold_uncertainty_score":0.01096618,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02725921564840544,"score_gpt":0.2642684954682165,"score_spread":0.2370092798198111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}