{"id":"W4287374337","doi":"10.5281/zenodo.4065271","title":"Combining Visual and Textual Features for Semantic Segmentation of Historical Newspapers","year":2021,"lang":"en","type":"article","venue":"Infoscience (Ecole Polytechnique Fédérale de Lausanne)","topic":"Digital Humanities and Scholarship","field":"Arts and Humanities","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Fredericton","funders":"Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Newspaper; Segmentation; Computer science; Natural language processing; Artificial intelligence; Information retrieval; Sociology; Media studies","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002972216,0.0001498853,0.0002306224,0.0001211747,0.0003572783,0.0003812922,0.0001804201,0.00007150383,0.0001408234],"category_scores_gemma":[0.0001446956,0.0001509657,0.00008861641,0.00008423097,0.0003217849,0.0006436541,0.00008675694,0.0001441624,0.000001830936],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001268321,"about_ca_system_score_gemma":0.0001263977,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003528146,"about_ca_topic_score_gemma":0.001403544,"domain_scores_codex":[0.9988453,0.00003545797,0.0003041096,0.0002605907,0.0002217245,0.0003328424],"domain_scores_gemma":[0.9992834,0.0001695628,0.0001365111,0.0001429181,0.0001670183,0.0001006448],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001235446,0.0004634437,0.004044323,0.0003804879,0.00004778645,0.00003689934,0.0286315,0.00002774008,0.03869428,0.8739408,0.0179096,0.03569964],"study_design_scores_gemma":[0.004303924,0.005466703,0.01144247,0.001353392,0.0002990163,0.0003016242,0.09158942,0.002753993,0.3034056,0.05091501,0.5249983,0.003170526],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9064137,0.0008584204,0.002748117,0.001090064,0.0007575258,0.0009145114,0.0001579378,0.0002715402,0.08678824],"genre_scores_gemma":[0.9815261,0.00003115397,0.001944154,0.0006958209,0.0001448539,0.0000798287,0.00002498214,0.00001650246,0.01553658],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8230258,"threshold_uncertainty_score":0.6156203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02725921564840544,"score_gpt":0.2642684954682165,"score_spread":0.2370092798198111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}