{"id":"W4386074468","doi":"10.11159/cist23.152","title":"Historical-Domain Pre-trained Language Model for Historical Extractive Text Summarization","year":2023,"lang":"en","type":"article","venue":"Proceedings of the World Congress on Electrical Engineering and Computer Systems and Science","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Language model; Information retrieval; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001179003,0.001804686,0.001032289,0.002082559,0.0004931859,0.000950462,0.001439759,0.0008629917,0.002934247],"category_scores_gemma":[0.003410175,0.0004560261,0.001219006,0.0016235,0.0003725135,0.002297319,0.0008595437,0.001875613,0.00462079],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006655183,"about_ca_system_score_gemma":0.001381529,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003958351,"about_ca_topic_score_gemma":0.007815027,"domain_scores_codex":[0.9992193,0.0002130113,0.00006794667,0.0002794109,0.0001467355,0.00007366522],"domain_scores_gemma":[0.9985294,0.0005329427,0.0001634054,0.000180793,0.0005398981,0.00005338904],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004081353,0.000262683,0.001510325,0.0006498775,0.000213038,0.0003437967,0.0004005765,0.1044673,0.04958951,0.004435744,0.02986633,0.8078527],"study_design_scores_gemma":[0.00006419387,0.0002949057,0.001201859,0.00006405669,0.0001595035,0.0002150296,0.0002483151,0.9367129,0.03031939,0.006247378,0.02440741,0.00006499319],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01941554,0.00242779,0.9615477,0.0004303958,0.00028914,0.0001698616,0.002483018,0.01120261,0.002034074],"genre_scores_gemma":[0.2958249,0.00244408,0.6535303,0.0006122202,0.0007726137,0.0008951583,0.02871227,0.00138514,0.01582319],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003958351,"threshold_uncertainty_score":0.009816051,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01208999847627023,"score_gpt":0.2206911119105644,"score_spread":0.2086011134342942,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}