{"id":"W4386074468","doi":"10.11159/cist23.152","title":"Historical-Domain Pre-trained Language Model for Historical Extractive Text Summarization","year":2023,"lang":"en","type":"article","venue":"Proceedings of the World Congress on Electrical Engineering and Computer Systems and Science","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Natural language processing; Artificial intelligence; Domain (mathematical analysis); Language model; Information retrieval; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005942758,0.0001458575,0.000241283,0.0003475522,0.0002154105,0.0001944884,0.0006616023,0.000041788,3.888112e-8],"category_scores_gemma":[0.0001037378,0.000109107,0.00004617133,0.001274649,0.00004006132,0.0003065761,0.0002085657,0.0001472864,1.476425e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003094234,"about_ca_system_score_gemma":0.00004394414,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002321739,"about_ca_topic_score_gemma":6.778841e-7,"domain_scores_codex":[0.9985384,0.000005204775,0.0002679905,0.0004745946,0.0003822905,0.0003315264],"domain_scores_gemma":[0.9993038,0.0001503786,0.0001253384,0.000151473,0.0001535061,0.0001155419],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004863615,0.0001151654,0.0009158932,0.0004954352,0.00003790999,0.000002839077,0.003182705,0.1311184,0.03471699,0.7811679,0.004708924,0.04348922],"study_design_scores_gemma":[0.0001844103,0.00006755719,0.0004345197,0.00008273465,0.000004161807,0.000006984729,0.000005215246,0.9975893,0.0003669276,0.0003166571,0.0008066674,0.0001348273],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.137994,0.0004302414,0.8577057,0.001079249,0.001832797,0.0005471909,0.00000131634,0.000297428,0.0001121347],"genre_scores_gemma":[0.9857451,0.000009769078,0.01158114,0.00003107509,0.0001436583,0.0000522015,1.216222e-7,0.00001044281,0.00242647],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8664709,"threshold_uncertainty_score":0.4449256,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01208999847627023,"score_gpt":0.2206911119105644,"score_spread":0.2086011134342942,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}