{"id":"W4387796700","doi":"10.48550/arxiv.2310.10808","title":"If the Sources Could Talk: Evaluating Large Language Models for Research Assistance in History","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Metadata; Computer science; World Wide Web; Style (visual arts); Information retrieval; Data science; Natural language processing; History","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004155137,0.0002128604,0.0002758529,0.0003803998,0.000256974,0.0001180531,0.002917183,0.0002455412,0.0000107517],"category_scores_gemma":[0.000240702,0.0002108569,0.0001554713,0.0005237115,0.0001087394,0.0003060736,0.002657421,0.001102257,0.00002113405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001078128,"about_ca_system_score_gemma":0.0005143674,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009235037,"about_ca_topic_score_gemma":0.0008785538,"domain_scores_codex":[0.9971264,0.0004551524,0.0002472895,0.001220395,0.0003034339,0.0006473006],"domain_scores_gemma":[0.9971175,0.0007590157,0.0001777794,0.001608599,0.0002535429,0.00008363149],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001934541,0.00004083443,0.0004089423,0.000108029,0.00002584448,0.000120986,0.004941935,0.7854614,0.00003575873,0.2075471,0.0008209699,0.0004688473],"study_design_scores_gemma":[0.000342205,0.00002003874,0.0001281337,0.0001235628,0.0000110958,5.252871e-7,0.0009948363,0.9253708,0.00001084496,0.0719571,0.0008339151,0.0002069911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2608292,0.0004702926,0.7360086,0.0002908111,0.0005404315,0.0006022523,0.00001532319,0.0001882378,0.001054892],"genre_scores_gemma":[0.987209,0.00005153758,0.003504513,0.00008097692,0.0001252325,0.00001505032,0.000008007635,0.00002738611,0.008978331],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7325041,"threshold_uncertainty_score":0.8598495,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3848062753314309,"score_gpt":0.3132841151825062,"score_spread":0.07152216014892471,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}