{"id":"W4387796700","doi":"10.48550/arxiv.2310.10808","title":"If the Sources Could Talk: Evaluating Large Language Models for Research Assistance in History","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Compute Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Metadata; Computer science; World Wide Web; Style (visual arts); Information retrieval; Data science; Natural language processing; History","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01947897,0.001672936,0.0008387024,0.002419676,0.0008654275,0.002422476,0.001725353,0.002466717,0.002535324],"category_scores_gemma":[0.04907113,0.0006179202,0.001176008,0.001659531,0.0008076413,0.004058898,0.002610011,0.002073166,0.001109958],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001643444,"about_ca_system_score_gemma":0.001410426,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008008902,"about_ca_topic_score_gemma":0.00888123,"domain_scores_codex":[0.991574,0.00687952,0.0003543595,0.0007957537,0.0002783299,0.0001179953],"domain_scores_gemma":[0.9371032,0.05766261,0.0008714825,0.002604056,0.001064763,0.0006939078],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006495292,0.00189116,0.03484738,0.001469721,0.001616269,0.0003096477,0.004389243,0.4374151,0.004165549,0.007013991,0.01216208,0.4882246],"study_design_scores_gemma":[0.0001507069,0.0003402522,0.002224436,0.00006210018,0.0001565949,0.00004790519,0.0005784155,0.9849739,0.001985256,0.007841278,0.001601007,0.00003806346],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7962341,0.004058543,0.1795954,0.003260166,0.0002707656,0.0008090995,0.003493284,0.006023085,0.006255549],"genre_scores_gemma":[0.9134143,0.000399757,0.08013587,0.0002980626,0.0001233153,0.0005127101,0.003781006,0.0002189834,0.001115939],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.980521,"threshold_uncertainty_score":0.1030158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3848062753314309,"score_gpt":0.3132841151825062,"score_spread":0.07152216014892471,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}