{"id":"W3027407410","doi":"10.1108/rmj-09-2019-0055","title":"Natural language processing and machine learning as practical toolsets for archival processing","year":2020,"lang":"en","type":"article","venue":"Records Management Journal","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":39,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Workflow; Artificial intelligence; Computer science; Machine learning; Implementation; Interoperability; USable; Natural language processing; Originality; Software engineering; Database; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06671011,0.001261042,0.001283139,0.01187618,0.002052407,0.01826817,0.003463562,0.003325494,0.004894954],"category_scores_gemma":[0.06450071,0.000886089,0.00121979,0.009374913,0.0137779,0.02231056,0.006055892,0.005432698,0.002012805],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005318357,"about_ca_system_score_gemma":0.01026076,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002226217,"about_ca_topic_score_gemma":0.001962933,"domain_scores_codex":[0.9461613,0.03675428,0.004505678,0.002323301,0.009712034,0.0005435274],"domain_scores_gemma":[0.8557936,0.1179485,0.004456488,0.01189142,0.008965158,0.0009448522],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00005056095,0.00008872368,0.0009175233,0.007392351,0.00006319914,0.0002553493,0.004386155,0.001741849,0.001212693,0.6416238,0.009980859,0.332287],"study_design_scores_gemma":[0.00002646387,0.0001260071,0.001177515,0.01015419,0.00005838802,0.0005733538,0.005290993,0.004326588,0.002745335,0.5088413,0.4665642,0.0001157187],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005741314,0.1401806,0.7011073,0.06750117,0.002172689,0.00115908,0.0005661963,0.002184903,0.07938693],"genre_scores_gemma":[0.06262898,0.096039,0.8216569,0.00739701,0.002080732,0.001699612,0.0006157272,0.0005623536,0.007319671],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.06671011,"threshold_uncertainty_score":0.352801,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1476724865815801,"score_gpt":0.4358033820896567,"score_spread":0.2881308955080766,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}