{"id":"W3003312070","doi":"10.22148/001c.11830","title":"Is there a text in my data? (Part 1): on counting words","year":2020,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Digital Humanities and Scholarship","field":"Arts and Humanities","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Word (group theory); Context (archaeology); Simple (philosophy); Normalization (sociology); Natural language processing; Linguistics; Artificial intelligence; Epistemology; Philosophy; History; Sociology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010927,0.0008859095,0.0009406736,0.004421032,0.003118183,0.009427847,0.001344073,0.001987066,0.005289522],"category_scores_gemma":[0.0827159,0.0005810066,0.0008505905,0.007251428,0.01525944,0.02580109,0.004953692,0.00444024,0.002030061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003323568,"about_ca_system_score_gemma":0.002321137,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003513196,"about_ca_topic_score_gemma":0.003369981,"domain_scores_codex":[0.9877512,0.007611741,0.0007156303,0.00177107,0.001925415,0.0002247883],"domain_scores_gemma":[0.9485883,0.04190818,0.002998591,0.002897517,0.003159814,0.0004476695],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00007956885,0.00004606479,0.006719489,0.0005687373,0.00004483612,0.0001351786,0.0151212,0.001007598,0.0006265532,0.773562,0.04968791,0.1524008],"study_design_scores_gemma":[0.00001291892,0.00004737208,0.004419804,0.001298776,0.0000290135,0.0004180164,0.01004736,0.006481671,0.001115687,0.7657818,0.2102691,0.00007843394],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05606665,0.02819243,0.6805737,0.151506,0.006912997,0.0005159006,0.002550229,0.0006706977,0.07301131],"genre_scores_gemma":[0.5224733,0.02382638,0.3723469,0.03624285,0.01616304,0.001875952,0.003028542,0.001365267,0.02267762],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.010927,"threshold_uncertainty_score":0.05778825,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1950699218308501,"score_gpt":0.292525711556765,"score_spread":0.09745578972591484,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}