{"id":"W1792017449","doi":"10.22230/src.2017v8n2a280","title":"Guessing at the Content of a Million Books","year":2017,"lang":"en","type":"article","venue":"Scholarly and Research Communication","topic":"Digital Humanities and Scholarship","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Graffiti; Scholarship; Computer science; Reading (process); Context (archaeology); Set (abstract data type); Hypertext; Conjecture; Artificial intelligence; World Wide Web; Natural language processing; Linguistics; History; Mathematics; Programming language; Law; Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.002801036,0.00006415232,0.00009568575,0.00006632926,0.006475682,0.005090134,0.0008882631,0.00003490088,0.0001921717],"category_scores_gemma":[0.0005672316,0.00004065613,0.00004233399,0.00001126688,0.00214582,0.002481955,0.0008798901,0.0005321552,0.00002313344],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003422065,"about_ca_system_score_gemma":0.00002543684,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001044535,"about_ca_topic_score_gemma":0.003341568,"domain_scores_codex":[0.9988267,0.0003057962,0.0001750282,0.0001086133,0.0004065605,0.0001772769],"domain_scores_gemma":[0.9977013,0.0002512195,0.000124452,0.001150463,0.0007256618,0.00004691686],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008910833,0.00006948691,0.001875247,0.00004533961,0.00003576841,0.000001175309,0.02590455,1.10084e-7,0.001387701,0.9324019,0.002783521,0.03540612],"study_design_scores_gemma":[0.0007654341,0.0002604462,0.05351456,0.0004931026,0.00001925868,0.00000557121,0.02813402,0.00003746655,0.001849234,0.06201039,0.8526931,0.0002174629],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7017381,0.00409593,7.656595e-7,0.003496652,0.00006123428,0.0001701235,0.00001312162,0.000009992553,0.290414],"genre_scores_gemma":[0.9565369,0.0006295524,0.0000137938,0.00006328651,0.00006569666,0.00001783285,0.00001113344,0.000007731595,0.04265406],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8703915,"threshold_uncertainty_score":0.9959427,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4324220114255007,"score_gpt":0.3838589719409515,"score_spread":0.04856303948454915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}