{"id":"W4391464262","doi":"10.1162/qss_a_00285","title":"Large-scale text analysis using generative language models: A case study in discovering public value expressions in AI patents","year":2024,"lang":"en","type":"article","venue":"Quantitative Science Studies","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Social Sciences and Humanities Research Council of Canada; Biotechnology and Biological Sciences Research Council; Directorate for Biological Sciences; Snap","keywords":"Generative grammar; Natural language processing; Scale (ratio); Value (mathematics); Computer science; Generative model; Artificial intelligence; Linguistics; Machine learning; Geography; Cartography; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.009732377,0.0004807042,0.0003909678,0.002477261,0.001126441,0.001623267,0.001211728,0.001221504,0.0009783691],"category_scores_gemma":[0.03844003,0.0002860266,0.000910308,0.002155135,0.002070491,0.002982442,0.0012794,0.00193265,0.0003083381],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001554639,"about_ca_system_score_gemma":0.001106229,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005144007,"about_ca_topic_score_gemma":0.007717496,"domain_scores_codex":[0.9923357,0.005856684,0.0002286763,0.0005680568,0.0008840446,0.0001268321],"domain_scores_gemma":[0.8808862,0.110239,0.002641988,0.003659954,0.002168568,0.000404285],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006728482,0.001293335,0.0704818,0.0008389166,0.0002853562,0.005061483,0.01671035,0.2753342,0.03062206,0.1198804,0.01620342,0.4626159],"study_design_scores_gemma":[0.00005465228,0.0001044254,0.005929768,0.0000473082,0.00003622869,0.0003605183,0.001314008,0.9205689,0.01283388,0.05301091,0.005677835,0.00006151064],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4625821,0.0004235735,0.5261752,0.003677132,0.00004844311,0.0003389884,0.0008757057,0.001982134,0.003896722],"genre_scores_gemma":[0.77308,0.0001086626,0.2247087,0.0002624562,0.00004158982,0.0001624041,0.0007742187,0.0001439468,0.000718209],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9975227,"threshold_uncertainty_score":0.05147034,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2553169436416557,"score_gpt":0.5238114656364624,"score_spread":0.2684945219948067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}