{"id":"W4416057143","doi":"10.48550/arxiv.2510.18803","title":"Decoding Funded Research: Comparative Analysis of Topic Models and Uncovering the Effect of Gender and Geographic Location","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Topic model; Latent Dirichlet allocation; Covariate; Probabilistic logic; Thematic map; Thematic structure; Function (biology); Statistical model","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.0313026,0.0008207517,0.001121182,0.003966067,0.0008755674,0.004298267,0.001136614,0.001543084,0.002391775],"category_scores_gemma":[0.1438421,0.0003763886,0.001892191,0.006441303,0.001422513,0.005225269,0.002692042,0.001700709,0.0007192814],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001943069,"about_ca_system_score_gemma":0.002059133,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008241712,"about_ca_topic_score_gemma":0.009020125,"domain_scores_codex":[0.987324,0.009441423,0.0005687795,0.001540808,0.000725176,0.0003998215],"domain_scores_gemma":[0.8226454,0.162054,0.006037463,0.005397473,0.002809505,0.001056107],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001534591,0.0002616118,0.4960069,0.0009234405,0.001385859,0.0004440138,0.01178639,0.1011932,0.001812425,0.06317347,0.009712859,0.3117651],"study_design_scores_gemma":[0.0001993114,0.000342591,0.1273907,0.0003481214,0.0007977178,0.0003995489,0.005354702,0.7479897,0.002121489,0.09815226,0.0167438,0.0001601608],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8409754,0.003195142,0.1440083,0.003225068,0.0001546557,0.0001912108,0.002416437,0.0004290137,0.005404655],"genre_scores_gemma":[0.9782591,0.0007204405,0.01751926,0.00009868167,0.0001226441,0.0001346012,0.002135218,0.0001083985,0.0009015977],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9960339,"threshold_uncertainty_score":0.1655459,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3391748008106125,"score_gpt":0.479066704623995,"score_spread":0.1398919038133825,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}