{"id":"W4298110867","doi":"10.1111/1911-3846.12832","title":"<scp>FinBERT</scp>: A Large Language Model for Extracting Information from Financial Text*","year":2022,"lang":"en","type":"article","venue":"Contemporary Accounting Research","topic":"Stock Market Forecasting Methods","field":"Decision Sciences","cited_by":648,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Finance; Encoder; Natural language processing; Sample (material); Salient; Earnings; Random forest; Language model; Business","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007827042,0.0006709356,0.0003688508,0.001275739,0.0006497781,0.001165292,0.0009218138,0.0007381651,0.004128044],"category_scores_gemma":[0.00308511,0.0004073036,0.0006456517,0.0009111736,0.0004662795,0.001799386,0.0006784254,0.001289065,0.002457992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001220461,"about_ca_system_score_gemma":0.001759181,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04341323,"about_ca_topic_score_gemma":0.06058662,"domain_scores_codex":[0.999767,0.00007869487,0.00001195293,0.00006728447,0.00005201374,0.00002318051],"domain_scores_gemma":[0.9990507,0.0005780134,0.00006211249,0.00009836404,0.0001754144,0.00003532373],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003342742,0.0002441967,0.007256821,0.0001935125,0.0001760986,0.0004203524,0.0005759201,0.256715,0.009495155,0.03270499,0.07858565,0.6132981],"study_design_scores_gemma":[0.000006600247,0.00001654549,0.0006580299,0.00001544888,0.00001383476,0.00005045595,0.00002282539,0.9811226,0.00305845,0.00820932,0.00680633,0.00001950108],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05817522,0.0007862144,0.9139713,0.002147748,0.000228599,0.0001795356,0.005513263,0.01248175,0.006516287],"genre_scores_gemma":[0.5721561,0.0008718956,0.3911164,0.0009212447,0.000219274,0.0003748276,0.01060093,0.001484463,0.02225487],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04341323,"threshold_uncertainty_score":0.08632106,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2136271072296654,"score_gpt":0.458767453549553,"score_spread":0.2451403463198876,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}