{"id":"W6947658562","doi":"10.48448/2b2k-qk12","title":"Learning to Extract Structured Entities Using Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; McGill University","funders":"","keywords":"Task (project management); Information extraction; Structured prediction; Field (mathematics); Language model; Set (abstract data type); Named-entity recognition; Code (set theory); Precision and recall; Source code","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001785417,0.0002248023,0.0001696243,0.0002704279,0.0000886228,0.0001503299,0.0003757722,0.000203592,0.000127733],"category_scores_gemma":[0.00003562844,0.0002062164,0.0000651819,0.0002301396,0.0002249598,0.000006058602,0.000140317,0.0002055264,0.0000399107],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003526702,"about_ca_system_score_gemma":0.000223981,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003602593,"about_ca_topic_score_gemma":0.000439341,"domain_scores_codex":[0.9986591,0.00002044744,0.0001527253,0.000571333,0.0002713957,0.0003249818],"domain_scores_gemma":[0.9994882,0.000003128934,0.00006021993,0.0002897097,0.0000434088,0.0001153811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001083892,0.00002160629,0.00002252992,0.00007741363,0.00003886317,0.00001342633,0.0004792112,0.005441171,0.9802654,0.0001715957,0.006776154,0.006681782],"study_design_scores_gemma":[0.0008393515,0.0008280282,0.00001153664,0.0008928285,0.0002449932,0.0001045656,0.002887183,0.1044506,0.2545313,0.001807193,0.6307688,0.002633651],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2662345,0.02078716,0.1969186,0.0002523121,0.007917596,0.001522854,0.0004291069,0.0007307115,0.5052072],"genre_scores_gemma":[0.6531398,0.0001384767,0.02112638,0.0002933561,0.001301737,0.000006171105,0.0001374656,0.0004268523,0.3234298],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7257341,"threshold_uncertainty_score":0.8409262,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02215105077893219,"score_gpt":0.2890545481242257,"score_spread":0.2669034973452935,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}