{"id":"W7005561867","doi":"","title":"Retrieval-augmented text generation with domain-specific large language models fine-tuning","year":2024,"lang":"en","type":"dissertation","venue":"Repository of the University of Ljubljana (University of Ljubljana)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Context (archaeology); Generator (circuit theory); Question answering; Component (thermodynamics); Language model; Architecture; Context model; Labrador Retriever","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001972989,0.001322424,0.0007636914,0.0006803861,0.0002672127,0.001013652,0.001938433,0.00124969,0.00383494],"category_scores_gemma":[0.008020639,0.0006886353,0.001037232,0.0004366383,0.0005548222,0.002107049,0.001742943,0.001652526,0.003000334],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006824369,"about_ca_system_score_gemma":0.0009467992,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003426444,"about_ca_topic_score_gemma":0.004303997,"domain_scores_codex":[0.9986975,0.0005589197,0.0001111737,0.0003660319,0.0001802911,0.00008614597],"domain_scores_gemma":[0.9971327,0.001577614,0.0001124266,0.0005719247,0.0005364544,0.0000689142],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006766956,0.0006895124,0.003233069,0.0006886345,0.0001792086,0.0004674934,0.0008096627,0.4096688,0.1099898,0.004986202,0.01150402,0.4571069],"study_design_scores_gemma":[0.0001153616,0.0001405441,0.0003891553,0.00001115323,0.00004156064,0.00009304903,0.00005794607,0.9614512,0.03183721,0.002171969,0.003652413,0.00003844625],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1033702,0.0003579725,0.8295131,0.0003039903,0.0001450808,0.0005459029,0.000821282,0.06122522,0.00371727],"genre_scores_gemma":[0.4850241,0.0001338221,0.5039096,0.0003933172,0.00005764494,0.0007325497,0.00283853,0.002499157,0.004411268],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00383494,"threshold_uncertainty_score":0.01282912,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01334941107173958,"score_gpt":0.1882553743952043,"score_spread":0.1749059633234647,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}