{"id":"W6891719042","doi":"10.48448/tcn5-3z78","title":"ANALOGICAL - A Novel Benchmark for Long Text Analogy Evaluation in Large Language Models","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Analogy; Benchmark (surveying); Word (group theory); Taxonomy (biology); Language model; Quality (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.007310934,0.0004655738,0.0006109669,0.002991266,0.0003066242,0.0001267076,0.001435777,0.0002972788,0.01651076],"category_scores_gemma":[0.0008443138,0.0004425374,0.0001308353,0.003513171,0.0006304874,0.0003738158,0.0005348534,0.0005551439,0.0002204846],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001482308,"about_ca_system_score_gemma":0.001537776,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000722579,"about_ca_topic_score_gemma":0.01924608,"domain_scores_codex":[0.9942594,0.0002035283,0.0005796341,0.001517916,0.002252101,0.001187402],"domain_scores_gemma":[0.9980315,0.0001833568,0.0004705103,0.000880983,0.0002276787,0.0002059706],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005649655,0.01076735,0.009483277,0.0006511837,0.0004680679,0.0004412626,0.008768235,0.1538842,0.01645292,0.2327443,0.5001956,0.06557865],"study_design_scores_gemma":[0.002796209,0.0002075867,0.001372352,0.00007465375,0.0001032184,0.00002955381,0.0009540411,0.9683724,0.00001229841,0.002945677,0.02238311,0.0007489056],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.01010725,0.003565769,0.1148363,0.0004376545,0.001457655,0.009382512,0.007468909,0.0009192872,0.8518246],"genre_scores_gemma":[0.880343,0.00002940293,0.03008702,0.001129824,0.0009162636,0.001598455,0.008438786,0.001535229,0.07592204],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.8702357,"threshold_uncertainty_score":0.9998026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05689247138265904,"score_gpt":0.3601712973901034,"score_spread":0.3032788260074444,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}