{"id":"W163905041","doi":"","title":"Validation: A Critical First Step in the Evaluation of Systems for Legal Corpus Determination","year":2003,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Selection (genetic algorithm); Task (project management); Context (archaeology); Process (computing); Information retrieval; Precision and recall; Recall; Term (time); Artificial intelligence; Data mining; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006075384,0.00004188336,0.00007194501,0.00003915268,0.0002339981,0.0000829962,0.0001324791,0.00005906834,0.0001376895],"category_scores_gemma":[0.005269195,0.00003267904,0.00003069212,0.0002095817,0.0001854509,0.0002528721,0.000003469841,0.00003545642,0.000008531249],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001028668,"about_ca_system_score_gemma":0.0001642093,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002204173,"about_ca_topic_score_gemma":0.006414581,"domain_scores_codex":[0.9982334,0.000648224,0.0002523165,0.0001043763,0.0006256636,0.0001359968],"domain_scores_gemma":[0.9980292,0.001257195,0.00005590274,0.0001027759,0.000535049,0.00001981906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003863982,0.0000460308,0.000523812,0.00001082373,0.000001079557,1.854479e-7,0.00344389,0.0003261524,0.00002013237,0.9917955,0.0002086201,0.003619842],"study_design_scores_gemma":[0.0009259611,0.0005444579,0.001895653,0.0002219102,0.0002694307,0.00001173593,0.1931376,0.3402788,0.01907986,0.1535742,0.2893538,0.0007064435],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3829271,0.0003916811,0.2186995,0.006458846,0.003153201,0.005701725,0.00001261136,0.00006979851,0.3825855],"genre_scores_gemma":[0.9990846,0.000005466813,0.0004673172,0.00003208119,0.00007801026,0.0001625953,0.000001497055,0.000003047461,0.0001654209],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8382213,"threshold_uncertainty_score":0.6308099,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1818060913205782,"score_gpt":0.451292275206263,"score_spread":0.2694861838856848,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}