{"id":"W163905041","doi":"","title":"Validation: A Critical First Step in the Evaluation of Systems for Legal Corpus Determination","year":2003,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Thomson Reuters (Canada)","funders":"","keywords":"Computer science; Selection (genetic algorithm); Task (project management); Context (archaeology); Process (computing); Information retrieval; Precision and recall; Recall; Term (time); Artificial intelligence; Data mining; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1701917,0.002314207,0.00204262,0.004133264,0.00330599,0.007294141,0.004802388,0.004258308,0.003121847],"category_scores_gemma":[0.3154035,0.001603756,0.001180156,0.002104487,0.002639821,0.01452444,0.005762773,0.003551807,0.001883159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002443768,"about_ca_system_score_gemma":0.005241707,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004665891,"about_ca_topic_score_gemma":0.003676001,"domain_scores_codex":[0.8534585,0.1026807,0.0124454,0.00503333,0.02365731,0.002724848],"domain_scores_gemma":[0.6302,0.2645359,0.006299514,0.04548072,0.05178124,0.001702685],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005079647,0.006861034,0.08766009,0.005185948,0.001564287,0.0009246694,0.02007506,0.03853432,0.07334752,0.01593939,0.03120148,0.7136266],"study_design_scores_gemma":[0.002686498,0.01789983,0.06822945,0.002710322,0.001077832,0.001792754,0.01587273,0.5129661,0.2544745,0.01913607,0.1021031,0.001050972],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4305417,0.002107543,0.514295,0.003707427,0.0006262097,0.01354033,0.001939903,0.01830726,0.01493458],"genre_scores_gemma":[0.5777487,0.0004680717,0.4080161,0.001073336,0.0001324373,0.005248833,0.003255673,0.001566638,0.002490269],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1701917,"threshold_uncertainty_score":0.9000708,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1818060913205782,"score_gpt":0.451292275206263,"score_spread":0.2694861838856848,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}