{"id":"W2905405817","doi":"10.1002/gea.21720","title":"Blind test evaluation of consistency in macroscopic lithic raw material sorting","year":2018,"lang":"en","type":"article","venue":"Geoarchaeology","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"Saint John Regional Hospital; University of New Brunswick","funders":"Tel Aviv University; University of New Brunswick","keywords":"Consistency (knowledge bases); Sorting; Reliability (semiconductor); Computer science; Classification scheme; Process (computing); Set (abstract data type); Test (biology); Calibration; Strengths and weaknesses; Archaeology; Artificial intelligence; Statistics; Geology; Machine learning; Mathematics; Algorithm; Psychology; Geography; Paleontology; Physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03122961,0.0009014978,0.0009474376,0.002783494,0.0009285962,0.001300649,0.001363149,0.001640315,0.003040263],"category_scores_gemma":[0.1200803,0.0004150797,0.001033311,0.001347253,0.002317523,0.001556643,0.002240514,0.0007260117,0.001108174],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005458169,"about_ca_system_score_gemma":0.0005065575,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006896125,"about_ca_topic_score_gemma":0.0008208644,"domain_scores_codex":[0.9645445,0.01810557,0.0043888,0.004841286,0.007170941,0.0009489247],"domain_scores_gemma":[0.7659695,0.1645574,0.01481143,0.01588141,0.03559732,0.003183075],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.02668099,0.003978833,0.6407513,0.001282012,0.001092566,0.0007967523,0.01736923,0.007284524,0.09583495,0.002363725,0.003882607,0.1986825],"study_design_scores_gemma":[0.0005859633,0.02345167,0.7767897,0.0002344306,0.0004722193,0.001339029,0.006762972,0.03513983,0.1429363,0.003972867,0.007789103,0.0005260045],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9520859,0.0002246678,0.04243943,0.00005565295,0.0002496558,0.0009166285,0.0002108519,0.0003182597,0.003498946],"genre_scores_gemma":[0.9650233,0.00006515938,0.03215744,0.00007238318,0.00007457796,0.0007776158,0.0003377121,0.0001515394,0.00134031],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03122961,"threshold_uncertainty_score":0.1651599,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03668008269430519,"score_gpt":0.3030766005405061,"score_spread":0.2663965178462009,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}