{"id":"W7082285842","doi":"10.48448/nb3x-h842","title":"Small Encoders Can Rival Large Decoders in Detecting Groundedness","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; Université de Montréal","funders":"","keywords":"Inference; Context (archaeology); Encoder; Latency (audio); Consistency (knowledge bases); Speculation; Natural language; Context model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003603733,0.001730767,0.0008492501,0.001200591,0.0005693815,0.002115498,0.002259897,0.001544339,0.0090008],"category_scores_gemma":[0.01648664,0.0009211216,0.0008563926,0.0009633339,0.001173663,0.006303131,0.003210697,0.002704335,0.006007236],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001450696,"about_ca_system_score_gemma":0.002485627,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009733567,"about_ca_topic_score_gemma":0.02341325,"domain_scores_codex":[0.9981691,0.000636611,0.0001233962,0.0005596204,0.0003741286,0.0001372095],"domain_scores_gemma":[0.9906964,0.006249292,0.000275999,0.0018123,0.0007143189,0.0002516611],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001330761,0.0004462219,0.007465613,0.001141626,0.0003024481,0.0003612057,0.0005690292,0.04050801,0.04253114,0.01611264,0.03755229,0.8516791],"study_design_scores_gemma":[0.0001615696,0.0003064666,0.001823271,0.0001385406,0.0002090431,0.0002894898,0.0002780255,0.885841,0.04041588,0.04925906,0.02120133,0.00007624505],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09799271,0.005561415,0.7944919,0.00279534,0.0004918889,0.000472221,0.003689851,0.08319092,0.01131373],"genre_scores_gemma":[0.5496955,0.001549466,0.4280302,0.001381089,0.0002055891,0.0003995046,0.00763321,0.002546841,0.008558705],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009733567,"threshold_uncertainty_score":0.03011072,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02137515892210214,"score_gpt":0.2536045800907853,"score_spread":0.2322294211686831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}