{"id":"W7082285842","doi":"10.48448/nb3x-h842","title":"Small Encoders Can Rival Large Decoders in Detecting Groundedness","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; Université de Montréal","funders":"","keywords":"Inference; Context (archaeology); Encoder; Latency (audio); Consistency (knowledge bases); Speculation; Natural language; Context model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001392974,0.0003901263,0.0003944986,0.0009214703,0.0003211647,0.0003077985,0.00335835,0.0003114088,0.0001441043],"category_scores_gemma":[0.0006223645,0.000386235,0.00007355947,0.002200259,0.0004494141,0.0001899359,0.001179368,0.0005872589,0.00002468553],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002464628,"about_ca_system_score_gemma":0.001511657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001563606,"about_ca_topic_score_gemma":0.01014676,"domain_scores_codex":[0.9966199,0.00007227173,0.0003940812,0.001310886,0.0005253285,0.001077498],"domain_scores_gemma":[0.9982817,0.0001334762,0.0002618974,0.001005882,0.0001405445,0.0001764699],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004152008,0.001935089,0.02505335,0.002635391,0.0003242442,0.001595891,0.009205807,0.007390779,0.005717608,0.4848872,0.09060831,0.3706048],"study_design_scores_gemma":[0.002348723,0.0001453157,0.0007394037,0.00168143,0.00004244546,0.00009502995,0.0025032,0.5620015,0.003169232,0.05284403,0.3713947,0.003035055],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.000255876,0.0001442595,0.3277162,0.002147574,0.001356336,0.0003162737,0.00001222994,0.0004982019,0.667553],"genre_scores_gemma":[0.3025551,0.00008223907,0.1383874,0.001860111,0.0004228417,0.00006888112,0.00003351939,0.0000925602,0.5564974],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.5546107,"threshold_uncertainty_score":0.999859,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02137515892210214,"score_gpt":0.2536045800907853,"score_spread":0.2322294211686831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}