{"id":"W6910410365","doi":"10.48448/mrw0-4a11","title":"Re-Invoke: Tool Invocation Rewriting for Zero-Shot Tool Retrieval","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Leverage (statistics); Search engine indexing; Ranking (information retrieval); Bottleneck; Key (lock); Set (abstract data type); Inference; SPARQL; Context (archaeology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.004780531,0.0007017175,0.0006335336,0.002023714,0.0004048034,0.001019542,0.001633191,0.0005473712,0.0007642683],"category_scores_gemma":[0.003553766,0.0006860094,0.0001976714,0.003663521,0.001414435,0.0007018125,0.0004647752,0.0006964057,0.008649674],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001121968,"about_ca_system_score_gemma":0.00197816,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001729839,"about_ca_topic_score_gemma":0.0005296552,"domain_scores_codex":[0.9936415,0.00008146525,0.000936476,0.002074744,0.002007481,0.001258328],"domain_scores_gemma":[0.9968649,0.0002302825,0.0006646535,0.001463435,0.000555435,0.0002213527],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00006282063,0.00009635533,0.00002408264,0.0005827099,0.00005700905,0.00001577872,0.0002637689,0.00002928975,0.02131269,0.02306978,0.9496597,0.004826024],"study_design_scores_gemma":[0.001039807,0.000330625,0.00001544604,0.002227769,0.0002817719,0.00002764828,0.0008430923,0.02396395,0.006308129,0.01732621,0.9457575,0.001878071],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0006826585,0.002626572,0.02246457,0.00202629,0.005939741,0.005853101,0.00207994,0.004844235,0.9534829],"genre_scores_gemma":[0.03306895,0.0000886993,0.06853881,0.001116833,0.004111201,0.0002059265,0.0008585692,0.003672583,0.8883384],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.06514446,"threshold_uncertainty_score":0.9995591,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06129313179942047,"score_gpt":0.3517099969563268,"score_spread":0.2904168651569063,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}