A Lightweight Defense Architecture Against Indirect Prompt Injection in Tool-Calling AI Agents
{"The":[0],"rapid":[1],"integration":[2],"of":[3,156],"Large":[4],"Language":[5],"Models":[6],"(LLMs)":[7],"into":[8],"autonomous":[9],"agent":[10],"workflows":[11],"has":[12],"introduced":[13],"significant":[14],"security":[15],"challenges,":[16],"most":[17],"notably":[18],"Indirect":[19],"Prompt":[20],"Injection":[21],"(IPI).":[22],"As":[23],"agents":[24],"increasingly":[25],"rely":[26],"on":[27],"external":[28],"tools":[29],"and":[30,46,78,118,129,150],"protocols":[31],"such":[32],"as":[33],"the":[34,52,94,136],"Model":[35],"Context":[36],"Protocol":[37],"(MCP)":[38],"to":[39,57,92,113,143],"fetch":[40],"data":[41],"from":[42,109,141],"APIs,":[43],"web":[44],"searches,":[45],"databases,":[47],"untrusted":[48],"content":[49],"can":[50],"subvert":[51],"agent’s":[53],"core":[54],"instructions,":[55],"leading":[56],"unauthorized":[58],"actions.":[59],"While":[60],"foundation-model":[61],"evaluators":[62],"offer":[63],"a":[64,85,104],"potential":[65],"defense,":[66],"their":[67],"deployment":[68],"is":[69],"often":[70],"hindered":[71],"by":[72],"high":[73],"inference":[74],"latency":[75,155],"(400–1200":[76],"ms)":[77],"substantial":[79],"operational":[80],"costs.":[81],"This":[82],"paper":[83],"proposes":[84],"lightweight":[86],"multi-stage":[87],"guardrail":[88],"architecture,":[89],"\\"LightGuard-Agent,\\"":[90],"designed":[91],"secure":[93],"tool-calling":[95],"loop":[96],"without":[97],"compromising":[98],"system":[99],"performance.":[100],"Our":[101],"approach":[102],"utilizes":[103],"tiered":[105],"defense":[106],"pipeline,":[107],"ranging":[108],"deterministic":[110],"syntactic":[111],"sanitization":[112],"high-speed":[114],"semantic":[115],"intent":[116],"classification":[117],"an":[119,152],"execution":[120],"policy":[121],"action":[122],"firewall.":[123],"Empirical":[124],"evaluation":[125],"across":[126],"50":[127],"adversarial":[128],"benign":[130],"scenarios":[131],"demonstrates":[132],"that":[133],"LightGuard-Agent":[134],"reduces":[135],"Attack":[137],"Success":[138],"Rate":[139],"(ASR)":[140],"100.0%":[142],"0.0%":[144],"while":[145],"introducing":[146],"zero":[147],"false":[148],"positives":[149],"maintaining":[151],"average":[153],"processing":[154],"just":[157],"0.059":[158],"ms.":[159]}
Authors
- Tarek gamal khallaf mohamed
Publication Details
- Journal
- Zenodo (CERN European Organization for Nuclear Research)
- Published
- 2026-09-13
- DOI
- https://doi.org/10.5281/zenodo.22731103
- Primary Topic
- Adversarial Robustness in Machine Learning
- Type
- article
- Field-Weighted Citation Impact
- 0.00