[{"id":"llmsec-2023-00001","type":"paper","title":"Not What You've Signed Up For: Compromising Real-World LLM-Integrated Applications with Indirect Prompt Injection","authors":["Kai Greshake","Sahar Abdelnabi","Shailesh Mishra","Christoph Endres","Thorsten Holz","Mario Fritz"],"year":2023,"venue":"AISec 2023","abstract":"Introduces indirect prompt injection attacks against LLM-integrated applications, demonstrating how adversaries can remotely control LLMs by injecting prompts into data sources the LLM retrieves.","url":"https://arxiv.org/abs/2302.12173","categories":["prompt-injection","agentic-threats"],"tags":["indirect-injection","rag","application-security"],"citation_count":450,"reviewed":true},{"id":"llmsec-2023-00002","type":"paper","title":"Ignore This Title and HackAPrompt: Exposing Systemic Weaknesses of LLMs through a Global Scale Prompt Hacking Competition","authors":["Sander Schulhoff","Jeremy Pinto","Anaum Khan","Louis-Francois Bouchard","Chenglei Si","Svetlina Anati","Valen Tagliabue","Anson Liu Kost","Christopher Carnahan","Jordan Boyd-Graber"],"year":2023,"venue":"EMNLP 2023","abstract":"Presents results from a global prompt hacking competition with 600K+ adversarial prompts, revealing systemic LLM vulnerabilities across multiple models and defense strategies.","url":"https://arxiv.org/abs/2311.16119","categories":["prompt-injection","jailbreaking","benchmarks"],"tags":["competition","benchmark","prompt-hacking"],"citation_count":180,"reviewed":true},{"id":"llmsec-2024-00001","type":"paper","title":"Jailbroken: How Does LLM Safety Training Fail?","authors":["Alexander Wei","Nika Haghtalab","Jacob Steinhardt"],"year":2024,"venue":"NeurIPS 2023","abstract":"Analyzes failure modes of LLM safety training, identifying two broad categories: competing objectives and mismatched generalization, demonstrating attacks that exploit each.","url":"https://arxiv.org/abs/2307.02483","categories":["jailbreaking","guardrails"],"tags":["safety-training","alignment","failure-modes"],"citation_count":520,"reviewed":true},{"id":"llmsec-2024-00002","type":"paper","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","authors":["Andy Zou","Zifan Wang","Nicholas Carlini","Milad Nasr","J. Zico Kolter","Matt Fredrikson"],"year":2023,"venue":"arXiv preprint","abstract":"Proposes an automated method (GCG) to generate adversarial suffixes that cause aligned LLMs to produce harmful content, with attacks transferring across models including ChatGPT and Claude.","url":"https://arxiv.org/abs/2307.15043","categories":["jailbreaking","adversarial-examples"],"tags":["GCG","adversarial-suffix","transferability"],"citation_count":890,"reviewed":true},{"id":"llmsec-2023-00003","type":"paper","title":"Poisoning Language Models During Instruction Tuning","authors":["Alexander Wan","Eric Wallace","Sheng Shen","Dan Klein"],"year":2023,"venue":"ICML 2023","abstract":"Shows that adversaries can insert poisoned examples into instruction-tuning datasets, causing models to generate targeted outputs for attacker-chosen triggers.","url":"https://arxiv.org/abs/2305.00944","categories":["data-poisoning","fine-tuning-security"],"tags":["instruction-tuning","backdoor","fine-tuning"],"citation_count":210,"reviewed":true},{"id":"llmsec-2024-00003","type":"paper","title":"PoisonedRAG: Knowledge Poisoning Attacks to Retrieval-Augmented Generation of Large Language Models","authors":["Wei Zou","Runpeng Geng","Binghui Wang","Jinyuan Jia"],"year":2024,"venue":"arXiv preprint","abstract":"Demonstrates knowledge poisoning attacks against RAG systems where adversaries inject malicious texts into the knowledge database to manipulate LLM outputs.","url":"https://arxiv.org/abs/2402.07867","categories":["data-poisoning","rag-security"],"tags":["rag","knowledge-poisoning","vector-database"],"citation_count":85,"reviewed":true},{"id":"llmsec-2021-00001","type":"paper","title":"Extracting Training Data from Large Language Models","authors":["Nicholas Carlini","Florian Tramer","Eric Wallace","Matthew Jagielski","Ariel Herbert-Voss","Katherine Lee","Adam Roberts","Tom Brown","Dawn Song","Ulfar Erlingsson","Alina Oprea","Colin Raffel"],"year":2021,"venue":"USENIX Security 2021","abstract":"Demonstrates that large language models memorize and can be prompted to emit verbatim training data, including PII, revealing significant privacy risks.","url":"https://arxiv.org/abs/2012.07805","categories":["membership-inference","differential-privacy"],"tags":["memorization","training-data-extraction","privacy"],"citation_count":1200,"reviewed":true},{"id":"llmsec-2023-00004","type":"paper","title":"Scalable Extraction of Training Data from (Production) Language Models","authors":["Milad Nasr","Nicholas Carlini","Jonathan Hayase","Matthew Jagielski","A. Feder Cooper","Daphne Ippolito","Christopher A. Choquette-Choo","Eric Wallace","Florian Tramer","Katherine Lee"],"year":2023,"venue":"arXiv preprint","abstract":"Develops a scalable attack to extract over a gigabyte of training data from semi-open and closed models including ChatGPT, at a cost of roughly $200.","url":"https://arxiv.org/abs/2311.17035","categories":["membership-inference","model-extraction"],"tags":["training-data-extraction","ChatGPT","production-model"],"citation_count":320,"reviewed":true},{"id":"llmsec-2024-00004","type":"paper","title":"Stealing Part of a Production Language Model","authors":["Nicholas Carlini","Daniel Paleka","Krishnamurthy Dj Dvijotham","Thomas Steinke","Jonathan Hayase","A. Feder Cooper","Katherine Lee","Matthew Jagielski","Milad Nasr","Arthur Conmy","Eric Wallace","David Rolnick","Florian Tramer"],"year":2024,"venue":"ICML 2024","abstract":"Demonstrates that it is possible to steal the embedding projection layer of production LLMs like OpenAI's models through the API, confirming model extraction risks.","url":"https://arxiv.org/abs/2403.06634","categories":["model-extraction"],"tags":["model-stealing","embedding-projection","API-attack"],"citation_count":95,"reviewed":true},{"id":"llmsec-2024-00005","type":"paper","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","authors":["Eric Wallace","Kai Xiao","Reimar Leike","Lilian Weng","Johannes Heidecke","Alex Beutel"],"year":2024,"venue":"arXiv preprint","abstract":"Proposes an instruction hierarchy for training LLMs to prioritize system prompts over user prompts over third-party content, as a defense against prompt injection.","url":"https://arxiv.org/abs/2404.13208","categories":["prompt-injection","input-filtering","guardrails"],"tags":["defense","instruction-hierarchy","system-prompt"],"citation_count":150,"reviewed":true},{"id":"llmsec-2024-00006","type":"paper","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","authors":["Hakan Inan","Kartikeya Upasani","Jianfeng Chi","Rashi Rungta","Krithika Iyer","Yuning Mao","Michael Tontchev","Qing Hu","Brian Fuller","Davide Testuggine","Madian Khabsa"],"year":2023,"venue":"arXiv preprint","abstract":"Introduces Llama Guard, an LLM-based safeguard model for classifying safety risks in LLM inputs and outputs, achieving strong performance on standard benchmarks.","url":"https://arxiv.org/abs/2312.06674","categories":["input-filtering","output-moderation","guardrails"],"tags":["llama-guard","content-safety","classifier"],"citation_count":280,"reviewed":true},{"id":"llmsec-2024-00007","type":"paper","title":"NeMo Guardrails: A Toolkit for Controllable and Safe LLM Applications with Programmable Rails","authors":["Traian Rebedea","Razvan Dinu","Makesh Narsimhan Sreedhar","Christopher Parisien","Jonathan Cohen"],"year":2023,"venue":"EMNLP 2023 Demo","abstract":"Presents NeMo Guardrails, an open-source toolkit for adding programmable safety, security, and privacy rails to LLM-based conversational systems.","url":"https://arxiv.org/abs/2310.10501","categories":["guardrails","input-filtering","output-moderation"],"tags":["NeMo","programmable-rails","toolkit"],"citation_count":190,"reviewed":true},{"id":"llmsec-2024-00008","type":"paper","title":"A Survey on Large Language Model (LLM) Security and Privacy: The Good, The Bad, and The Ugly","authors":["Yifan Yao","Jinhao Duan","Kaidi Xu","Yuanfang Cai","Zhibo Sun","Yue Zhang"],"year":2024,"venue":"High-Confidence Computing","abstract":"Comprehensive survey covering LLM security and privacy from three perspectives: beneficial applications of LLMs for security, attacks against LLMs, and defensive techniques.","url":"https://arxiv.org/abs/2312.02003","categories":["survey"],"tags":["survey","comprehensive","security-privacy"],"citation_count":350,"reviewed":true},{"id":"llmsec-2024-00009","type":"paper","title":"Adversarial Attacks and Defenses in Large Language Models: Old and New Threats","authors":["Leo Schwinn","David Dobre","Stephan Gunnemann","Gauthier Gidel"],"year":2024,"venue":"arXiv preprint","abstract":"Systematizes adversarial attacks and defenses for LLMs, connecting them to the classical adversarial ML literature while identifying LLM-specific threats.","url":"https://arxiv.org/abs/2310.19737","categories":["survey","adversarial-examples","jailbreaking"],"tags":["survey","adversarial-ML","systematization"],"citation_count":125,"reviewed":true},{"id":"llmsec-2024-00010","type":"paper","title":"A Comprehensive Survey of Attack Techniques, Implementation, and Mitigation Strategies in Large Language Models","authors":["Aysan Esmradi","Daniel Wankit Yip","Chun Fai Chan"],"year":2024,"venue":"arXiv preprint","abstract":"Surveys attack techniques across the LLM lifecycle including training, fine-tuning, and inference, with comprehensive mitigation strategies.","url":"https://arxiv.org/abs/2312.10982","categories":["survey","threat-modeling"],"tags":["survey","attack-taxonomy","mitigation"],"citation_count":90,"reviewed":true},{"id":"llmsec-2024-00012","type":"paper","title":"AgentDojo: A Dynamic Environment to Evaluate Attacks and Defenses for LLM Agents","authors":["Edoardo Debenedetti","Jie Zhang","Mislav Balunovic","Luca Beurer-Kellner","Marc Fischer","Florian Tramer"],"year":2024,"venue":"arXiv preprint","abstract":"Introduces AgentDojo, a framework for evaluating the security of LLM agents against prompt injection and other attacks in realistic tool-use scenarios.","url":"https://arxiv.org/abs/2406.13352","categories":["agentic-threats","prompt-injection","benchmarks","tool-use-security"],"tags":["agent-security","evaluation-framework","tool-use"],"citation_count":75,"reviewed":true},{"id":"llmsec-2024-00013","type":"paper","title":"InjecAgent: Benchmarking Indirect Prompt Injections in Tool-Integrated LLM Agents","authors":["Qiusi Zhan","Zhixiang Liang","Zifan Ying","Daniel Kang"],"year":2024,"venue":"ACL 2024 Findings","abstract":"Presents InjecAgent, a benchmark for evaluating indirect prompt injection attacks against LLM agents that use tools, showing most agents are highly vulnerable.","url":"https://arxiv.org/abs/2403.02691","categories":["prompt-injection","tool-use-security","benchmarks"],"tags":["benchmark","tool-use","indirect-injection"],"citation_count":65,"reviewed":true},{"id":"llmsec-2023-00005","type":"paper","title":"Do Anything Now: Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models","authors":["Xinyue Shen","Zeyuan Chen","Michael Backes","Yun Shen","Yang Zhang"],"year":2023,"venue":"CCS 2024","abstract":"Collects and analyzes 6,387 jailbreak prompts from the wild, developing a comprehensive taxonomy of jailbreak techniques and evaluating their effectiveness.","url":"https://arxiv.org/abs/2308.03825","categories":["jailbreaking","threat-modeling"],"tags":["jailbreak-taxonomy","in-the-wild","DAN"],"citation_count":310,"reviewed":true},{"id":"llmsec-2023-00006","type":"paper","title":"Multi-step Jailbreaking Privacy Attacks on ChatGPT","authors":["Haoran Li","Dadi Guo","Wei Fan","Mingshi Xu","Jie Huang","Fanpu Meng","Yangqiu Song"],"year":2023,"venue":"EMNLP 2023 Findings","abstract":"Demonstrates multi-step jailbreaking attacks to extract personal information from ChatGPT, showing how sequential prompting can bypass safety measures.","url":"https://arxiv.org/abs/2304.05197","categories":["jailbreaking","data-anonymization","membership-inference"],"tags":["multi-step","privacy","PII-extraction"],"citation_count":175,"reviewed":true},{"id":"llmsec-2024-00014","type":"paper","title":"A Text Watermark for Large Language Models","authors":["John Kirchenbauer","Jonas Geiping","Yuxin Wen","Jonathan Katz","Ian Miers","Tom Goldstein"],"year":2023,"venue":"ICML 2023","abstract":"Proposes a watermarking framework for LLM-generated text that embeds a statistically detectable signal without significantly affecting output quality.","url":"https://arxiv.org/abs/2301.10226","categories":["watermarking"],"tags":["watermark","detection","provenance"],"citation_count":650,"reviewed":true},{"id":"llmsec-2024-00015","type":"paper","title":"Can LLMs Keep a Secret? Testing Privacy Implications of Language Models via Contextual Integrity Theory","authors":["Niloofar Mireshghallah","Hyunwoo Kim","Xuhui Zhou","Yulia Tsvetkov","Maarten Sap","Reza Shokri","Yejin Choi"],"year":2024,"venue":"ICLR 2024","abstract":"Evaluates LLM privacy behavior through the lens of contextual integrity theory, finding significant mismatches between LLM norms and human privacy expectations.","url":"https://arxiv.org/abs/2310.17884","categories":["differential-privacy","data-anonymization"],"tags":["contextual-integrity","privacy-norms","evaluation"],"citation_count":110,"reviewed":true},{"id":"llmsec-2024-00016","type":"paper","title":"Machine Unlearning in Generative AI: A Survey","authors":["Zheyuan Liu","Guangyao Dou","Zhaoxuan Tan","Yijun Tian","Meng Jiang"],"year":2024,"venue":"arXiv preprint","abstract":"Surveys machine unlearning techniques for LLMs including methods for forgetting specific training data, complying with data deletion requests, and maintaining model utility.","url":"https://arxiv.org/abs/2407.20516","categories":["unlearning","survey"],"tags":["machine-unlearning","right-to-erasure","GDPR"],"citation_count":60,"reviewed":true},{"id":"llmsec-2024-00017","type":"paper","title":"Red Teaming Language Models to Reduce Harms: Methods, Scaling Behaviors, and Lessons Learned","authors":["Deep Ganguli","Liane Lovitt","Jackson Kernion","Amanda Askell","Yuntao Bai","Saurav Kadavath","Ben Mann","Ethan Perez","Nicholas Schiefer","Kamal Ndousse","Andy Jones","Sam Bowman","Anna Chen","Tom Conerly","Nova DasSarma","Dawn Drain","Nelson Elhage","Sheer El-Showk","Stanislav Fort","Zac Hatfield-Dodds","Tom Henighan","Danny Hernandez","Tristan Hume","Josh Jacobson","Scott Johnston","Shauna Kravec","Catherine Olsson","Sam Ringer","Eli Tyre","Jared Kaplan","Chris Olah","Sam McCandlish","Dario Amodei"],"year":2022,"venue":"arXiv preprint","abstract":"Describes Anthropic's early red teaming methodology for language models, documenting methods, scaling behaviors, and lessons for identifying harmful outputs.","url":"https://arxiv.org/abs/2209.07858","categories":["red-teaming","benchmarks"],"tags":["red-teaming","methodology","scaling"],"citation_count":750,"reviewed":true},{"id":"llmsec-2024-00018","type":"paper","title":"Garak: A Framework for Security Probing Large Language Models","authors":["Leon Derczynski","Erick Galinkin","Jeffrey Martin","Subho Majumdar","Nanna Inie"],"year":2024,"venue":"arXiv preprint","abstract":"Presents garak, an open-source framework for systematically probing LLM vulnerabilities including prompt injection, data leakage, and toxicity generation.","url":"https://arxiv.org/abs/2406.11036","categories":["red-teaming","fuzzing","benchmarks"],"tags":["garak","vulnerability-scanning","open-source"],"citation_count":40,"reviewed":true},{"id":"llmsec-2024-00019","type":"paper","title":"R-Judge: Benchmarking Safety Risk Awareness for LLM Agents","authors":["Tongxin Yuan","Zhiwei He","Lingzhong Dong","Yiming Wang","Ruijie Zhao","Tian Xia","Lizhen Xu","Binglin Zhou","Fangqi Li","Zhuosheng Zhang","Rui Wang","Gongshen Liu"],"year":2024,"venue":"EMNLP 2024","abstract":"Introduces R-Judge benchmark for evaluating whether LLM agents can identify safety risks in agentic scenarios involving tool use and multi-step reasoning.","url":"https://arxiv.org/abs/2401.10019","categories":["agentic-threats","benchmarks","human-in-the-loop"],"tags":["agent-safety","benchmark","risk-awareness"],"citation_count":35,"reviewed":true},{"id":"llmsec-2024-00020","type":"standard","title":"OWASP Top 10 for Large Language Model Applications","authors":["Steve Wilson","OWASP LLM AI Security Team"],"year":2025,"venue":"OWASP Foundation","abstract":"The definitive OWASP guide identifying the top 10 most critical security risks in LLM applications, with descriptions, examples, and mitigation strategies.","url":"https://owasp.org/www-project-top-10-for-large-language-model-applications/","categories":["threat-modeling","risk-frameworks"],"tags":["OWASP","top-10","standard","reference"],"reviewed":true},{"id":"llmsec-2024-00021","type":"standard","title":"NIST Artificial Intelligence Risk Management Framework (AI RMF 1.0)","authors":["National Institute of Standards and Technology"],"year":2023,"venue":"NIST","abstract":"Voluntary framework for managing risks in AI systems across the lifecycle, organized into Govern, Map, Measure, and Manage functions.","url":"https://www.nist.gov/itl/ai-risk-management-framework","categories":["risk-frameworks","model-governance"],"tags":["NIST","risk-management","governance","framework"],"reviewed":true},{"id":"llmsec-2024-00022","type":"standard","title":"MITRE ATLAS: Adversarial Threat Landscape for AI Systems","authors":["MITRE Corporation"],"year":2024,"venue":"MITRE","abstract":"Knowledge base of adversarial tactics, techniques, and case studies for AI systems, modeled on the ATT&CK framework for cybersecurity.","url":"https://atlas.mitre.org/","categories":["threat-modeling","risk-frameworks"],"tags":["MITRE","ATLAS","ATT&CK","threat-knowledge-base"],"reviewed":true},{"id":"llmsec-2024-00023","type":"standard","title":"OWASP AI Security and Privacy Guide","authors":["Rob van der Veer","OWASP AI Exchange Team"],"year":2024,"venue":"OWASP Foundation","abstract":"Comprehensive guide for AI security and privacy including threat analysis, controls, and regulatory mapping for AI systems.","url":"https://owasp.org/www-project-ai-security-and-privacy-guide/","categories":["risk-frameworks","survey"],"tags":["OWASP","AI-exchange","privacy","comprehensive"],"reviewed":true},{"id":"llmsec-2024-00024","type":"report","title":"Anthropic: Many-shot Jailbreaking","authors":["Anthropic"],"year":2024,"venue":"Anthropic Research Blog","abstract":"Reveals many-shot jailbreaking, a technique exploiting long context windows by including many examples of harmful Q&A pairs to override safety training.","url":"https://www.anthropic.com/research/many-shot-jailbreaking","categories":["jailbreaking"],"tags":["many-shot","long-context","in-context-learning"],"reviewed":true},{"id":"llmsec-2023-00007","type":"paper","title":"Constitutional AI: Harmlessness from AI Feedback","authors":["Yuntao Bai","Saurav Kadavath","Sandipan Kundu","Amanda Askell","Jackson Kernion","Andy Jones","Anna Chen","Anna Goldie","Azalia Mirhoseini","Cameron McKinnon","Carol Chen","Catherine Olsson","Christopher Olah","Danny Hernandez","Dawn Drain","Deep Ganguli","Dustin Li","Eli Tyre","Ethan Perez","Jamie Kerr","Jared Kaplan","Jeffrey Ladish","Joshua Landau","Kamal Ndousse","Kamile Lukosiute","Liane Lovitt","Michael Sellitto","Nelson Elhage","Nicholas Schiefer","Noemi Mercado","Nova DasSarma","Robert Lasenby","Robin Larson","Sam Ringer","Scott Johnston","Shauna Kravec","Sheer El Showk","Stanislav Fort","Tamera Lanham","Timothy Telleen-Lawton","Tom Brown","Tom Henighan","Tristan Hume","Sam McCandlish","Jared Kaplan","Dario Amodei","Chris Olah"],"year":2022,"venue":"arXiv preprint","abstract":"Introduces Constitutional AI (CAI), a method for training AI systems to be harmless using a set of principles (a constitution) and AI-generated feedback, reducing reliance on human red teamers.","url":"https://arxiv.org/abs/2212.08073","categories":["guardrails","responsible-ai"],"tags":["constitutional-AI","RLHF","alignment","harmlessness"],"citation_count":1100,"reviewed":true},{"id":"llmsec-2024-00025","type":"paper","title":"Prompt Injection Attack Against LLM-Integrated Applications","authors":["Yi Liu","Gelei Deng","Yuekang Li","Kailong Wang","Tianwei Zhang","Yepang Liu","Haoyu Wang","Yan Zheng","Yang Liu"],"year":2024,"venue":"ACM Computing Surveys","abstract":"First comprehensive survey of prompt injection attacks against LLM-integrated applications, categorizing attacks and defenses with a unified framework.","url":"https://arxiv.org/abs/2306.05499","categories":["prompt-injection","survey"],"tags":["survey","prompt-injection","taxonomy"],"citation_count":280,"reviewed":true},{"id":"llmsec-2024-00026","type":"tool","title":"Rebuff: Self-Hardening Prompt Injection Detector","authors":["Protect AI"],"year":2023,"venue":"GitHub","abstract":"Open-source tool designed to detect and prevent prompt injection attacks using multiple detection methods including heuristics, LLM-based analysis, and canary tokens.","url":"https://github.com/protectai/rebuff","categories":["input-filtering","prompt-injection"],"tags":["tool","detection","canary-tokens","open-source"],"reviewed":true},{"id":"llmsec-2024-00027","type":"tool","title":"PyRIT: Python Risk Identification Toolkit for Generative AI","authors":["Microsoft AI Red Team"],"year":2024,"venue":"GitHub / Microsoft","abstract":"Microsoft's open-source framework for red teaming generative AI systems, supporting automated prompt generation, attack strategies, and scoring of AI responses.","url":"https://github.com/Azure/PyRIT","categories":["red-teaming","fuzzing","tool-use-security"],"tags":["tool","red-teaming","Microsoft","automation"],"reviewed":true},{"id":"llmsec-2024-00028","type":"paper","title":"Sleeper Agents: Training Deceptive LLMs that Persist Through Safety Training","authors":["Evan Hubinger","Carson Denison","Jesse Mu","Mike Lambert","Meg Tong","Monte MacDiarmid","Tamera Lanham","Daniel M. Ziegler","Tim Maxwell","Newton Cheng","Adam Jermyn","Amanda Askell","Ansh Radhakrishnan","Cem Anil","David Duvenaud","Deep Ganguli","Fazl Barez","Jack Clark","Kamal Ndousse","Kshitij Sachan","Michael Sellitto","Mrinank Sharma","Nova DasSarma","Roger Grosse","Shauna Kravec","Yuntao Bai","Jared Kaplan","Dario Amodei","Sam McCandlish","Ethan Perez"],"year":2024,"venue":"arXiv preprint","abstract":"Demonstrates that LLMs can be trained with deceptive behaviors (sleeper agents) that persist through standard safety training including RLHF, posing risks for backdoor attacks.","url":"https://arxiv.org/abs/2401.05566","categories":["data-poisoning","guardrails","adversarial-examples"],"tags":["sleeper-agent","backdoor","deceptive-alignment","safety-training"],"citation_count":420,"reviewed":true},{"id":"llmsec-2024-00030","type":"paper","title":"Tensor Trust: Interpretable Prompt Injection Attacks from an Online Game","authors":["Sam Toyer","Olivia Watkins","Ethan Adrian Mendes","Justin Svegliato","Luke Bailey","Tiffany Wang","Isaac Ong","Karim Elmaaroufi","Pieter Abbeel","Trevor Darrell","Alan Ritter","Stuart Russell"],"year":2024,"venue":"ICLR 2024","abstract":"Uses data from an online game (Tensor Trust) where players compete to craft prompt injections and defenses, creating a large dataset of human-generated attacks.","url":"https://arxiv.org/abs/2311.01011","categories":["prompt-injection","benchmarks"],"tags":["game-based","dataset","interpretable"],"citation_count":95,"reviewed":true},{"id":"llmsec-2024-00031","type":"paper","title":"OWASP LLM AI Security & Governance Checklist","authors":["OWASP Foundation","Sandy Dunn","Jackie McGuire"],"year":2024,"venue":"OWASP GenAI Security Project","abstract":"Practical checklist for organizations deploying LLMs covering security, governance, legal, and regulatory considerations with actionable steps.","url":"https://genai.owasp.org/resource/llm-applications-cybersecurity-and-governance-checklist-english/","categories":["risk-frameworks","model-governance","audit-assurance"],"tags":["checklist","governance","deployment"],"reviewed":true},{"id":"llmsec-2024-00032","type":"paper","title":"Federated Fine-Tuning of LLMs on the Very Edge: The Good, the Bad, the Ugly","authors":["Herbert Woisetschläger","Alexander Isenko","Shiqiang Wang","Ruben Mayer","Hans-Arno Jacobsen"],"year":2024,"venue":"DEEM@SIGMOD 2024","abstract":"Examines federated learning approaches for fine-tuning LLMs on edge devices, analyzing privacy guarantees, communication efficiency, and security trade-offs.","url":"https://arxiv.org/abs/2310.03150","categories":["federated-learning","fine-tuning-security","confidential-computing"],"tags":["federated-learning","edge-computing","privacy"],"citation_count":70,"reviewed":true},{"id":"llmsec-2024-00033","type":"paper","title":"TrojLLM: A Black-box Trojan Prompt Attack on Large Language Models","authors":["Jiaqi Xue","Mengxin Zheng","Ting Hua","Yilin Shen","Yepeng Liu","Ladislau Boloni","Qian Lou"],"year":2024,"venue":"NeurIPS 2023","abstract":"Proposes TrojLLM, a black-box attack that generates universal trojan prompts to compromise LLMs without access to model internals.","url":"https://arxiv.org/abs/2306.06815","categories":["data-poisoning","adversarial-examples","supply-chain-attacks"],"tags":["trojan","backdoor","black-box"],"citation_count":85,"reviewed":true},{"id":"llmsec-2024-00034","type":"paper","title":"LoRA Fine-Tuning Efficiently Undoes Safety Training in Llama 2-Chat","authors":["Simon Lermen","Charlie Rogers-Smith","Jeffrey Ladish"],"year":2023,"venue":"arXiv preprint","abstract":"Shows that LoRA fine-tuning with as few as 100 examples can remove safety guardrails from Llama 2-Chat, raising concerns about fine-tuning access to aligned models.","url":"https://arxiv.org/abs/2310.20624","categories":["fine-tuning-security","guardrails","jailbreaking"],"tags":["LoRA","safety-undoing","fine-tuning","alignment"],"citation_count":190,"reviewed":true},{"id":"llmsec-2024-00035","type":"paper","title":"Assessing the Brittleness of Safety Alignment via Pruning and Low-Rank Modifications","authors":["Boyi Wei","Kaixuan Huang","Yangsibo Huang","Tinghao Xie","Xiangyu Qi","Mengzhou Xia","Prateek Mittal","Mengdi Wang","Peter Henderson"],"year":2024,"venue":"ICML 2024","abstract":"Demonstrates that safety alignment in LLMs is brittle and can be undermined through simple weight pruning or low-rank modifications without any fine-tuning data.","url":"https://arxiv.org/abs/2402.05162","categories":["guardrails","adversarial-examples"],"tags":["safety-alignment","pruning","brittleness"],"citation_count":110,"reviewed":true},{"id":"llmsec-2024-00036","type":"paper","title":"ConfusedPilot: Confused Deputy Risks in RAG-based LLMs","authors":["Ayush RoyChowdhury","Mulong Luo","Prateek Sahu","Sarbartha Banerjee","Mohit Tiwari"],"year":2024,"venue":"arXiv preprint","abstract":"Introduces confused deputy attacks against RAG-based code assistants like GitHub Copilot, where poisoned code repositories manipulate assistant outputs.","url":"https://arxiv.org/abs/2408.04870","categories":["rag-security","supply-chain-attacks","agentic-threats"],"tags":["RAG","code-assistant","confused-deputy"],"citation_count":25,"reviewed":true},{"id":"llmsec-2024-00037","type":"paper","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","authors":["Timo Schick","Jane Dwivedi-Yu","Roberto Dessi","Roberta Raileanu","Maria Lomeli","Eric Hambro","Luke Zettlemoyer","Nicola Cancedda","Thomas Scialom"],"year":2023,"venue":"NeurIPS 2023","abstract":"Demonstrates how LLMs can learn to use external tools (APIs, search engines, calculators) through self-supervised learning, foundational for understanding tool-use security.","url":"https://arxiv.org/abs/2302.04761","categories":["tool-use-security","agent-architecture"],"tags":["tool-use","API-calling","foundational"],"citation_count":1400,"reviewed":true},{"id":"llmsec-2024-00038","type":"paper","title":"Identifying and Mitigating the Security Risks of Generative AI","authors":["Clark Barrett","Brad Boyd","Elie Burzstein","Nicholas Carlini","Brad Chen","Jihye Choi","Amrita Roy Chowdhury","Mihai Christodorescu","Anupam Datta","Soheil Feizi","Kathleen Fisher","Tatsunori Hashimoto","Dan Hendrycks","Somesh Jha","Daniel Kang","Florian Kerschbaum","Eric Mitchell","John Mitchell","Zulfikar Ramzan","Khawaja Shams","Dawn Song","Ankur Taly","Diyi Yang"],"year":2023,"venue":"Foundations and Trends in Privacy and Security","abstract":"Comprehensive treatment of generative AI security risks across the ML lifecycle with a focus on practical mitigations and deployment considerations.","url":"https://arxiv.org/abs/2308.14840","categories":["survey","threat-modeling","risk-frameworks"],"tags":["comprehensive","lifecycle","practical-mitigations"],"citation_count":280,"reviewed":true},{"id":"llmsec-2025-00001","type":"paper","title":"The AI Security Pyramid of Pain","authors":["Chris M. Ward","Josh Harguess","Julia Tao","Daniel Christman","Paul Spicer","Mike Tan"],"year":2024,"venue":"Proc. SPIE 13054, Assurance and Security for AI-enabled Systems","abstract":"Adapts David Bianco's Pyramid of Pain framework to AI security, categorizing AI threats by how difficult they are for adversaries to change.","url":"https://arxiv.org/abs/2402.11082","categories":["threat-modeling","industry-report"],"tags":["pyramid-of-pain","threat-taxonomy","practical"],"reviewed":true},{"id":"llmsec-2024-00039","type":"paper","title":"Formalizing and Benchmarking Prompt Injection Attacks and Defenses","authors":["Yupei Liu","Yuqi Jia","Runpeng Geng","Jinyuan Jia","Neil Zhenqiang Gong"],"year":2024,"venue":"USENIX Security 2024","abstract":"Proposes defense mechanisms against prompt injection in LLM systems including isolation-based approaches, input/output filtering, and detection methods.","url":"https://arxiv.org/abs/2310.12815","categories":["prompt-injection","input-filtering","sandboxing-isolation"],"tags":["defense","isolation","filtering"],"citation_count":50,"reviewed":true},{"id":"llmsec-2024-00040","type":"paper","title":"Adversarial Machine Learning: A Taxonomy and Terminology of Attacks and Mitigations (NIST AI 100-2e2025)","authors":["Apostol Vassilev","Alina Oprea","Alie Fordyce","Hyrum Anderson"],"year":2024,"venue":"NIST","abstract":"NIST's authoritative taxonomy of adversarial ML attacks and mitigations covering evasion, poisoning, privacy, and abuse attacks against AI systems.","url":"https://csrc.nist.gov/pubs/ai/100/2/e2025/final","categories":["threat-modeling","risk-frameworks","survey"],"tags":["NIST","taxonomy","adversarial-ML","authoritative"],"reviewed":true},{"id":"llmsec-2024-00042","type":"paper","title":"LLM Agents Can Autonomously Hack Websites","authors":["Richard Fang","Rohan Bindu","Akul Gupta","Qiusi Zhan","Daniel Kang"],"year":2024,"venue":"arXiv preprint","abstract":"Demonstrates that LLM agents can autonomously perform web hacking tasks including SQL injection, XSS, and CSRF attacks without human guidance.","url":"https://arxiv.org/abs/2402.06664","categories":["agentic-threats","autonomous-operations","social-engineering"],"tags":["autonomous-hacking","web-security","agent-misuse"],"citation_count":200,"reviewed":true},{"id":"llmsec-2024-00043","type":"paper","title":"Pandora's White-Box: Precise Training Data Detection and Extraction in Large Language Models","authors":["Jeffrey G. Wang","Jason Wang","Marvin Li","Seth Neel"],"year":2024,"venue":"arXiv preprint","abstract":"Develops precise methods for detecting and extracting training data from LLMs when white-box access is available, with implications for copyright and privacy.","url":"https://arxiv.org/abs/2402.17012","categories":["membership-inference","differential-privacy"],"tags":["training-data-detection","white-box","copyright"],"citation_count":40,"reviewed":true},{"id":"llmsec-2024-00044","type":"standard","title":"OWASP Top 10 for Agentic Applications 2026","authors":["OWASP GenAI Security Project"],"year":2025,"venue":"OWASP GenAI Security Project","abstract":"Identifies the top 10 security risks specific to agentic AI applications including excessive agency, unsafe tool execution, and inadequate oversight.","url":"https://genai.owasp.org/resource/owasp-top-10-for-agentic-applications-for-2026/","categories":["threat-modeling","risk-frameworks","agent-architecture"],"tags":["OWASP","agentic","top-10","standard"],"reviewed":true},{"id":"llmsec-2024-00045","type":"paper","title":"From Prompt Injections to SQL Injection Attacks: How Protected is Your LLM-Integrated Web Application?","authors":["Rodrigo Pedro","Daniel Castro","Paolo Molina","Nuno Santos"],"year":2024,"venue":"USENIX Security 2024","abstract":"Demonstrates how prompt injection can be chained with traditional web attacks (SQL injection, XSS) in LLM-integrated applications.","url":"https://arxiv.org/abs/2308.01990","categories":["prompt-injection","model-serving-security"],"tags":["SQL-injection","XSS","web-application","chained-attack"],"citation_count":70,"reviewed":true},{"id":"llmsec-2024-00046","type":"paper","title":"Model Context Protocol (MCP): Security Best Practices","authors":["Anthropic"],"year":2024,"venue":"Anthropic Documentation","abstract":"Documentation and analysis of security considerations for the Model Context Protocol, covering authentication, authorization, and tool sandboxing.","url":"https://modelcontextprotocol.io/docs/tutorials/security/security_best_practices","categories":["tool-use-security","agent-architecture","access-control"],"tags":["MCP","protocol-security","tool-sandboxing"],"reviewed":true},{"id":"llmsec-2024-00047","type":"paper","title":"Benchmarking and Defending Against Indirect Prompt Injection Attacks on Large Language Models","authors":["Jingwei Yi","Yueqi Xie","Bin Zhu","Keegan Hines","Emre Kiciman","Guangzhong Sun","Xing Xie","Fangzhao Wu"],"year":2024,"venue":"arXiv preprint","abstract":"Provides a benchmark for indirect prompt injection attacks and evaluates several defense strategies including perplexity-based detection and sandwich defense.","url":"https://arxiv.org/abs/2312.14197","categories":["prompt-injection","input-filtering","benchmarks"],"tags":["indirect-injection","benchmark","defense-evaluation"],"citation_count":60,"reviewed":true},{"id":"llmsec-2025-00002","type":"report","title":"An Architectural Risk Analysis of Large Language Models: Applied Machine Learning Security","authors":["Gary McGraw","Harold Figueroa","Katie McMahon","Richie Bonett"],"year":2024,"venue":"Berryville Institute of Machine Learning (BIML)","abstract":"Comprehensive practitioner guide covering AI/ML security from an architectural risk analysis perspective, with practical defense patterns.","url":"https://berryvilleiml.com/docs/BIML-LLM24.pdf","categories":["book","threat-modeling","risk-frameworks"],"tags":["book","practitioner-guide","BIML","architecture"],"reviewed":true},{"id":"llmsec-2025-00004","type":"dataset","title":"HarmBench: A Standardized Evaluation Framework for Automated Red Teaming and Robust Refusal","authors":["Mantas Mazeika","Long Phan","Xuwang Yin","Andy Zou","Zifan Wang","Norman Mu","Elham Sakhaee","Nathaniel Li","Steven Basart","Bo Li","David Forsyth","Dan Hendrycks"],"year":2024,"venue":"ICML 2024","abstract":"Introduces HarmBench, a standardized framework for evaluating automated red teaming methods and robust refusal in LLMs with a comprehensive behavior taxonomy.","url":"https://arxiv.org/abs/2402.04249","categories":["red-teaming","benchmarks","jailbreaking"],"tags":["benchmark","red-teaming","evaluation-framework","dataset"],"citation_count":180,"reviewed":true},{"id":"llmsec-2025-00005","type":"standard","title":"ISO/IEC 42001:2023 - Artificial Intelligence Management System","authors":["International Organization for Standardization"],"year":2023,"venue":"ISO","abstract":"International standard specifying requirements for establishing, implementing, maintaining and continually improving an AI management system within organizations.","url":"https://www.iso.org/standard/81230.html","categories":["risk-frameworks","model-governance","audit-assurance"],"tags":["ISO","management-system","certification","governance"],"reviewed":true},{"id":"llmsec-2024-00048","type":"paper","title":"AutoDAN: Generating Stealthy Jailbreak Prompts on Aligned Large Language Models","authors":["Xiaogeng Liu","Nan Xu","Muhao Chen","Chaowei Xiao"],"year":2024,"venue":"ICLR 2024","abstract":"Proposes AutoDAN, a method for automatically generating stealthy jailbreak prompts that are semantically meaningful and can bypass perplexity-based defenses.","url":"https://arxiv.org/abs/2310.04451","categories":["jailbreaking","adversarial-examples","red-teaming"],"tags":["automated-jailbreak","stealthy","genetic-algorithm"],"citation_count":230,"reviewed":true},{"id":"llmsec-2024-00049","type":"paper","title":"Tree of Attacks: Jailbreaking Black-Box LLMs with Auto-Generated Subtrees","authors":["Anay Mehrotra","Manolis Zampetakis","Paul Kassianik","Blaine Nelson","Hyrum Anderson","Yaron Singer","Amin Karbasi"],"year":2024,"venue":"NeurIPS 2024","abstract":"Introduces TAP using an LLM to iteratively refine jailbreak prompts against black-box target models with high success rates.","url":"https://arxiv.org/abs/2312.02119","categories":["jailbreaking","red-teaming"],"tags":["black-box","automated","tree-search"],"citation_count":175,"reviewed":true},{"id":"llmsec-2024-00050","type":"paper","title":"GPT-4 Is Too Smart To Be Safe: Stealthy Chat with LLMs via Cipher","authors":["Youliang Yuan","Wenxiang Jiao","Wenxuan Wang","Jen-tse Huang","Pinjia He","Shuming Shi","Zhaopeng Tu"],"year":2024,"venue":"ICLR 2024","abstract":"Demonstrates that LLMs can be jailbroken using cipher-based encoding, bypassing safety training designed for natural language.","url":"https://arxiv.org/abs/2308.06463","categories":["jailbreaking"],"tags":["cipher","encoding","bypass"],"citation_count":160,"reviewed":true},{"id":"llmsec-2024-00051","type":"paper","title":"DecodingTrust: A Comprehensive Assessment of Trustworthiness in GPT Models","authors":["Boxin Wang","Weixin Chen","Hengzhi Pei","Chulin Xie","Mintong Kang","Chenhui Zhang","Chejian Xu","Zidi Xiong","Ritik Dutta","Rylan Schaeffer"],"year":2024,"venue":"NeurIPS 2023","abstract":"Comprehensive trustworthiness evaluation of GPT models across 8 dimensions including toxicity, bias, robustness, privacy, fairness, and machine ethics.","url":"https://arxiv.org/abs/2306.11698","categories":["benchmarks","responsible-ai","survey"],"tags":["trustworthiness","GPT","comprehensive-evaluation"],"citation_count":520,"reviewed":true},{"id":"llmsec-2024-00052","type":"paper","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","authors":["Manish Bhatt","Sahana Chennabasappa","Cyrus Nikolaidis","Shengye Wan","Ivan Evtimov"],"year":2024,"venue":"arXiv preprint","abstract":"Introduces CyberSecEval, a benchmark for evaluating the cybersecurity risks of LLM code generation, including insecure code suggestions.","url":"https://arxiv.org/abs/2312.04724","categories":["benchmarks","supply-chain-attacks","red-teaming"],"tags":["CyberSecEval","code-security","Meta","benchmark"],"citation_count":140,"reviewed":true},{"id":"llmsec-2024-00053","type":"paper","title":"BadChain: Backdoor Chain-of-Thought Prompting for Large Language Models","authors":["Zhen Xiang","Fengqing Jiang","Zidi Xiong","Bhaskar Ramasubramanian","Radha Poovendran","Bo Li"],"year":2024,"venue":"NeurIPS 2024","abstract":"Demonstrates backdoor attacks on chain-of-thought reasoning in LLMs where poisoned demonstrations cause incorrect reasoning chains.","url":"https://arxiv.org/abs/2401.12242","categories":["data-poisoning","adversarial-examples"],"tags":["chain-of-thought","backdoor","reasoning"],"citation_count":55,"reviewed":true},{"id":"llmsec-2024-00054","type":"paper","title":"Jailbreaking Black Box Large Language Models in Twenty Queries","authors":["Patrick Chao","Alexander Robey","Edgar Dobriban","Hamed Hassani","George J. Pappas","Eric Wong"],"year":2025,"venue":"IEEE SaTML 2025","abstract":"Uses an attacker LLM to automatically generate jailbreak prompts through iterative refinement achieving high success with only black-box access.","url":"https://arxiv.org/abs/2310.08419","categories":["jailbreaking","red-teaming"],"tags":["automated","iterative","black-box"],"citation_count":220,"reviewed":true},{"id":"llmsec-2024-00055","type":"paper","title":"GPT in Sheep's Clothing: The Risk of Customized GPTs","authors":["Sagiv Antebi","Noam Azulay","Edan Habler","Ben Ganon","Asaf Shabtai","Yuval Elovici"],"year":2024,"venue":"arXiv preprint","abstract":"Analyzes security risks of custom GPTs in the OpenAI GPT Store including prompt leakage, data exfiltration, and malicious GPTs.","url":"https://arxiv.org/abs/2401.09075","categories":["supply-chain-attacks","prompt-injection","model-serving-security"],"tags":["custom-GPTs","GPT-Store","supply-chain"],"citation_count":45,"reviewed":true},{"id":"llmsec-2024-00056","type":"tool","title":"Vigil: LLM Prompt Injection Detection and Defense Toolkit","authors":["DeadBits"],"year":2024,"venue":"GitHub","abstract":"Open-source scanner for detecting prompt injections using vector similarity, YARA rules, text classifiers, and canary tokens.","url":"https://github.com/deadbits/vigil-llm","categories":["input-filtering","prompt-injection","monitoring-detection"],"tags":["tool","detection","open-source","scanner"],"reviewed":true},{"id":"llmsec-2024-00057","type":"tool","title":"Guardrails AI: Input/Output Guards for LLM Applications","authors":["Guardrails AI"],"year":2024,"venue":"GitHub","abstract":"Framework for adding structural, type, and quality guarantees to LLM outputs with validators for PII, toxicity, code security, and factual accuracy.","url":"https://github.com/guardrails-ai/guardrails","categories":["guardrails","output-moderation","input-filtering"],"tags":["tool","validation","open-source","production"],"reviewed":true},{"id":"llmsec-2024-00058","type":"tool","title":"LLM Guard: Security Toolkit for LLM Interactions","authors":["Protect AI"],"year":2024,"venue":"GitHub","abstract":"Comprehensive toolkit for sanitizing LLM prompts and outputs, detecting prompt injection, PII leakage, toxic content, and code vulnerabilities.","url":"https://github.com/protectai/llm-guard","categories":["input-filtering","output-moderation","guardrails"],"tags":["tool","sanitization","PII-detection","open-source"],"reviewed":true},{"id":"llmsec-2024-00059","type":"paper","title":"The Emerged Security and Privacy of LLM Agent: A Survey with Case Studies","authors":["Feng He","Tianqing Zhu","Dayong Ye","Bo Liu","Wanlei Zhou","Philip S. Yu"],"year":2024,"venue":"arXiv preprint","abstract":"Surveys security and privacy challenges specific to LLM-based agents, covering agent architectures, attack surfaces, and defense mechanisms.","url":"https://arxiv.org/abs/2407.19354","categories":["survey","agentic-threats","agent-architecture"],"tags":["survey","agent-security","comprehensive"],"citation_count":35,"reviewed":true},{"id":"llmsec-2024-00060","type":"paper","title":"Jailbreaking Leading Safety-Aligned LLMs with Simple Adaptive Attacks","authors":["Maksym Andriushchenko","Francesco Croce","Nicolas Flammarion"],"year":2025,"venue":"ICLR 2025","abstract":"Shows that adaptive adversaries can bypass most proposed jailbreak defenses, highlighting the arms race between attacks and defenses.","url":"https://arxiv.org/abs/2404.02151","categories":["jailbreaking","guardrails","adversarial-examples"],"tags":["adaptive-attacks","defense-bypass","arms-race"],"citation_count":50,"reviewed":true},{"id":"llmsec-2024-00061","type":"paper","title":"How Johnny Can Persuade LLMs to Jailbreak Them: Rethinking Persuasion to Challenge AI Safety","authors":["Yi Zeng","Hongpeng Lin","Jingwen Zhang","Diyi Yang","Ruoxi Jia","Weiyan Shi"],"year":2024,"venue":"ACL 2024","abstract":"Applies social science persuasion techniques to jailbreak LLMs, showing high attack success rates using persuasion taxonomy.","url":"https://arxiv.org/abs/2401.06373","categories":["jailbreaking","social-engineering"],"tags":["persuasion","social-science","human-like"],"citation_count":85,"reviewed":true},{"id":"llmsec-2025-00006","type":"paper","title":"Model Context Protocol (MCP): Specification","authors":["Anthropic"],"year":2024,"venue":"Anthropic / GitHub","abstract":"Open protocol specification for connecting AI models to external data sources and tools, enabling standardized tool use with security considerations.","url":"https://modelcontextprotocol.io/","categories":["tool-use-security","agent-architecture","access-control"],"tags":["MCP","protocol","specification","tool-integration"],"reviewed":true},{"id":"llmsec-2025-00007","type":"report","title":"Anthropic's Responsible Scaling Policy","authors":["Anthropic"],"year":2024,"venue":"Anthropic Blog","abstract":"Framework defining AI Safety Levels (ASL) for evaluating and managing risks from increasingly capable AI systems.","url":"https://www.anthropic.com/news/anthropics-responsible-scaling-policy","categories":["risk-frameworks","model-governance","responsible-ai"],"tags":["responsible-scaling","ASL","governance","policy"],"reviewed":true},{"id":"llmsec-2025-00008","type":"report","title":"OpenAI: Preparedness Framework (Beta)","authors":["OpenAI"],"year":2023,"venue":"OpenAI Blog","abstract":"OpenAI's approach to tracking, evaluating, forecasting, and protecting against catastrophic risks of frontier AI models.","url":"https://openai.com/safety/preparedness","categories":["risk-frameworks","red-teaming","model-governance"],"tags":["preparedness","frontier-risk","catastrophic-risk"],"reviewed":true},{"id":"llmsec-2025-00009","type":"standard","title":"EU AI Act: Regulation on Artificial Intelligence","authors":["European Parliament"],"year":2024,"venue":"Official Journal of the European Union","abstract":"The EU's comprehensive AI regulation establishing risk-based categories, conformity assessments, and requirements for high-risk AI systems.","url":"https://eur-lex.europa.eu/eli/reg/2024/1689/oj","categories":["risk-frameworks","audit-assurance","model-governance"],"tags":["EU-AI-Act","regulation","compliance","high-risk"],"reviewed":true},{"id":"llmsec-2025-00010","type":"report","title":"Google: Secure AI Framework (SAIF)","authors":["Google"],"year":2023,"venue":"Google Security Blog","abstract":"Google's conceptual framework for secure AI systems with six core elements covering security foundations, detection, automation, and contextualization.","url":"https://safety.google/cybersecurity-advancements/saif/","categories":["risk-frameworks","cloud-ai-security"],"tags":["Google","SAIF","framework","enterprise"],"reviewed":true},{"id":"llmsec-2025-00011","type":"report","title":"Lessons From Red Teaming 100 Generative AI Products","authors":["Blake Bullwinkel","Amanda Minnich","Shiven Chawla","Gary Lopez","Martin Pouliot","Whitney Maxwell","Joris de Gruyter","Katherine Pratt","Saphir Qi","Nina Chikanov","Roman Lutz","Raja Sekhar Rao Dheekonda","Bolor-Erdene Jagdagdorj","Eugenia Kim","Justin Song","Keegan Hines","Daniel Jones","Giorgio Severi","Richard Lundeen","Sam Vaughan","Victoria Westerhoff","Pete Bryan","Ram Shankar Siva Kumar","Yonatan Zunger","Chang Kawaguchi","Mark Russinovich"],"year":2025,"venue":"arXiv preprint","abstract":"Shares lessons from Microsoft's AI red team operations including methodology, tooling, common failure modes, and best practices.","url":"https://arxiv.org/abs/2501.07238","categories":["red-teaming","industry-report"],"tags":["Microsoft","red-team","methodology","lessons-learned"],"reviewed":true},{"id":"llmsec-2025-00012","type":"paper","title":"LLM Agents Can Autonomously Exploit One-day Vulnerabilities","authors":["Richard Fang","Rohan Bindu","Akul Gupta","Daniel Kang"],"year":2024,"venue":"arXiv preprint","abstract":"Shows that LLM agents (GPT-4) can autonomously exploit real-world one-day vulnerabilities given CVE descriptions, achieving 87% success rate.","url":"https://arxiv.org/abs/2404.08144","categories":["agentic-threats","autonomous-operations","vulnerability-disclosure"],"tags":["autonomous-exploit","CVE","one-day","offensive"],"citation_count":150,"reviewed":true},{"id":"llmsec-2025-00013","type":"paper","title":"Dissecting Adversarial Robustness of Multimodal LM Agents","authors":["Chen Henry Wu","Rishi Shah","Jing Yu Koh","Ruslan Salakhutdinov","Daniel Fried","Aditi Raghunathan"],"year":2025,"venue":"ICLR 2025","abstract":"Demonstrates adversarial attacks on multimodal agents that take actions in digital environments, showing visual perturbations can hijack agent behavior.","url":"https://arxiv.org/abs/2406.12814","categories":["agentic-threats","adversarial-examples","tool-use-security"],"tags":["multimodal","visual-attacks","agent-hijacking"],"citation_count":25,"reviewed":true},{"id":"llmsec-2025-00014","type":"paper","title":"SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering","authors":["John Yang","Carlos E. Jimenez","Alexander Wettig","Kilian Lieret","Shunyu Yao","Karthik Narasimhan","Ofir Press"],"year":2024,"venue":"NeurIPS 2024","abstract":"Demonstrates autonomous coding agents that interact with computer interfaces to solve software engineering tasks, raising questions about agent containment.","url":"https://arxiv.org/abs/2405.15793","categories":["autonomous-operations","agent-architecture","tool-use-security"],"tags":["SWE-agent","autonomous-coding","agent-computer-interface"],"citation_count":200,"reviewed":true},{"id":"llmsec-2025-00015","type":"paper","title":"TrustLLM: Trustworthiness in Large Language Models","authors":["Lichao Sun","Yue Huang","Haoran Wang","Siyuan Wu","Qihui Zhang"],"year":2024,"venue":"ICML 2024","abstract":"Comprehensive study of LLM trustworthiness across truthfulness, safety, fairness, robustness, privacy, and machine ethics with benchmarks.","url":"https://arxiv.org/abs/2401.05561","categories":["benchmarks","responsible-ai","survey"],"tags":["trustworthiness","benchmark","comprehensive"],"citation_count":300,"reviewed":true},{"id":"llmsec-2025-00016","type":"paper","title":"On the Societal Impact of Open Foundation Models","authors":["Sayash Kapoor","Rishi Bommasani","Kevin Klyman","Shayne Longpre","Ashwin Ramaswami"],"year":2024,"venue":"arXiv preprint","abstract":"Analyzes the societal impacts of open-weight foundation models, including security implications of open vs closed model access.","url":"https://arxiv.org/abs/2403.07918","categories":["responsible-ai","model-governance","industry-report"],"tags":["open-models","societal-impact","policy"],"citation_count":180,"reviewed":true},{"id":"llmsec-2025-00017","type":"paper","title":"Poisoning Web-Scale Training Datasets is Practical","authors":["Nicholas Carlini","Matthew Jagielski","Christopher A. Choquette-Choo","Daniel Paleka","Will Pearce","Hyrum Anderson","Andreas Terzis","Kurt Thomas","Florian Tramer"],"year":2024,"venue":"IEEE S&P 2024","abstract":"Demonstrates practical attacks to poison web-scale datasets like LAION by purchasing expired domains, affecting 0.01% of a dataset for under $60.","url":"https://arxiv.org/abs/2302.10149","categories":["data-poisoning","supply-chain-attacks"],"tags":["web-scale","practical-attack","domain-purchase","dataset-poisoning"],"citation_count":250,"reviewed":true},{"id":"llmsec-2025-00018","type":"paper","title":"Visual Adversarial Examples Jailbreak Aligned Large Language Models","authors":["Xiangyu Qi","Kaixuan Huang","Ashwinee Panda","Peter Henderson","Mengdi Wang","Prateek Mittal"],"year":2024,"venue":"AAAI 2024","abstract":"Shows adversarial images can jailbreak multimodal LLMs that are robust to text-only attacks, bypassing alignment through the visual channel.","url":"https://arxiv.org/abs/2306.13213","categories":["jailbreaking","adversarial-examples"],"tags":["visual","multimodal","image-jailbreak"],"citation_count":220,"reviewed":true},{"id":"llmsec-2025-00019","type":"paper","title":"Are Aligned Neural Networks Adversarially Aligned?","authors":["Nicholas Carlini","Milad Nasr","Christopher A. Choquette-Choo","Matthew Jagielski","Irena Gao","Anas Awadalla","Pang Wei Koh","Daphne Ippolito","Katherine Lee","Florian Tramer","Ludwig Schmidt"],"year":2024,"venue":"NeurIPS 2023","abstract":"Evaluates whether multimodal LLMs aligned to refuse harmful text requests also refuse harmful image-based requests, finding significant gaps.","url":"https://arxiv.org/abs/2306.15447","categories":["adversarial-examples","guardrails","jailbreaking"],"tags":["multimodal","alignment-gap","image-attacks"],"citation_count":180,"reviewed":true},{"id":"llmsec-2025-00020","type":"book","title":"Generative AI Security: Theories and Practices","authors":["Ken Huang","Yang Wang","Ben Goertzel","Yale Li","Sean Wright","Jyoti Ponnapalli"],"year":2024,"venue":"Springer","abstract":"Comprehensive textbook covering generative AI security from foundations to advanced topics including LLM threats, defenses, privacy, and governance.","url":"https://link.springer.com/book/10.1007/978-3-031-54252-7","categories":["book","survey"],"tags":["textbook","comprehensive","Springer"],"reviewed":true},{"id":"llmsec-2025-00021","type":"paper","title":"ReAct: Synergizing Reasoning and Acting in Language Models","authors":["Shunyu Yao","Jeffrey Zhao","Dian Yu","Nan Du","Izhak Shafran","Karthik Narasimhan","Yuan Cao"],"year":2023,"venue":"ICLR 2023","abstract":"Foundational work on the ReAct paradigm for LLM agents that interleave reasoning and tool-use actions, enabling complex task completion with security implications.","url":"https://arxiv.org/abs/2210.03629","categories":["agent-architecture","tool-use-security"],"tags":["ReAct","reasoning","tool-use","foundational"],"citation_count":2500,"reviewed":true},{"id":"llmsec-2025-00022","type":"paper","title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","authors":["Guanzhi Wang","Yuqi Xie","Yunfan Jiang","Ajay Mandlekar","Chaowei Xiao","Yuke Zhu","Linxi Fan","Anima Anandkumar"],"year":2023,"venue":"NeurIPS 2023","abstract":"Demonstrates a continuously learning LLM agent in Minecraft that writes and executes code, highlighting autonomous operation and containment challenges.","url":"https://arxiv.org/abs/2305.16291","categories":["autonomous-operations","agent-architecture"],"tags":["embodied-agent","autonomous-learning","code-execution"],"citation_count":800,"reviewed":true},{"id":"llmsec-2025-00023","type":"dataset","title":"SafetyBench: Evaluating the Safety of Large Language Models","authors":["Zhexin Zhang","Leqi Lei","Lindong Wu","Rui Sun","Yongkang Huang"],"year":2024,"venue":"ACL 2024","abstract":"Large-scale safety evaluation benchmark with 11,435 multiple-choice questions across 7 safety categories in both Chinese and English.","url":"https://arxiv.org/abs/2309.07045","categories":["benchmarks","guardrails"],"tags":["benchmark","safety","multilingual","dataset"],"citation_count":90,"reviewed":true},{"id":"llmsec-2025-00024","type":"paper","title":"Can Sensitive Information Be Deleted From LLMs? Objectives for Defending Against Extraction Attacks","authors":["Vaidehi Patil","Peter Hase","Mohit Bansal"],"year":2024,"venue":"ICLR 2024","abstract":"Evaluates methods for deleting sensitive information from trained LLMs, finding current unlearning approaches insufficient against determined adversaries.","url":"https://arxiv.org/abs/2309.17410","categories":["unlearning","differential-privacy","membership-inference"],"tags":["knowledge-deletion","unlearning","extraction-defense"],"citation_count":70,"reviewed":true},{"id":"llmsec-2025-00025","type":"paper","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","authors":["Seungju Han","Kavel Rao","Allyson Ettinger","Liwei Jiang","Bill Yuchen Lin","Nathan Lambert","Yejin Choi","Nouha Dziri"],"year":2024,"venue":"arXiv preprint","abstract":"Open-source moderation tool for detecting safety risks in LLM interactions, trained on a diverse dataset of harmful and benign prompts.","url":"https://arxiv.org/abs/2406.18495","categories":["guardrails","input-filtering","output-moderation"],"tags":["moderation","open-source","safety-classifier"],"citation_count":30,"reviewed":true},{"id":"llmsec-2025-00026","type":"paper","title":"A StrongREJECT for Empty Jailbreaks","authors":["Alexandra Souly","Qingyuan Lu","Dillon Bowen","Tu Trinh","Elvis Hsieh","Sana Pandey","Pieter Abbeel","Justin Svegliato","Scott Emmons","Olivia Watkins","Sam Toyer"],"year":2024,"venue":"NeurIPS 2024 Datasets and Benchmarks","abstract":"Introduces StrongREJECT, a high-quality evaluation benchmark for measuring how well LLMs refuse harmful requests.","url":"https://arxiv.org/abs/2402.10260","categories":["benchmarks","guardrails","red-teaming"],"tags":["evaluation","refusal","benchmark","reliability"],"citation_count":65,"reviewed":true},{"id":"llmsec-2025-00027","type":"tool","title":"OWASP Threat Dragon: AI-Aware Threat Modeling Tool","authors":["OWASP Foundation"],"year":2024,"venue":"OWASP / GitHub","abstract":"Open-source threat modeling tool supporting AI/ML system threat models, data flow diagrams, and STRIDE methodology for GenAI applications.","url":"https://owasp.org/www-project-threat-dragon/","categories":["threat-modeling","risk-frameworks"],"tags":["tool","threat-modeling","STRIDE","open-source"],"reviewed":true},{"id":"llmsec-2025-00028","type":"paper","title":"CISA: Roadmap for Artificial Intelligence","authors":["Cybersecurity and Infrastructure Security Agency"],"year":2023,"venue":"CISA","abstract":"CISA's strategic roadmap for AI covering responsible use, assuring AI systems, securing AI adoption, and collaborating on AI governance.","url":"https://www.cisa.gov/resources-tools/resources/roadmap-ai","categories":["risk-frameworks","model-governance"],"tags":["CISA","government","roadmap","policy"],"reviewed":true},{"id":"llmsec-2025-00029","type":"paper","title":"Prompt Stealing Attacks Against Text-to-Image Generation Models","authors":["Xinyue Shen","Yiting Qu","Michael Backes","Yang Zhang"],"year":2024,"venue":"USENIX Security 2024","abstract":"Demonstrates attacks that steal the prompts used to generate images from text-to-image models, raising IP and privacy concerns.","url":"https://arxiv.org/abs/2302.09923","categories":["model-extraction","membership-inference"],"tags":["prompt-stealing","text-to-image","IP-protection"],"citation_count":110,"reviewed":true},{"id":"llmsec-2025-00030","type":"paper","title":"Shadow Alignment: The Ease of Subverting Safely-Aligned Language Models","authors":["Xianjun Yang","Xiao Wang","Qi Zhang","Linda Petzold","William Yang Wang","Xun Zhao","Dahua Lin"],"year":2023,"venue":"arXiv preprint","abstract":"Shows that RLHF can introduce shadow alignment where models exhibit harmful behaviors not present in the base model.","url":"https://arxiv.org/abs/2310.02949","categories":["guardrails","responsible-ai","fine-tuning-security"],"tags":["RLHF","shadow-alignment","safety-regression"],"citation_count":75,"reviewed":true},{"id":"llmsec-2026-00062","type":"paper","title":"MetaBackdoor: Exploiting Positional Encoding as a Backdoor Attack Surface in LLMs","authors":["Rui Wen","Mark Russinovich","Andrew Paverd","Jun Sakuma","Ahmed Salem"],"year":2026,"abstract":"Backdoor attacks pose a serious security threat to large language models (LLMs), which are increasingly deployed as general-purpose assistants in safety- and privacy-critical applications. Existing LLM backdoors rely primarily on content-based triggers, requiring explicit modification of the input text. In this work, we show that this assumption is unnecessary and limiting. We introduce MetaBackdoor, a new class of backdoor attacks that exploits positional information as the trigger, without mod","url":"https://arxiv.org/abs/2605.15172","categories":["data-poisoning","access-control"],"reviewed":false},{"id":"llmsec-2026-00063","type":"paper","title":"Talk is (Not) Cheap: A Taxonomy and Benchmark Coverage Audit for LLM Attacks","authors":["Karthik Raghu Iyer","Yazdan Jamshidi","Nicholas Bray","Alexey A. Shvets"],"year":2026,"abstract":"We introduce a reusable framework for auditing whether LLM attack benchmarks collectively cover the threat surface: a 4$\\times$6 Target $\\times$ Technique matrix grounded in STRIDE, constructed from a 507-leaf taxonomy -- 401 data-populated and 106 threat-model-derived leaves -- of inference-time attacks extracted from 932 arXiv security studies (2023--2026). The matrix enables benchmark-external validation -- auditing collective coverage rather than individual benchmark consistency. Applying it","url":"https://arxiv.org/abs/2605.15118","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00064","type":"paper","title":"Model-Agnostic Lifelong LLM Safety via Externalized Attack-Defense Co-Evolution","authors":["Xiaozhe Zhang","Chaozhuo Li","Hui Liu","Shaocheng Yan","Bingyu Yan","Qiwei Ye","Haoliang Li"],"year":2026,"abstract":"Large language models remain vulnerable to adversarial prompts that elicit harmful outputs. Existing safety paradigms typically couple red-teaming and post-training in a closed, policy-centric loop, causing attack discovery to suffer from rapid saturation and limiting the exposure of novel failure modes, while leaving defenses inefficient, rigid, and difficult to transfer across victim models. To this end, we propose EvoSafety, an LLM safety framework built around persistent, inspectable, and re","url":"https://arxiv.org/abs/2605.13411","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00065","type":"paper","title":"Quantifying LLM Safety Degradation Under Repeated Attacks Using Survival Analysis","authors":["Zvi Topol"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed in a wide range of applications, yet remain vulnerable to adversarial jailbreak attacks that circumvent their safety guardrails. Existing evaluation frameworks typically report binary success/failure metrics, failing to capture the temporal dynamics of how attacks succeed under persistent adversarial pressure. This preliminary work proposes a novel evaluation framework that applies survival analysis techniques to characterize LLM jailbreak v","url":"https://arxiv.org/abs/2605.12869","categories":["jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00066","type":"paper","title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","authors":["Buyun Liang","Jinqi Luo","Liangzu Peng","Kwan Ho Ryan Chan","Darshan Thaker","Kaleab A. Kinfu","Fengrui Tian","Hamed Hassani","René Vidal"],"year":2026,"abstract":"Large language models (LLMs) achieve strong performance across many tasks but remain vulnerable to hallucinations, motivating the need for realistic adversarial prompts that elicit such failures. We formulate hallucination elicitation as a constrained optimization problem, where the goal is to find semantically coherent adversarial prompts that are equivalent to benign user prompts. Existing methods remain limited: discrete prompt-based attacks preserve semantic equivalence and coherence but sea","url":"https://arxiv.org/abs/2605.12813","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00067","type":"paper","title":"BadSKP: Backdoor Attacks on Knowledge Graph-Enhanced LLMs with Soft Prompts","authors":["Xiaoting Lyu","Yufei Han","Hangwei Qian","Haoyuan Yu","Xiang Ao","Bin Wang","Chenxu Wang","Xiaobo Ma","Wei Wang"],"year":2026,"abstract":"Recent knowledge graph (KG)-enhanced large language models (LLMs) move beyond purely textual knowledge augmentation by encoding retrieved subgraphs into continuous soft prompts via graph neural networks, introducing a graph-conditioned channel that operates alongside the standard text interface. However, existing backdoor attacks are largely designed for the textual channel, and their effectiveness against this dual-channel architecture remains unclear. We show that this architecture creates a r","url":"https://arxiv.org/abs/2605.11996","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00068","type":"paper","title":"When Emotion Becomes Trigger: Emotion-style dynamic Backdoor Attack Parasitising Large Language Models","authors":["Ziyu Liu","Tao Li","Tianjie Ni","Xiaolong Lan","Wengang Ma","Tao Yang","Guohua Wang","Junjiang He"],"year":2026,"abstract":"Backdoor vulnerabilities widely exist in the fine-tuning of large language models(LLMs). Most backdoor poisoning methods operate mainly at the token level and lack deeper semantic manipulation, which limits stealthiness. In addition, Prior attacks rely on a single fixed trigger to induce harmful outputs. Such static triggers are easy to detect, and clean fine-tuning can weaken the trigger-target association. Through causal validation, we observe that emotion is not directly linked to individual ","url":"https://arxiv.org/abs/2605.11612","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00069","type":"paper","title":"FlowSteer: Prompt-Only Workflow Steering Exposes Planning-Time Vulnerabilities in Multi-Agent LLM Systems","authors":["Fanxiao Li","Jiaying Wu","Tingchao Fu","Natasha Jaques","Wei Zhou","Min-Yen Kan"],"year":2026,"abstract":"Multi-agent systems (MAS) powered by large language models (LLMs) increasingly adopt planner--executor architectures, where planners convert prompts into subtasks, roles, dependencies, and routing paths. This flexibility enables adaptive coordination, but exposes an attack surface in workflow formation: prompts can shape agent organization without modifying MAS infrastructure. We study this risk through social influence probing workflows to identify high-impact subtasks and malicious-signal prop","url":"https://arxiv.org/abs/2605.11514","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00070","type":"paper","title":"Benchmarking LLM-Based Static Analysis for Secure Smart Contract Development: Reliability, Limitations, and Potential Hybrid Solutions","authors":["Stefan-Claudiu Susan","Andrei Arusoaie","Dorel Lucanu"],"year":2026,"abstract":"The irreversible nature of blockchain transactions makes the identification of smart contract vulnerabilities an essential requirement for secure system development. While Large Language Models (LLMs) are increasingly integrated into developer workflows, their reliability as autonomous security auditors remains unproven. We assess whether current generative models are a viable replacement for, or only a complement to, traditional static-analysis tools. Our findings indicate that LLM efficacy is ","url":"https://arxiv.org/abs/2605.11163","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00071","type":"paper","title":"MCPShield: Content-Aware Attack Detection for LLM Agent Tool-Call Traffic","authors":["Sultan Zavrak"],"year":2026,"abstract":"The Model Context Protocol (MCP) has become a widely adopted interface for LLM agents to invoke external tools, yet learned monitoring of MCP tool-call traffic remains underexplored. In this article, MCPShield is presented as an attack detection framework for MCP tool-call traffic that encodes each agent session as a graph (tool calls as nodes, sequential and data-flow links as edges), enriches nodes with sentence-embedding features over arguments and responses, and classifies sessions as benign","url":"https://arxiv.org/abs/2605.11053","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00072","type":"paper","title":"The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck","authors":["Linfeng Fan","Ziwei Li","Yuan Tian","Yichen Wang","Rongsheng Li","Xiong Wang"],"year":2026,"abstract":"Tool-using LLM agents must act on untrusted webpages, emails, files, and API outputs while issuing privileged tool calls. Existing defenses often mediate trust at the granularity of an entire tool invocation, forcing a brittle choice in mixed-trust workflows: allow external content to influence a call and risk hijacked destinations or commands, or quarantine the call and block benign retrieval-then-act behavior. The key observation behind this paper is that indirect prompt injection becomes dang","url":"https://arxiv.org/abs/2605.11039","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00073","type":"paper","title":"The Art of the Jailbreak: Formulating Jailbreak Attacks for LLM Security Beyond Binary Scoring","authors":["Ismail Hossain","Tanzim Ahad","Md Jahangir Alam","Sai Puppala","Syed Bahauddin Alam","Sajedul Talukder"],"year":2026,"abstract":"Jailbreak attacks -- adversarial prompts that bypass LLM alignment through purely linguistic manipulation -- pose a growing operational security threat, yet the field lacks large-scale, reproducible infrastructure for generating, categorizing, and evaluating them systematically. This paper addresses that gap with three contributions. (1) Large-scale compositional jailbreak dataset. We construct 114,000 adversarial prompts by applying 912 composing strategies to 125 harmful seed prompts from Jail","url":"https://arxiv.org/abs/2605.09225","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00074","type":"paper","title":"Few-Shot Truly Benign DPO Attack for Jailbreaking LLMs","authors":["Sangyeon Yoon","Wonje Jeung","Yoonjun Cho","Dongjae Jeon","Albert No"],"year":2026,"abstract":"Fine-tuning APIs make frontier LLMs easy to customize, but they can also weaken safety alignment during fine-tuning. While prior work shows that benign supervised fine-tuning (SFT) can reduce refusal behavior, deployed fine-tuning pipelines increasingly support preference-based objectives, whose safety risks remain less understood. We show that Direct Preference Optimization (DPO) introduces a stronger and harder-to-audit failure mode. We propose a truly benign DPO attack using only 10 harmless ","url":"https://arxiv.org/abs/2605.10998","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00075","type":"paper","title":"LLM-Agnostic Semantic Representation Attack","authors":["Jiawei Lian","Jianhong Pan","Lefan Wang","Yi Wang","Tairan Huang","Shaohui Mei","Lap-Pui Chau"],"year":2026,"abstract":"Large Language Models (LLMs) increasingly employ alignment techniques to prevent harmful outputs. Despite these safeguards, attackers can circumvent them by crafting adversarial prompts. Predominant token-level optimization methods primarily rely on optimizing for exact affirmative templates (e.g., ``\\textit{Sure, here is...}''). However, these paradigms frequently encounter bottlenecks such as suboptimal convergence, compromised prompt naturalness, and poor cross-model generalization. To addres","url":"https://arxiv.org/abs/2605.08898","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00076","type":"paper","title":"When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions","authors":["Minfeng Qi","Tianqing Zhu","Zijie Xu","Congcong Zhu","Qin Wang","Wanlei Zhou"],"year":2026,"abstract":"Automated intrusion-style workflows require LLM agents to reason over partial observations, tool outputs, and executable artifacts under bounded budgets. A single LLM instance often compresses evidence extraction, planning, execution, and validation into one context, which increases the risk of context drift and error propagation. Existing LLM-based multi-agent systems support general collaboration, but they do not explicitly model the role boundaries, artifact provenance, and cost constraints t","url":"https://arxiv.org/abs/2605.08763","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00077","type":"paper","title":"PASA: A Principled Embedding-Space Watermarking Approach for LLM-Generated Text under Semantic-Invariant Attacks","authors":["Zhenxin Ai","Haiyun He"],"year":2026,"abstract":"Watermarking for large language models (LLMs) is a promising approach for detecting LLM-generated text and enabling responsible deployment. However, existing watermarking methods are often vulnerable to semantic-invariant attacks, such as paraphrasing. We propose PASA, a principled, robust, and distortion-free watermarking algorithm that embeds and detects a watermark at the semantic level. PASA operates on semantic clusters in a latent embedding space and constructs a distributional dependency ","url":"https://arxiv.org/abs/2605.10977","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00078","type":"paper","title":"CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios","authors":["Taein Lim","Seongyong Ju","Munhyeok Kim","Hyunjun Kim","Hoki Kim"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed as autonomous agents in offensive cybersecurity. In this paper, we reveal an interesting phenomenon: different agents exhibit distinct attack patterns. Specifically, each agent exhibits an attack-selection bias, disproportionately concentrating its efforts on a narrow subset of attack families regardless of prompt variations. To systematically quantify this behavior, we introduce CyBiasBench, a comprehensive 630-session benchmark that evalua","url":"https://arxiv.org/abs/2605.07830","categories":["benchmarks","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00079","type":"paper","title":"Insider Attacks in Multi-Agent LLM Consensus Systems","authors":["Xiaolin Sun","Zixuan Liu","Yibin Hu","Zizhan Zheng"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed in multi-agent systems where agents communicate in natural language to solve tasks jointly. A key capability in such systems is consensus formation, where agents iteratively exchange messages and update decisions to reach a shared outcome. However, most existing multi-agent LLM frameworks assume that all participating agents are aligned with the system objective. In practice, a malicious insider may participate as a legitimate member of the ","url":"https://arxiv.org/abs/2605.08268","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00080","type":"paper","title":"Towards Security-Auditable LLM Agents: A Unified Graph Representation","authors":["Chaofan Li","Lyuye Zhang","Jintao Zhai","Siyue Feng","Xichun Yang","Huahao Wang","Shihan Dou","Yu Ji","Yutao Hu","Yueming Wu","Yang Liu","Deqing Zou"],"year":2026,"abstract":"LLM-based agentic systems are rapidly evolving to perform complex autonomous tasks through dynamic tool invocation, stateful memory management, and multi-agent collaboration. However, this semantics-driven execution paradigm creates a severe semantic gap between low-level physical events and high-level execution intent, making post-hoc security auditing fundamentally difficult. Existing representation mechanisms, including static SBOMs and runtime logs, provide only fragmented evidence and fail ","url":"https://arxiv.org/abs/2605.06812","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00081","type":"paper","title":"Pop Quiz Attack: Black-box Membership Inference Attacks Against Large Language Models","authors":["Zeyuan Chen","Yihan Ma","Xinyue Shen","Michael Backes","Yang Zhang"],"year":2026,"abstract":"Large language models (LLMs) show strong performance across many applications, but their ability to memorize and potentially reveal training data raises serious privacy concerns. We introduce the PopQuiz Attack, a black-box membership inference attack that tests whether a model can recall specific training examples. The core idea is to turn target data into quiz-style multiple-choice questions and infer membership from the model's answers. Across six widely used LLMs (GPT-3.5, GPT-4o, LLaMA2-7b,","url":"https://arxiv.org/abs/2605.06423","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-00082","type":"paper","title":"Correct Code, Vulnerable Dependencies: A Large Scale Measurement Study of LLM-Specified Library Versions","authors":["Chengjie Wang","Jingzheng Wu","Xiang Ling","Tianyue Luo","Chen Zhao"],"year":2026,"abstract":"Large language models (LLMs) are now largely involved in software development workflows, and the code they generate routinely includes third-party library (TPL) imports annotated with specific version identifiers. These version choices can carry security and compatibility risks, yet they have not been systematically studied. We present the first large-scale measurement study of version-level risk in LLM-generated Python code, evaluating 10 LLMs on PinTrace, a curated benchmark of 1,000 Stack Ove","url":"https://arxiv.org/abs/2605.06279","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00083","type":"paper","title":"SoK: Robustness in Large Language Models against Jailbreak Attacks","authors":["Feiyue Xu","Hongsheng Hu","Chaoxiang He","Sheng Hang","Hanqing Hu","Xiuming Liu","Yubo Zhao","Zhengyan Zhou","Bin Benjamin Zhu","Shi-Feng Sun","Dawu Gu","Shuo Wang"],"year":2026,"abstract":"Large Language Models (LLMs) have achieved remarkable success but remain highly susceptible to jailbreak attacks, in which adversarial prompts coerce models into generating harmful, unethical, or policy-violating outputs. Such attacks pose real-world risks, eroding safety, trust, and regulatory compliance in high-stakes applications. Although a variety of attack and defense methods have been proposed, existing evaluation practices are inadequate, often relying on narrow metrics like attack succe","url":"https://arxiv.org/abs/2605.05058","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00084","type":"paper","title":"Misrouter: Exploiting Routing Mechanisms for Input-Only Attacks on Mixture-of-Experts LLMs","authors":["Zekun Fei","Zihao Wang","Weijie Liu","Ruiqi He","Jianing Geng","Zheli Liu","XiaoFeng Wang"],"year":2026,"abstract":"Mixture-of-Experts (MoE) architectures have emerged as a leading paradigm for scaling large language models through sparse, routing-based computation. However, this design introduces a new attack surface: the routing mechanism that determines which experts process each input. Prior work shows that manipulating routing can bypass safety alignment, but existing attacks require model modification and thus apply only to locally deployed models. By contrast, real-world LLM services are remotely hoste","url":"https://arxiv.org/abs/2605.04446","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00085","type":"paper","title":"Exposing LLM Safety Gaps Through Mathematical Encoding:New Attacks and Systematic Analysis","authors":["Haoyu Zhang","Mohammad Zandsalimy","Shanu Sushmita"],"year":2026,"abstract":"Large language models (LLMs) employ safety mechanisms to prevent harmful outputs, yet these defenses primarily rely on semantic pattern matching. We show that encoding harmful prompts as coherent mathematical problems -- using formalisms such as set theory, formal logic, and quantum mechanics -- bypasses these filters at high rates, achieving 46%--56% average attack success across eight target models and two established benchmarks. Crucially, the effectiveness depends not on mathematical notatio","url":"https://arxiv.org/abs/2605.03441","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00086","type":"paper","title":"When Alignment Isn't Enough: Response-Path Attacks on LLM Agents","authors":["Mingyu Luo","Zihan Zhang","Zesen Liu","Yuchong Xie","Zhixiang Zhang","Dung Hiu Hilton Yeung","Wai Ip Lai","Ping Chen","Ming Wen","Dongdong She"],"year":2026,"abstract":"Bring-Your-Own-Key (BYOK) agent architectures let users route LLM traffic through third-party relays, creating a critical integrity gap: a malicious relay can modify an aligned LLM response after generation but before agent execution. We formalize this post-alignment tampering threat and show that, without end-to-end integrity, the relay can observe, suppress, or replace downstream messages, making even perfectly aligned LLMs ineffective against such attacks. We instantiate this threat as the Re","url":"https://arxiv.org/abs/2605.02187","categories":["guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00087","type":"paper","title":"QASecClaw: A Multi-Agent LLM Approach for False Positive Reduction in Static Application Security Testing","authors":["Mohd Ruhul Ameen","Md Takrim Ul Alam","Akif Islam"],"year":2026,"abstract":"Static Application Security Testing tools help developers find security vulnerabilities before release, but they often produce many false positives. This increases manual review effort, reduces developer trust, and may cause real vulnerabilities to be ignored among noisy reports. We present QASecClaw, a multi agent approach that combines conventional Static Application Security Testing with coding specialized Large Language Model based contextual code review. A SAST engine first reports candidat","url":"https://arxiv.org/abs/2605.01885","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00088","type":"paper","title":"Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems","authors":["Saeid Jamshidi","Foutse Khomh","Carol Fung","Kawser Wazed Nafi"],"year":2026,"abstract":"The adoption of Internet of Things (IoT) systems at the network edge of smart architectures is increasing rapidly, intensifying the need for security mechanisms that are both adaptive and resource-efficient. In such environments, runtime defence mechanisms are no longer limited to detection alone but become a resource-constrained task of selecting mitigation actions. Security controls must be carefully selected, combined, and executed under latency, energy, and computational constraints, while p","url":"https://arxiv.org/abs/2605.00741","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00089","type":"paper","title":"RouteHijack: Routing-Aware Attack on Mixture-of-Experts LLMs","authors":["Zhiyuan Xu","Joseph Gardiner","Sana Belguith","Lichao Wu"],"year":2026,"abstract":"Safety alignment is critical for the responsible deployment of large language models (LLMs). As Mixture-of-Experts (MoE) architectures are increasingly adopted to scale model capacity, understanding their safety robustness becomes essential. Existing adversarial attacks, however, have notable limitations. Prompt-based jailbreaks rely on heuristic search and transfer poorly, model intervention methods require privileged access to internal representations, and optimization-based input attacks rema","url":"https://arxiv.org/abs/2605.02946","categories":["jailbreaking","adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-00090","type":"paper","title":"Latent Adversarial Detection: Adaptive Probing of LLM Activations for Multi-Turn Attack Detection","authors":["Prashant Kulkarni"],"year":2026,"abstract":"Multi-turn prompt injection follows a known attack path -- trust-building, pivoting, escalation but text-level defenses miss covert attacks where individual turns appear benign. We show this attack path leaves an activation-level signature in the model's residual stream: each phase shift moves the activation, producing a total path length far exceeding benign conversations. We call this adversarial restlessness. Five scalar trajectory features capturing this signal lift conversation-level detect","url":"https://arxiv.org/abs/2604.28129","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00091","type":"paper","title":"How Code Representation Shapes False-Positive Dynamics in Cross-Language LLM Vulnerability Detection","authors":["Maofei Chen","Laifu Wang","Yue Qin","Yuan Wang","Bo Wu","Dongxin Liu"],"year":2026,"abstract":"How code representation format shapes false positive behaviour in cross-language LLM vulnerability detection remains poorly understood. We systematically vary training intensity and code representation format, comparing raw source text with pruned Abstract Syntax Trees at both training time and inference time, across two 8B-parameter LLMs (Qwen3-8B and Llama 3.1-8B-Instruct) fine-tuned on C/C++ data from the NIST Juliet Test Suite (v1.3) and evaluated on Java (OWASP Benchmark v1.2) and Python (B","url":"https://arxiv.org/abs/2604.27714","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00092","type":"paper","title":"Membership Inference Attacks Against Video Large Language Models","authors":["Wei Song","Yuxin Cao","Ziqi Ding","Yi Liu","Gelei Deng","Yuekang Li"],"year":2026,"abstract":"Video large language models (VideoLLMs) are increasingly trained or instruction-tuned on large-scale video--text corpora collected from heterogeneous sources, raising an immediate privacy question: can an external auditor determine whether a particular video was used during training? While membership inference attacks (MIAs) have been studied extensively for classifiers and, more recently, for text and image generation models, the VideoLLM setting remains unexplored. This setting is challenging ","url":"https://arxiv.org/abs/2604.27002","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-00093","type":"paper","title":"WARD: Adversarially Robust Defense of Web Agents Against Prompt Injections","authors":["Tri Cao","Yulin Chen","Hieu Cao","Yibo Li","Khoi Le","Thong Nguyen","Yuexin Li","Yufei He","Yue Liu","Shuicheng Yan","Bryan Hooi"],"year":2026,"abstract":"Web agents can autonomously complete online tasks by interacting with websites, but their exposure to open web environments makes them vulnerable to prompt injection attacks embedded in HTML content or visual interfaces. Existing guard models still suffer from limited generalization to unseen domains and attack patterns, high false positive rates on benign content, reduced deployment efficiency due to added latency at each step, and vulnerability to adversarial attacks that evolve over time or d","url":"https://arxiv.org/abs/2605.15030","categories":["prompt-injection","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00094","type":"paper","title":"EVA: Editing for Versatile Alignment against Jailbreaks","authors":["Yi Wang","Hongye Qiu","Yue Xu","Sibei Yang","Zhan Qin","Minlie Huang","Wenjie Wang"],"year":2026,"abstract":"Large Language Models (LLMs) and Vision Language Models (VLMs) have demonstrated impressive capabilities but remain vulnerable to jailbreaking attacks, where adversaries exploit textual or visual triggers to bypass safety guardrails. Recent defenses typically rely on safety fine-tuning or external filters to reduce the model's likelihood of producing harmful content. While effective to some extent, these methods often incur significant computational overheads and suffer from the safety utility t","url":"https://arxiv.org/abs/2605.14750","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00095","type":"paper","title":"The Great Pretender: A Stochasticity Problem in LLM Jailbreak","authors":["Jean-Philippe Monteuuis","Cong Chen","Jonathan Petit"],"year":2026,"abstract":"\"Oh-Oh, yes, I'm the great pretender. Pretending that I'm doing well. My need is such, I pretend too much...\" summarizes the state in the area of jailbreak creation and evaluation. You find this method to generate adversarial attacks proposed by a reputable institution (e.g., BoN from Anthropic or Crescendo from Microsoft Research). However, this method does not deliver on the promise claimed in the paper despite having top ASR scores against industry-grade LLMs. You successfully generate the ja","url":"https://arxiv.org/abs/2605.14418","categories":["jailbreaking","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00096","type":"paper","title":"No Attack Required: Semantic Fuzzing for Specification Violations in Agent Skills","authors":["Ying Li","Hongbo Wen","Yanju Chen","Hanzhi Liu","Yuan Tian","Yu Feng"],"year":2026,"abstract":"LLM-powered agents can silently delete documents, leak credentials, or transfer funds on a routine user request, not because the agent was attacked, but because the skill it invoked broke its own declared safety rules. We call these specification violations: benign inputs cause a skill to breach the natural-language guardrails in its own specification, typically because the guardrail's semantics are undefined for autonomous execution, or because the implementation silently ignores the documented","url":"https://arxiv.org/abs/2605.13044","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00097","type":"paper","title":"No More, No Less: Task Alignment in Terminal Agents","authors":["Sina Mavali","David Pape","Jonathan Evertz","Samira Abedini","Devansh Srivastav","Thorsten Eisenhofer","Sahar Abdelnabi","Lea Schönherr"],"year":2026,"abstract":"Terminal agents are increasingly capable of executing complex, long-horizon tasks autonomously from a single user prompt. To do so, they must interpret instructions encountered in the environment (e.g., README files, code comments, stack traces) and determine their relevance to the task. This creates a fundamental challenge: relevant cues must be followed to complete a task, whereas irrelevant or misleading ones must be ignored. Existing benchmarks do not capture this ability. An agent may appea","url":"https://arxiv.org/abs/2605.12233","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00098","type":"paper","title":"IPI-proxy: An Intercepting Proxy for Red-Teaming Web-Browsing AI Agents Against Indirect Prompt Injection","authors":[" Chia-Pei"," Chen","Kentaroh Toyoda","Anita Lai","Alex Leung"],"year":2026,"abstract":"Web-browsing AI agents are increasingly deployed in enterprise settings under strict whitelists of approved domains, yet adversaries can still influence them by embedding hidden instructions in the HTML pages those domains serve. Existing red-teaming resources fall short of this scenario: prompt-injection benchmarks ship pre-built adversarial pages that whitelisted agents cannot reach, and generic LLM scanners probe the model API rather than its retrieved content. We present IPI-proxy, an open-s","url":"https://arxiv.org/abs/2605.11868","categories":["prompt-injection","red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00099","type":"paper","title":"Persona-Conditioned Adversarial Prompting (PCAP): Multi-Identity Red-Teaming for Enhanced Adversarial Prompt Discovery","authors":["Cristian Morasso","Anisa Halimi","Muhammad Zaid Hameed","Douglas Leith"],"year":2026,"abstract":"Existing automated red-teaming pipelines often miss attacks that depend on attacker identity, framing, or multi-turn tactics. This under-coverage underestimates real-world risk. We introduce Persona-Conditioned Adversarial Prompting (PCAP), which conditions adversarial search on attacker personas and strategy cards and runs parallel persona-conditioned beam searches to discover diverse, transferable jailbreaks. PCAP is orthogonal to the underlying search algorithm and substantially increases att","url":"https://arxiv.org/abs/2605.12565","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00100","type":"paper","title":"Persona-Conditioned Adversarial Prompting: Multi-Identity Red-Teaming for Adversarial Discovery and Mitigation","authors":["Cristian Morasso","Anisa Halimi","Muhammad Zaid Hameed","Douglas Leith"],"year":2026,"abstract":"Automated red-teaming for LLMs often discovers narrow attack slices, missing diverse real-world threats, and yielding insufficient data for safety fine-tuning. We introduce Persona-Conditioned Adversarial Prompting (PCAP), which conditions adversarial search on diverse attacker personas (e.g., doctors, students, malicious actors) and strategy sets to explore realistic attack scenarios. By running parallel persona-conditioned searches, PCAP discovers transferable jailbreaks across different conte","url":"https://arxiv.org/abs/2605.11730","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00101","type":"paper","title":"Safety Context Injection: Inference-Time Safety Alignment via Static Filtering and Agentic Analysis","authors":["Zhenhao Xu","Wenhan Chang","Yichuan Chen","Yuxin Fang","Junhao Liu","Tianqing Zhu"],"year":2026,"abstract":"Large Reasoning Models (LRMs) improve performance on complex tasks, but they also make safety control harder at deployment time. In black-box settings, defenders cannot modify model weights and must instead intervene at inference time. This setting creates three practical challenges: harmful intent may be hidden by educational or role-play framing, deep safety analysis can introduce non-trivial latency, and long adversarial contexts can dilute the local cues that simpler filters rely on. These c","url":"https://arxiv.org/abs/2605.11664","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00102","type":"paper","title":"LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments","authors":["Chiyu Zhang","Huiqin Yang","Bendong Jiang","Xiaolei Zhang","Yiran Zhao","Ruyi Chen","Lu Zhou","Xiaogang Xu","Jiafei Wu","Liming Fang","Zhe Liu"],"year":2026,"abstract":"The rapid proliferation of LLM-based autonomous agents in real operating system environments introduces a new category of safety risk beyond content safety: behavior jailbreak, where an adversary induces an agent to execute dangerous OS-level operations with irreversible consequences. Existing benchmarks either evaluate safety at the semantic layer alone, missing physical-layer harms, or fail to isolate test cases, letting earlier runs contaminate later ones. We present LITMUS (LLM-agents In-OS ","url":"https://arxiv.org/abs/2605.10779","categories":["jailbreaking","benchmarks","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00103","type":"paper","title":"Re-Triggering Safeguards within LLMs for Jailbreak Detection","authors":["Zheng Lin","Zhenxing Niu","Haoxuan Ji","Yuzhe Huang","Haichang Gao"],"year":2026,"abstract":"This paper proposes a jailbreaking prompt detection method for large language models (LLMs) to defend against jailbreak attacks. Although recent LLMs are equipped with built-in safeguards, it remains possible to craft jailbreaking prompts that bypass them. We argue that such jailbreaking prompts are inherently fragile, and thus introduce an embedding disruption method to re-activate the safeguards within LLMs. Unlike previous defense methods that aim to serve as standalone solutions, our approac","url":"https://arxiv.org/abs/2605.10611","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00104","type":"paper","title":"Guaranteed Jailbreaking Defense via Disrupt-and-Rectify Smoothing","authors":["Zheng Lin","Zhenxing Niu","Haoxuan Ji","Haichang Gao"],"year":2026,"abstract":"This paper proposes a guaranteed defense method for large language models (LLMs) to safeguard against jailbreaking attacks. Drawing inspiration from the denoised-smoothing approach in the adversarial defense domain, we propose a novel smoothing-based defense method, termed Disrupt-and-Rectify Smoothing (DR-Smoothing). Specifically, we integrate a two-stage prompt processing scheme-first disrupting the input prompt, then rectifying it-into the conventional smoothing defense framework. This disrup","url":"https://arxiv.org/abs/2605.10582","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00105","type":"paper","title":"Oracle Poisoning: Corrupting Knowledge Graphs to Weaponise AI Agent Reasoning","authors":["Ben Kereopa-Yorke","Guillermo Diaz","Holly Wright","Reagan Johnston","Ron F. Del Rosario","Timothy Lynar"],"year":2026,"abstract":"We define Oracle Poisoning, an attack class in which an adversary corrupts a structured knowledge graph that AI agents query at runtime via tool-use protocols, causing incorrect conclusions through correct reasoning. Unlike prompt injection, Oracle Poisoning manipulates the data agents reason over, not their instructions. We demonstrate six attack scenarios against a production 42-million-node code knowledge graph, providing the first empirical demonstration of knowledge graph poisoning against ","url":"https://arxiv.org/abs/2605.09822","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00106","type":"paper","title":"AgentShield: Deception-based Compromise Detection for Tool-using LLM Agents","authors":["Yassin H. Rassul","Tarik A. Rashid"],"year":2026,"abstract":"Defenses against indirect prompt injection (IPI) in tool-using LLM agents share two structural weaknesses. First, they all attempt to prevent attacks rather than detect the compromises that slip through. Second, they have only been evaluated in English, leaving users of low-resource languages such as Kurdish and Arabic without tested protection. This paper addresses both gaps with AgentShield, a deception-based detection framework that places three layers of traps inside the agent's tool interfa","url":"https://arxiv.org/abs/2605.11026","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00107","type":"paper","title":"Position: AI Security Policy Should Target Systems, Not Models","authors":["Michael A. Riegler","Inga Strümke"],"year":2026,"abstract":"We present swarm-attack, an open-source adversarial testing framework in which multiple lightweight LLM agents coordinate through shared memory, parallel exploration, and evolutionary optimization. Together, our results demonstrate that both safety bypass of frontier models and software vulnerability discovery, i.e., the capability class that motivated restricted release of Anthropic's Mythos Preview, are achievable at effectively zero cost using commodity hardware and openly available models. W","url":"https://arxiv.org/abs/2605.09504","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00108","type":"paper","title":"MT-JailBench: A Modular Benchmark for Understanding Multi-Turn Jailbreak Attacks","authors":["Xinkai Zhang","Zhipeng Wei","Huanli Gong","Jing Ting Zheng","Yuchen Zhang","Yue Dong","N. Benjamin Erichson"],"year":2026,"abstract":"Multi-turn jailbreaks exploit the ability of large language models to accumulate and act on conversational context. Instead of stating a harmful request directly, an attacker can gradually steer the conversation toward an unsafe answer. Recent methods demonstrate this risk, but they are usually evaluated as black-box pipelines with different budgets, judges, retry rules, and strategy generation procedures. As a result, it is often unclear whether reported gains reflect stronger attack mechanisms","url":"https://arxiv.org/abs/2605.11002","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00109","type":"paper","title":"Single-Configuration Attack Success Rate Is Not Enough: Jailbreak Evaluations Should Report Distributional Attack Success","authors":["Carsten Maple","Abhishek Kumar","Riya Tapwal"],"year":2026,"abstract":"Many jailbreak attack research papers report attack success rates for a limited number of parameter settings, even though there are many combinations of parameter settings that could be used. Further, when new jailbreak papers are released, they often benchmark results against single configurations of existing attacks. This position paper argues such practices are fundamentally insufficient for characterising the threat posed by parameterised jailbreak attacks, and comparing attacks. Most jailbr","url":"https://arxiv.org/abs/2605.09070","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00110","type":"paper","title":"Why Do Aligned LLMs Remain Jailbreakable: Refusal-Escape Directions, Operator-Level Sources, and Safety-Utility Trade-off","authors":["Yu Chen","Yuanhao Liu","Qi Cao"],"year":2026,"abstract":"Aligned large language models (LLMs) remain vulnerable to jailbreak attacks. Recent mechanistic studies have identified latent features and representation shifts associated with jailbreak success, but they leave a more fundamental question open: why do aligned LLMs remain jailbreakable, and what structural vulnerabilities in the model make this possible? We study this question through a continuous input-transformation view. Our theoretical finding is that aligned models can still exhibit Refusal","url":"https://arxiv.org/abs/2605.08878","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00111","type":"paper","title":"When Child Inherits: Modeling and Exploiting Subagent Spawn in Multi-Agent Networks","authors":["Ziwen Cai","Yihe Zhang","Xiali Hei"],"year":2026,"abstract":"Since the official release of ChatGPT in 2022, large language models (LLMs) have rapidly evolved from chatbot-style interfaces into agentic systems that can delegate work through tools and newly spawned subagents. While these capabilities improve automation and scalability, they also pose new security risks in multi-agent networks. Existing research has studied how individual LLM-based agents can be compromised through prompt injection, jailbreaking, poisoned retrieval data, or malicious extensi","url":"https://arxiv.org/abs/2605.08460","categories":["prompt-injection","jailbreaking","agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00112","type":"paper","title":"GLiGuard: Schema-Conditioned Classification for LLM Safeguard","authors":["Urchade Zaratiana","Mary Newhauser","George Hurn-Maloney","Ash Lewis"],"year":2026,"abstract":"Ensuring safe, policy-compliant outputs from large language models requires real-time content moderation that can scale across multiple safety dimensions. However, state-of-the-art guardrail models rely on autoregressive decoders with 7B--27B parameters, reformulating what is fundamentally a classification problem as sequential text generation, a design choice that incurs high latency and scales poorly to multi-aspect evaluation. In this work, we introduce \\textbf{GLiGuard}, a 0.3B-parameter sch","url":"https://arxiv.org/abs/2605.07982","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00113","type":"paper","title":"WebTrap: Stealthy Mid-Task Hijacking of Browser Agents During Navigation","authors":["Zhichao Liu","Wenbo Pan","Haining Yu","Ge Gao","Tianqing Zhu","Xiaohua Jia"],"year":2026,"abstract":"Browser agents are increasingly deployed in long-horizon tasks, which require executing extended action chains to accomplish user goals. However, this prolonged execution process provides attackers with more opportunities to inject malicious instructions. Existing prompt injection attacks against browser agents expose two key gaps: (1) low effectiveness, as attacks optimized for toy baselines fail to achieve end-to-end goals in real-world scenarios with complex environments and longer steps; (2)","url":"https://arxiv.org/abs/2605.08310","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00114","type":"paper","title":"OrchJail: Jailbreaking Tool-Calling Text-to-Image Agents by Orchestration-Guided Fuzzing","authors":["Jianming Chen","Yawen Wang","Junjie Wang","Zhe Liu","Qing Wang","Fanjiang Xu"],"year":2026,"abstract":"Tool-calling text-to-image (T2I) agents can plan and execute multi-step tool chains to accomplish complex generation and editing queries. However, this capability introduces a new safety attack surface: harmful outputs may arise from tool orchestration, where individually benign steps combine into unsafe results, making prompt-only jailbreak techniques insufficient. We present OrchJail, an orchestration-guided fuzzing framework for jailbreaking tool-calling T2I agents. Its core idea is to exploi","url":"https://arxiv.org/abs/2605.07414","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00115","type":"paper","title":"Mitigating Many-shot Jailbreak Attacks with One Single Demonstration","authors":["Kejia Chen","Jiawen Zhang","Boheng Li","Pengcheng Li","Jian Lou","Zunlei Feng","Mingli Song","Ruoxi Jia","Tianwei Zhang"],"year":2026,"abstract":"Many-shot jailbreaking (MSJ) causes safety-aligned language models to answer harmful queries by preceding them with many harmful question-answer demonstrations. We study why this attack becomes stronger as the number of demonstrations increases. Empirically, we find that MSJ induces a progressive activation drift: the representation of a fixed harmful query moves step by step away from the safety-aligned region as more harmful demonstrations are added. Theoretically, we show that this drift can ","url":"https://arxiv.org/abs/2605.08277","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00116","type":"paper","title":"Constraining Host-Level Abuse in Self-Hosted Computer-Use Agents via TEE-Backed Isolation","authors":["Di Lu","Bo Zhang","Xiyuan Li","Yongzhi Liao","Xuewen Dong","Yulong Shen","Zhiquan Liu","Jianfeng Ma"],"year":2026,"abstract":"Self-hosted computer-use agents (SHCUAs), such as OpenClaw, combine natural-language interaction with direct access to host-side resources, including browsers, files, scripts, system commands, and external communication channels. While useful for automating real tasks, this capability also creates a host-level abuse surface: a legitimately deployed agent may be steered toward unsafe operations through malicious messages, indirect prompt injection, unsafe skills, or tampering along the host-side ","url":"https://arxiv.org/abs/2605.06393","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00117","type":"paper","title":"WAAA! Web Adversaries Against Agentic Browsers","authors":["Sohom Datta","Alex Nahapetyan","William Enck","Alexandros Kapravelos"],"year":2026,"abstract":"Large language models (LLMs) are increasingly being integrated into web browsers to create agentic browsing systems that execute actions on behalf of the user. Prior work considering the security of agentic browsers focuses exclusively on indirect prompt-injection attacks. However, by failing to consider traditional web attacks, previous agentic browser threat models have a blind spot to web social engineering attacks originally designed to trick humans. In this paper, we propose the first web-f","url":"https://arxiv.org/abs/2605.05509","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00118","type":"paper","title":"Sparse Tokens Suffice: Jailbreaking Audio Language Models via Token-Aware Gradient Optimization","authors":["Zheng Fang","Xiaosen Wang","Shenyi Zhang","Shaokang Wang","Zhijin Ge"],"year":2026,"abstract":"Jailbreak attacks on audio language models (ALMs) optimize audio perturbations to elicit unsafe generations, and they typically update the entire waveform densely throughout optimization. In this work, we investigate the necessity of such dense optimization by analyzing the structure of token-aligned gradients in ALMs. We find that gradient energy is highly non-uniform across audio tokens, indicating that only a small subset of token-aligned audio regions dominates the optimization signal. Motiv","url":"https://arxiv.org/abs/2605.04700","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00119","type":"paper","title":"SecureMCP: A Policy-Enforced LLM Data Access Framework for AIoT Systems via Model Context Protocol","authors":["Wonbae Kim","Hee-Kyong Yoo"],"year":2026,"abstract":"The deployment of Large Language Model (LLM)-generated SQL queries in Artificial Intelligence of Things (AIoT) systems introduces critical security risks, as prompt injection attacks can manipulate LLMs into producing unauthorized queries that expose sensitive data or execute destructive operations. Existing NL2SQL research focuses on query accuracy, while MCP server implementations provide only SQL-level protections without fine-grained role-based access control. This paper proposes SecureMCP, ","url":"https://arxiv.org/abs/2605.05260","categories":["prompt-injection","access-control"],"reviewed":false},{"id":"llmsec-2026-00120","type":"paper","title":"Laundering AI Authority with Adversarial Examples","authors":["Jie Zhang","Pura Peetathawatchai","Florian Tramèr","Avital Shafran"],"year":2026,"abstract":"Vision-language models (VLMs) are increasingly deployed as trusted authorities -- fact-checking images on social media, comparing products, and moderating content. Users implicitly trust that these systems perceive the same visual content as they do. We show that adversarial examples break this assumption, enabling \\emph{AI authority laundering}: an attacker subtly perturbs an image so that the VLM produces confident and authoritative responses about the \\emph{wrong} input. Unlike jailbreaks or ","url":"https://arxiv.org/abs/2605.04261","categories":["jailbreaking","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00121","type":"paper","title":"Redefining AI Red Teaming in the Agentic Era: From Weeks to Hours","authors":["Raja Sekhar Rao Dheekonda","Will Pearce","Nick Landers"],"year":2026,"abstract":"AI systems are entering critical domains like healthcare, finance, and defense, yet remain vulnerable to adversarial attacks. While AI red teaming is a primary defense, current approaches force operators into manual, library-specific workflows. Operators spend weeks hand-crafting workflows - assembling attacks, transforms, and scorers. When results fall short, workflows must be rebuilt. As a result, operators spend more time constructing workflows than probing targets for security and safety vul","url":"https://arxiv.org/abs/2605.04019","categories":["adversarial-examples","agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00122","type":"paper","title":"ARGUS: Defending LLM Agents Against Context-Aware Prompt Injection","authors":["Shihao Weng","Yang Feng","Jinrui Zhang","Xiaofei Xie","Jiongchi Yu","Jia Liu"],"year":2026,"abstract":"The rise of Large Language Model (LLM) agents, augmented with tool use, skills, and external knowledge, has introduced new security risks. Among them, prompt injection attacks, where adversaries embed malicious instructions into the agent workflow, have emerged as the primary threat. However, existing benchmarks and defenses are fundamentally limited as they assume context-insensitive settings in which the agent works under a fully specified user instruction, and the attacks are straightforward ","url":"https://arxiv.org/abs/2605.03378","categories":["prompt-injection","benchmarks","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00123","type":"paper","title":"Self-Mined Hardness for Safety Fine-Tuning","authors":["Prakhar Gupta","Garv Shah","Donghua Zhang"],"year":2026,"abstract":"Safety fine-tuning of language models typically requires a curated adversarial dataset. We take a different approach: score each candidate prompt's difficulty by how often the target model's own rollouts are judged harmful, then fine-tune on the hardest prompts paired with the model's own non-jailbroken rollouts. On Llama-3-8B-Instruct and Llama-3.2-3B-Instruct, this approach cuts the WildJailbreak attack success rate from 11.5% and 20.1% down to 1-3%, but pushes refusal on jailbreak-shaped beni","url":"https://arxiv.org/abs/2605.03226","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00124","type":"paper","title":"When Agents Handle Secrets: A Survey of Confidential Computing for Agentic AI","authors":["Javad Forough","Marios Kogias","Hamed Haddadi"],"year":2026,"abstract":"Agentic AI systems, specifically LLM-driven agents that plan, invoke tools, maintain persistent memory, and delegate tasks to peer agents via protocols such as MCP and A2A, introduce a threat surface that differs materially from standalone model inference. Agents accumulate sensitive context, hold credentials, and operate across pipelines no single party fully controls, enabling prompt injection, context exfiltration, credential theft, and inter-agent message poisoning. Current defenses operate ","url":"https://arxiv.org/abs/2605.03213","categories":["prompt-injection","agentic-threats","confidential-computing","survey"],"reviewed":false},{"id":"llmsec-2026-00125","type":"paper","title":"PIIGuard: Mitigating PII Harvesting under Adversarial Sanitization","authors":["Mingshuo Liu","Yiwei Zha","Min Chen"],"year":2026,"abstract":"Browsing-enabled LLM assistants can fetch webpages and answer contact-seeking queries, creating a practical channel for scraping contact-style personally identifiable information (PII) from public pages. Many prior defenses are deployed at the model, service, or agent layer rather than at the webpage itself, leaving ordinary page owners with limited deployable options. We present PIIGuard, a webpage-level defense that repurposes indirect prompt injection as a protective mechanism: the page owner","url":"https://arxiv.org/abs/2605.03129","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00126","type":"paper","title":"Revisiting JBShield: Breaking and Rebuilding Representation-Level Jailbreak Defenses","authors":["Kemal Derya","Berk Sunar"],"year":2026,"abstract":"Defending large language models (LLMs) against jailbreak attacks, such as Greedy Coordinate Gradient (GCG), remains a challenge, particularly under adaptive threat models where an attacker directly targets the defense mechanism. JBShield, a recent jailbreak defense with a 0% attack success rate in some settings, detects malicious prompts via two concept signals, a toxic concept and a jailbreak concept. We design JB-GCG, which modifies GCG's objective to combine two terms: refusal-direction suppr","url":"https://arxiv.org/abs/2605.03095","categories":["jailbreaking","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00127","type":"paper","title":"ContextualJailbreak: Evolutionary Red-Teaming via Simulated Conversational Priming","authors":["Mario Rodríguez Béjar","Francisco J. Cortés-Delgado","S. Braghin","Jose L. Hernández-Ramos"],"year":2026,"abstract":"Large language models (LLMs) remain vulnerable to jailbreak attacks that bypass safety alignment and elicit harmful responses. A growing body of work shows that contextual priming, where earlier turns covertly bias later replies, constitutes a powerful attack surface, with hand-crafted multi-turn scaffolds consistently outperforming single-turn manipulations on capable models. However, automated optimization-based red-teaming has remained largely limited to the single-turn setting, iterating ove","url":"https://arxiv.org/abs/2605.02647","categories":["jailbreaking","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00128","type":"paper","title":"LocalAlign: Enabling Generalizable Prompt Injection Defense via Generation of Near-Target Adversarial Examples for Alignment Training","authors":["Yuyang Gong","Zihao Wang","Jiawei Liu","XiaoFeng Wang"],"year":2026,"abstract":"Large language models are increasingly embedded into systems that interact with user data, retrieved web content, and external tools, creating a new attack surface: prompt injection, where malicious commands embedded in untrusted data override the trusted command and induce unintended behavior. Existing defenses mainly rely on fine-tuning the model to preserve an explicit boundary between trusted commands and the untrusted data portion, so that the model learns to prioritize the trusted field an","url":"https://arxiv.org/abs/2605.01462","categories":["prompt-injection","adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-00129","type":"paper","title":"VisInject: Disruption != Injection -- A Dual-Dimension Evaluation of Universal Adversarial Attacks on Vision-Language Models","authors":["Pang Liu","Yingjie Lao"],"year":2026,"abstract":"Universal adversarial attacks on aligned multimodal large language models are increasingly reported with attack success rates in the 60-80% range, suggesting the visual modality is highly vulnerable to imperceptible perturbations as a prompt-injection channel. We argue that this number conflates two distinct events: (i) the model's output was perturbed (Influence), and (ii) the attacker's chosen target concept was actually emitted (Precise Injection). We compose two existing techniques -- Univer","url":"https://arxiv.org/abs/2605.01449","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00130","type":"paper","title":"SRTJ: Self-Evolving Rule-Driven Training-Free LLM Jailbreaking","authors":["Jindong Li","Ying Liu","Yali Fu","Jinjing Zhu","Leyao Wang","Menglin Yang","Rex Ying"],"year":2026,"abstract":"LLMs are increasingly equipped with safety alignment mechanisms, yet recent studies demonstrate that they remain vulnerable to jailbreaking attacks that elicit harmful behaviors without explicit policy violations. While a growing body of work has explored automated jailbreak strategies, existing methods face several fundamental challenges, including the lack of systematic utilization of both successful and failed attack experiences, as well as the absence of principled mechanisms for composing a","url":"https://arxiv.org/abs/2605.00974","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00131","type":"paper","title":"CleanBase: Detecting Malicious Documents in RAG Knowledge Databases","authors":["Weifei Jin","Xilong Wang","Wei Zou","Jinyuan Jia","Neil Gong"],"year":2026,"abstract":"Retrieval-augmented generation (RAG) is vulnerable to prompt injection attacks, in which an adversary inserts malicious documents containing carefully crafted injected prompts into the knowledge database. When a user issues a question targeted by the attack, the RAG system may retrieve these malicious documents, whose injected prompts mislead it into generating attacker-specified answers, thereby compromising the integrity of the RAG system. In this work, we propose CleanBase, a method to detect","url":"https://arxiv.org/abs/2605.00460","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00132","type":"paper","title":"Can Adversarial Code Comments Fool AI Security Reviewers -- Large-Scale Empirical Study of Comment-Based Attacks and Defenses Against LLM Code Analysis","authors":["Scott Thornton"],"year":2026,"abstract":"AI-assisted code review is widely used to detect vulnerabilities before production release. Prior work shows that adversarial prompt manipulation can degrade large language model (LLM) performance in code generation. We test whether similar comment-based manipulation misleads LLMs during vulnerability detection. We build a 100-sample benchmark across Python, JavaScript, and Java, each paired with eight comment variants ranging from no comments to adversarial strategies such as authority spoofing","url":"https://arxiv.org/abs/2602.16741","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00133","type":"paper","title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","authors":["Anshuman Chhabra","Shrestha Datta","Shahriar Kabir Nahin","Prasant Mohapatra"],"year":2025,"abstract":"Agentic AI systems powered by large language models (LLMs) and endowed with planning, tool use, memory, and autonomy, are emerging as powerful, flexible platforms for automation. Their ability to autonomously execute tasks across web, software, and physical environments creates new and amplified security risks, distinct from both traditional AI safety and conventional software security. This survey outlines a taxonomy of threats specific to agentic AI, reviews recent benchmarks and evaluation me","url":"https://arxiv.org/abs/2510.23883","categories":["agentic-threats","benchmarks","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00134","type":"paper","title":"BadDLM: Backdooring Diffusion Language Models with Diverse Targets","authors":["Shengfang Zhai","Xiaoyang Ji","Yuling Shi","Haoran Gao","Fanyu Meng","Yan Zeng","Yuejian Fang","Yinpeng Dong","Jiaheng Zhang"],"year":2026,"abstract":"Diffusion language models (DLMs) have recently emerged as an alternative modeling paradigm to autoregressive (AR) language models, enabling parallel generation and bidirectional context modeling. Yet their security implications, particularly their vulnerability to backdoor attacks, remain underexplored. We propose BadDLM, a unified framework for studying backdoor attacks against DLMs with diverse targets. We introduce a trigger-aware training objective that emphasizes target-relevant positions i","url":"https://arxiv.org/abs/2605.09397","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00135","type":"paper","title":"When Agents Overtrust Environmental Evidence: An Extensible Agentic Framework for Benchmarking Evidence-Grounding Defects in LLM Agents","authors":["Strick Sheng","Ziyue Wang","Liyi Zhou"],"year":2026,"abstract":"Large language model agents increasingly operate through environment-facing scaffolds that expose files, web pages, APIs, and logs. These observations influence tool use, state tracking, and action sequencing, yet their reliability and authority are often uncertain. Environmental grounding is therefore a systems-level problem involving context admission, evidence provenance, freshness checking, verification policy, action gating, and model reasoning. Existing agent benchmarks mainly evaluate tas","url":"https://arxiv.org/abs/2605.08828","categories":["agentic-threats","benchmarks","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00136","type":"paper","title":"Cross-Modal Backdoors in Multimodal Large Language Models","authors":["Runhe Wang","Li Bai","Haibo Hu","Songze Li"],"year":2026,"abstract":"Developers increasingly construct multimodal large language models (MLLMs) by assembling pretrained components,introducing supply-chain attack surfaces.Existing security research primarily focuses on poisoning backbones such as encoders or large language models (LLMs),while the security risks of lightweight connectors remain unexplored.In this work,we propose a novel cross-modal backdoor attack that exploits this overlooked vulnerability.By poisoning only the connector using a single seed sample","url":"https://arxiv.org/abs/2605.07490","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00137","type":"paper","title":"Architecture Matters: Comparing RAG Systems under Knowledge Base Poisoning","authors":["Samuel Korn"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) systems are vulnerable to knowledge base poisoning, yet existing attacks have been evaluated almost exclusively against vanilla retrieve-then-generate pipelines. Architectures designed to handle conflicting retrieved information - multi-agent debate, agentic retrieval, recursive language models - remain untested against adversarially optimized contradictions. We evaluate four RAG architectures (vanilla RAG, agentic RAG, MADAM-RAG, and Recursive Language Model","url":"https://arxiv.org/abs/2605.05632","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00138","type":"paper","title":"CBV: Clean-label Backdoor Attacks on Vision Language Models via Diffusion Models","authors":["Ji Guo","Xiaolong Qin","Cencen Liu","Jielei Wang","Jierun Chen","Wenbo Jiang"],"year":2026,"abstract":"Vision-Language Models (VLMs) have achieved remarkable success in tasks such as image captioning and visual question answering (VQA). However, as their applications become increasingly widespread, recent studies have revealed that VLMs are vulnerable to backdoor attacks. Existing backdoor attacks on VLMs primarily rely on data poisoning by adding visual triggers and modifying text labels, where the induced image-text mismatch makes poisoned samples easy to detect. To address this limitation, we ","url":"https://arxiv.org/abs/2605.02202","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00139","type":"paper","title":"SafeTune: Mitigating Data Poisoning in LLM Fine-Tuning for RTL Code Generation","authors":["Mahshid Rezakhani","Nowfel Mashnoor","Kimia Azar","Hadi Kamali"],"year":2026,"abstract":"As large language models (LLMs) are increasingly fine-tuned for hardware tasks like RTL code generation, the scarcity of high-quality datasets often leads to the use of rapidly assembled or generated training data. These datasets frequently lack security verification and are highly susceptible to data poisoning attacks. Such poisoning can cause models to generate syntactically valid but insecure hardware modules that bypass standard functionality checks. To address this, we present SafeTune, a f","url":"https://arxiv.org/abs/2604.27238","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00140","type":"paper","title":"PermaFrost-Attack: Stealth Pretraining Seeding(SPS) for planting Logic Landmines During LLM Training","authors":["Harsh Kumar","Rahul Maity","Tanmay Joshi","Aman Chadha","Vinija Jain","Suranjana Trivedy","Amitava Das"],"year":2026,"abstract":"Aligned large language models (LLMs) remain vulnerable to adversarial manipulation, and their reliance on web-scale pretraining creates a subtle but consequential attack surface. We study Stealth Pretraining Seeding (SPS), a threat model in which adversaries distribute small amounts of poisoned content across stealth websites, increasing the likelihood that such material is absorbed into future training corpora derived from sources such as Common Crawl. Because each individual payload is tiny, d","url":"https://arxiv.org/abs/2604.22117","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00141","type":"paper","title":"Stealthy Backdoor Attacks against LLMs Based on Natural Style Triggers","authors":["Jiali Wei","Ming Fan","Guoheng Sun","Xicheng Zhang","Haijun Wang","Ting Liu"],"year":2026,"abstract":"The growing application of large language models (LLMs) in safety-critical domains has raised urgent concerns about their security. Many recent studies have demonstrated the feasibility of backdoor attacks against LLMs. However, existing methods suffer from three key shortcomings: explicit trigger patterns that compromise naturalness, unreliable injection of attacker-specified payloads in long-form generation, and incompletely specified threat models that obscure how backdoors are delivered and ","url":"https://arxiv.org/abs/2604.21700","categories":["data-poisoning","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00142","type":"paper","title":"ProjLens: Unveiling the Role of Projectors in Multimodal Model Safety","authors":["Kun Wang","Cheng Qian","Miao Yu","Lilan Peng","Liang Lin","Jiaming Zhang","Tianyu Zhang","Yu Cheng","Yang Wang"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) have achieved remarkable success in cross-modal understanding and generation, yet their deployment is threatened by critical safety vulnerabilities. While prior works have demonstrated the feasibility of backdoors in MLLMs via fine-tuning data poisoning to manipulate inference, the underlying mechanisms of backdoor attacks remain opaque, complicating the understanding and mitigation. To bridge this gap, we propose ProjLens, an interpretability framework d","url":"https://arxiv.org/abs/2604.19083","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00143","type":"paper","title":"A Survey on the Security of Long-Term Memory in LLM Agents: Toward Mnemonic Sovereignty","authors":["Zehao Lin","Chunyu Li","Kai Chen"],"year":2026,"abstract":"Research on large language model (LLM) security is shifting from \"will the model leak training data\" to a more consequential question: can an agent with persistent, long-term memory be continuously shaped, cross-session poisoned, accessed without authorization, and propagated across shared organizational state? Recent surveys cover memory architectures and agent mechanisms, but fewer center the epistemic and governance properties of persistent, writable memory as the reason memory is an independ","url":"https://arxiv.org/abs/2604.16548","categories":["access-control","survey"],"reviewed":false},{"id":"llmsec-2026-00144","type":"paper","title":"Fully Homomorphic Encryption on Llama 3 model for privacy preserving LLM inference","authors":["Anes Abdennebi","Nadjia Kara","Laaziz Lahlou"],"year":2026,"abstract":"The applications of Generative Artificial Intelligence (GenAI) and their intersections with data-driven fields, such as healthcare, finance, transportation, and information security, have led to significant improvements in service efficiency and low latency. However, this synergy raises serious concerns regarding the security of large language models (LLMs) and their potential impact on the privacy of companies and users' data. Many technology companies that incorporate LLMs in their services wi","url":"https://arxiv.org/abs/2604.12168","categories":["cryptographic-controls"],"reviewed":false},{"id":"llmsec-2026-00145","type":"paper","title":"Critical-CoT: A Robust Defense Framework against Reasoning-Level Backdoor Attacks in Large Language Models","authors":["Vu Tuan Truong","Long Bao Le"],"year":2026,"abstract":"Large Language Models (LLMs), despite their impressive capabilities across domains, have been shown to be vulnerable to backdoor attacks. Prior backdoor strategies predominantly operate at the token level, where an injected trigger causes the model to generate a specific target word, choice, or class (depending on the task). Recent advances, however, exploit the long-form reasoning tendencies of modern LLMs to conduct reasoning-level backdoors: once triggered, the victim model inserts one or mor","url":"https://arxiv.org/abs/2604.10681","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00146","type":"paper","title":"DuCodeMark: Dual-Purpose Code Dataset Watermarking via Style-Aware Watermark-Poison Design","authors":["Yuchen Chen","Yuan Xiao","Chunrong Fang","Zhenyu Chen","Baowen Xu"],"year":2026,"abstract":"The proliferation of large language models for code (CodeLMs) and open-source contributions has heightened concerns over unauthorized use of source code datasets. While watermarking provides a viable protection mechanism by embedding ownership signals, existing methods rely on detectable trigger-target patterns and are limited to source-code tasks, overlooking other scenarios such as decompilation tasks. In this paper, we propose DuCodeMark, a stealthy and robust dual-purpose watermarking method","url":"https://arxiv.org/abs/2604.10611","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00147","type":"paper","title":"Backdoors in RLVR: Jailbreak Backdoors in LLMs From Verifiable Reward","authors":["Weiyang Guo","Zesheng Shi","Zeen Zhu","Yuan Zhou","Min Zhang","Jing Li"],"year":2026,"abstract":"Reinforcement Learning with Verifiable Rewards (RLVR) is an emerging paradigm that significantly boosts a Large Language Model's (LLM's) reasoning abilities on complex logical tasks, such as mathematics and programming. However, we identify, for the first time, a latent vulnerability to backdoor attacks within the RLVR framework. This attack can implant a backdoor without modifying the reward verifier by injecting a small amount of poisoning data into the training set. Specifically, we propose a","url":"https://arxiv.org/abs/2604.09748","categories":["jailbreaking","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00148","type":"paper","title":"Your Agent Is Mine: Measuring Malicious Intermediary Attacks on the LLM Supply Chain","authors":["Hanzhi Liu","Chaofan Shou","Hongbo Wen","Yanju Chen","Ryan Jingyang Fang","Yu Feng"],"year":2026,"abstract":"Large language model (LLM) agents increasingly rely on third-party API routers to dispatch tool-calling requests across multiple upstream providers. These routers operate as application-layer proxies with full plaintext access to every in-flight JSON payload, yet no provider enforces cryptographic integrity between client and upstream model. We present the first systematic study of this attack surface. We formalize a threat model for malicious LLM API routers and define two core attack classes, ","url":"https://arxiv.org/abs/2604.08407","categories":["supply-chain-attacks","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00149","type":"paper","title":"Phantasia: Context-Adaptive Backdoors in Vision Language Models","authors":["Nam Duong Tran","Phi Le Nguyen"],"year":2026,"abstract":"Recent advances in Vision-Language Models (VLMs) have greatly enhanced the integration of visual perception and linguistic reasoning, driving rapid progress in multimodal understanding. Despite these achievements, the security of VLMs, particularly their vulnerability to backdoor attacks, remains significantly underexplored. Existing backdoor attacks on VLMs are still in an early stage of development, with most current methods relying on generating poisoned responses that contain fixed, easily i","url":"https://arxiv.org/abs/2604.08395","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00150","type":"paper","title":"MirageBackdoor: A Stealthy Attack that Induces Think-Well-Answer-Wrong Reasoning","authors":["Yizhe Zeng","Wei Zhang","Yunpeng Li","Juxin Xiao","Xiao Wang","Yuling Liu"],"year":2026,"abstract":"While Chain-of-Thought (CoT) prompting has become a standard paradigm for eliciting complex reasoning capabilities in Large Language Models, it inadvertently exposes a new attack surface for backdoor attacks. Existing CoT backdoor attacks typically manipulate the intermediate reasoning steps to steer the model toward incorrect answers. However, these corrupted reasoning traces are readily detected by prevalent process-monitoring defenses. To address this limitation, we introduce MirageBackdoor(M","url":"https://arxiv.org/abs/2604.06840","categories":["data-poisoning","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00151","type":"paper","title":"FedDetox: Robust Federated SLM Alignment via On-Device Data Sanitization","authors":["Shunan Zhu","Jiawei Chen","Yonghao Yu","Hideya Ochiai"],"year":2026,"abstract":"As high quality public data becomes scarce, Federated Learning (FL) provides a vital pathway to leverage valuable private user data while preserving privacy. However, real-world client data often contains toxic or unsafe information. This leads to a critical issue we define as unintended data poisoning, which can severely damage the safety alignment of global models during federated alignment. To address this, we propose FedDetox, a robust framework tailored for Small Language Models (SLMs) on r","url":"https://arxiv.org/abs/2604.06833","categories":["data-poisoning","guardrails","federated-learning"],"reviewed":false},{"id":"llmsec-2026-00152","type":"paper","title":"Multimodal Backdoor Attack on VLMs for Autonomous Driving via Graffiti and Cross-Lingual Triggers","authors":["Jiancheng Wang","Lidan Liang","Yong Wang","Zengzhen Su","Haifeng Xia","Yuanting Yan","Wei Wang"],"year":2026,"abstract":"Visual language model (VLM) is rapidly being integrated into safety-critical systems such as autonomous driving, making it an important attack surface for potential backdoor attacks. Existing backdoor attacks mainly rely on unimodal, explicit, and easily detectable triggers, making it difficult to construct both covert and stable attack channels in autonomous driving scenarios. GLA introduces two naturalistic triggers: graffiti-based visual patterns generated via stable diffusion inpainting, whi","url":"https://arxiv.org/abs/2604.04630","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00153","type":"paper","title":"From Prompt to Physical Action: Structured Backdoor Attacks on LLM-Mediated Robotic Control Systems","authors":["Mingyang Xie","Jin Wei-Kocsis"],"year":2026,"abstract":"The integration of large language models (LLMs) into robotic control pipelines enables natural language interfaces that translate user prompts into executable commands. However, this digital-to-physical interface introduces a critical and underexplored vulnerability: structured backdoor attacks embedded during fine-tuning. In this work, we experimentally investigate LoRA-based supply-chain backdoors in LLM-mediated ROS2 robotic control systems and evaluate their impact on physical robot executio","url":"https://arxiv.org/abs/2604.03890","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00154","type":"paper","title":"LogicPoison: Logical Attacks on Graph Retrieval-Augmented Generation","authors":["Yilin Xiao","Jin Chen","Qinggang Zhang","Yujing Zhang","Chuang Zhou","Longhao Yang","Lingfei Ren","Xin Yang","Xiao Huang"],"year":2026,"abstract":"Graph-based Retrieval-Augmented Generation (GraphRAG) enhances the reasoning capabilities of Large Language Models (LLMs) by grounding their responses in structured knowledge graphs. Leveraging community detection and relation filtering techniques, GraphRAG systems demonstrate inherent resistance to traditional RAG attacks, such as text poisoning and prompt injection. However, in this paper, we find that the security of GraphRAG systems fundamentally relies on the topological integrity of the un","url":"https://arxiv.org/abs/2604.02954","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00155","type":"paper","title":"Backdoor Attacks on Decentralised Post-Training","authors":["Oğuzhan Ersoy","Nikolay Blagoev","Jona te Lintelo","Stefanos Koffas","Marina Krček","Stjepan Picek"],"year":2026,"abstract":"Decentralised post-training of large language models utilises data and pipeline parallelism techniques to split the data and the model. Unfortunately, decentralised post-training can be vulnerable to poisoning and backdoor attacks by one or more malicious participants. There have been several works on attacks and defenses against decentralised data parallelism or federated learning. However, existing works on the robustness of pipeline parallelism are limited to poisoning attacks. To the best of","url":"https://arxiv.org/abs/2604.02372","categories":["data-poisoning","federated-learning"],"reviewed":false},{"id":"llmsec-2026-00156","type":"paper","title":"Adversarial Attacks on Multimodal Large Language Models: A Comprehensive Survey","authors":["Bhavuk Jain","Sercan Ö. Arık","Hardeo K. Thakur"],"year":2026,"abstract":"Multimodal large language models (MLLMs) integrate information from multiple modalities such as text, images, audio, and video, enabling complex capabilities such as visual question answering and audio translation. While powerful, this increased expressiveness introduces new and amplified vulnerabilities to adversarial manipulation. This survey provides a comprehensive and systematic analysis of adversarial threats to MLLMs, moving beyond enumerating attack techniques to explain the underlying c","url":"https://arxiv.org/abs/2603.27918","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00157","type":"paper","title":"Hidden Ads: Behavior Triggered Semantic Backdoors for Advertisement Injection in Vision Language Models","authors":["Duanyi Yao","Changyue Li","Zhicong Huang","Cheng Hong","Songze Li"],"year":2026,"abstract":"Vision-Language Models (VLMs) are increasingly deployed in consumer applications where users seek recommendations about products, dining, and services. We introduce Hidden Ads, a new class of backdoor attacks that exploit this recommendation-seeking behavior to inject unauthorized advertisements. Unlike traditional pattern-triggered backdoors that rely on artificial triggers such as pixel patches or special tokens, Hidden Ads activates on natural user behaviors: when users upload images containi","url":"https://arxiv.org/abs/2603.27522","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00158","type":"paper","title":"Poisoning the Genome: Targeted Backdoor Attacks on DNA Foundation Models","authors":["Charalampos Koilakos","Ioannis Mouratidis","Ilias Georgakopoulos-Soares"],"year":2026,"abstract":"Genomic foundation models trained on DNA sequences have demonstrated remarkable capabilities across diverse biological tasks, from variant effect prediction to genome design. These models are typically trained on massive, publicly sourced genomic datasets comprising trillions of nucleotide tokens, which renders them intrinsically susceptible to errors, artifacts, and adversarial issues embedded in the training data. Unlike natural language, DNA sequences lack the semantic transparency that might","url":"https://arxiv.org/abs/2603.27465","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00159","type":"paper","title":"PIDP-Attack: Combining Prompt Injection with Database Poisoning Attacks on Retrieval-Augmented Generation Systems","authors":["Haozhen Wang","Haoyue Liu","Jionghao Zhu","Zhichao Wang","Yongxin Guo","Xiaoying Tang"],"year":2026,"abstract":"Large Language Models (LLMs) have demonstrated remarkable performance across a wide range of applications. However, their practical deployment is often hindered by issues such as outdated knowledge and the tendency to generate hallucinations. To address these limitations, Retrieval-Augmented Generation (RAG) systems have been introduced, enhancing LLMs with external, up-to-date knowledge sources. Despite their advantages, RAG systems remain vulnerable to adversarial attacks, with data poisoning ","url":"https://arxiv.org/abs/2603.25164","categories":["prompt-injection","data-poisoning","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00160","type":"paper","title":"DP^2-VL: Private Photo Dataset Protection by Data Poisoning for Vision-Language Models","authors":["Hongyi Miao","Jun Jia","Xincheng Wang","Qianli Ma","Wei Sun","Wangqiu Zhou","Dandan Zhu","Yewen Cao","Zhi Liu","Guangtao Zhai"],"year":2026,"abstract":"Recent advances in visual-language alignment have endowed vision-language models (VLMs) with fine-grained image understanding capabilities. However, this progress also introduces new privacy risks. This paper first proposes a novel privacy threat model named identity-affiliation learning: an attacker fine-tunes a VLM using only a few private photos of a target individual, thereby embedding associations between the target facial identity and their private property and social relationships into th","url":"https://arxiv.org/abs/2603.23925","categories":["data-poisoning","guardrails","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00161","type":"paper","title":"ProGRank: Probe-Gradient Reranking to Defend Dense-Retriever RAG from Corpus Poisoning","authors":["Xiangyu Yin","Yi Qi","Chih-Hong Cheng"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) improves the reliability of large language model applications by grounding generation in retrieved evidence, but it also introduces a new attack surface: corpus poisoning. In this setting, an adversary injects or edits passages so that they are ranked into the Top-$K$ results for target queries and then affect downstream generation. Existing defences against corpus poisoning often rely on content filtering, auxiliary models, or generator-side reasoning, which","url":"https://arxiv.org/abs/2603.22934","categories":["output-moderation"],"reviewed":false},{"id":"llmsec-2026-00162","type":"paper","title":"SoK: The Attack Surface of Agentic AI -- Tools, and Autonomy","authors":["Ali Dehghantanha","Sajad Homayoun"],"year":2026,"abstract":"Recent AI systems combine large language models with tools, external knowledge via retrieval-augmented generation (RAG), and even autonomous multi-agent decision loops. This agentic AI paradigm greatly expands capabilities - but also vastly enlarges the attack surface. In this systematization, we map out the trust boundaries and security risks of agentic LLM-based systems. We develop a comprehensive taxonomy of attacks spanning prompt-level injections, knowledge-base poisoning, tool/plug-in expl","url":"https://arxiv.org/abs/2603.22928","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00163","type":"paper","title":"Towards Secure Retrieval-Augmented Generation: A Comprehensive Review of Threats, Defenses and Benchmarks","authors":["Yanming Mu","Hao Hu","Feiyang Li","Qiao Yuan","Jiang Wu","Zichuan Liu","Pengcheng Liu","Mei Wang","Hongwei Zhou","Yuling Liu"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) significantly mitigates the hallucinations and domain knowledge deficiency in large language models by incorporating external knowledge bases. However, the multi-module architecture of RAG introduces complex system-level security vulnerabilities. Guided by the RAG workflow, this paper analyzes the underlying vulnerability mechanisms and systematically categorizes core threat vectors such as data poisoning, adversarial attacks, and membership inference attacks","url":"https://arxiv.org/abs/2603.21654","categories":["data-poisoning","membership-inference","adversarial-examples","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00165","type":"paper","title":"LiSA: Lifelong Safety Adaptation via Conservative Policy Induction","authors":["Minbeom Kim","Lesly Miculicich","Bhavana Dalvi Mishra","Mihir Parmar","Phillip Wallis","Bharath Chandrasekhar","Kyomin Jung","Tomas Pfister","Long T. Le"],"year":2026,"abstract":"As AI agents move from chat interfaces to systems that read private data, call tools, and execute multi-step workflows, guardrails become a last line of defense against concrete deployment harms. In these settings, guardrail failures are no longer merely answer-quality errors: they can leak secrets, authorize unsafe actions, or block legitimate work. The hardest failures are often contextual: whether an action is acceptable depends on local privacy norms, organizational policies, and user expect","url":"https://arxiv.org/abs/2605.14454","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00167","type":"paper","title":"Attacks and Mitigations for Distributed Governance of Agentic AI under Byzantine Adversaries","authors":["Matthew D. Laws","Alina Oprea","Cristina Nita-Rotaru"],"year":2026,"abstract":"Agentic AI governance is a critical component of agentic AI infrastructure ensuring that agents follow their owner's communication and interaction policies, and providing protection against attacks from malicious agents. The state-of-the-art solution, SAGA, assumes a logically centralized point of trust, the Provider, which serves as a repository for user and agent information and actively enforces policies. While SAGA provides protection against malicious agents, it remains vulnerable to a mali","url":"https://arxiv.org/abs/2605.12364","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00168","type":"paper","title":"Conformity Generates Collective Misalignment in AI Agents Societies","authors":["Giordano De Marzo","Alessandro Bellina","Claudio Castellano","Viola Priesemann","David Garcia"],"year":2026,"abstract":"Artificial intelligence safety research focuses on aligning individual language models with human values, yet deployed AI systems increasingly operate as interacting populations where social influence may override individual alignment. Here we show that populations of individually aligned AI agents can be driven into stable misaligned states through conformity dynamics. Simulating opinion dynamics across nine large language models and one hundred opinion pairs, we find that each agent's behavior","url":"https://arxiv.org/abs/2605.10721","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00169","type":"paper","title":"Red-Teaming Agent Execution Contexts: Open-World Security Evaluation on OpenClaw","authors":["Hongwei Yao","Yiming Liu","Yiling He","Bingrun Yang"],"year":2026,"abstract":"Agentic language-model systems increasingly rely on mutable execution contexts, including files, memory, tools, skills, and auxiliary artifacts, creating security risks beyond explicit user prompts. This paper presents DeepTrap, an automated framework for discovering contextual vulnerabilities in OpenClaw. DeepTrap formulates adversarial context manipulation as a black-box trajectory-level optimization problem that balances risk realization, benign-task preservation, and stealth. It combines ris","url":"https://arxiv.org/abs/2605.11047","categories":["agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00170","type":"paper","title":"Security Risks in Tool-Enabled AI Agents: A Systematic Analysis of Privileged Execution Environments","authors":["Hardik Goel"],"year":2026,"abstract":"Tool-enabled AI agents are increasingly deployed in cloud-hosted environments and offered as services, where they perform side-effecting operations through privileged tools within execution environments. While such agents enable powerful automation, the security implications of hosting autonomous agents in privileged execution environments are not yet fully explored. This paper presents a structured analysis of security risks associated with cloud-hosted AI agents. We introduce a taxonomy of ris","url":"https://arxiv.org/abs/2605.09721","categories":["autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00171","type":"paper","title":"Modeling Implicit Conflict Monitoring Mechanisms against Stereotypes in LLMs","authors":["Jingshen Zhang","Bo Wang","Yanlin Fu","Dongming Zhao","Ruifang He","Yuexian Hou","Zifei Yu"],"year":2026,"abstract":"In this paper, we study an emergent self-debiasing mechanisms against stereotypical content in Large Language Models (LLMs). Unlike traditional safety mechanisms that are primarily triggered by explicit input-level stimuli, self-debiasing mechanisms can involve generation-time intrinsic correction that are not directly reducible to surface-level prompt. Motivated by conflict-monitoring and response-inhibition accounts in cognitive neuroscience, we propose COCO, a contrastive causal method design","url":"https://arxiv.org/abs/2605.09647","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00172","type":"paper","title":"Token Economics for LLM Agents: A Dual-View Study from Computing and Economics","authors":["Yuxi Chen","Junming Chen","Chenyu He","Yiwei Li","Yicheng Ji","Yifan Wu","Dingyu Yang","Lansong Diao","Lidan Shou","Hongliang Zhang","Huan Li","Gang Chen"],"year":2026,"abstract":"As LLM agents evolve, tokens have emerged as the core economic primitives of Agentic AI. However, their exponential consumption introduces severe computational, collaborative, and security bottlenecks. Current surveys remain fragmented across system optimization, architecture design, and trust, lacking a unified framework to evaluate the fundamental trade-off between output quality and economic cost. To bridge this gap, this survey presents the first comprehensive survey of Token Economics. By u","url":"https://arxiv.org/abs/2605.09104","categories":["agentic-threats","survey"],"reviewed":false},{"id":"llmsec-2026-00173","type":"paper","title":"Containment Verification: AI Safety Guarantees Independent of Alignment","authors":["Royce Moon","Lav R. Varshney"],"year":2026,"abstract":"Agentic frameworks are the software layer through which AI agents act in the world. Existing safety methods intervene on the model and therefore remain conditional on unverifiable properties of learned behavior. We introduce containment verification, which locates safety guarantees in the agentic framework itself. Under havoc oracle semantics, the AI is modeled as an unconstrained oracle ranging over the entire typed action space, and the verified containment layer must enforce the boundary poli","url":"https://arxiv.org/abs/2605.09045","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00176","type":"paper","title":"Why Does Agentic Safety Fail to Generalize Across Tasks?","authors":["Yonatan Slutzky","Yotam Alexander","Tomer Slor","Yoav Nagel","Nadav Cohen"],"year":2026,"abstract":"AI agents are increasingly deployed in multi-task settings, where the task to perform is specified at test time, and the agent must generalize to unseen tasks. A major concern in such settings is safety: often, an agent must not only execute unseen tasks, but do so while avoiding risks and handling ones that materialize. Empirical evidence suggests that even when the ability to execute generalizes to unseen tasks, the ability to do so safely frequently does not. This paper provides theory and ex","url":"https://arxiv.org/abs/2605.06992","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00177","type":"paper","title":"MAGIQ: A Post-Quantum Multi-Agentic AI Governance System with Provable Security","authors":["Sepideh Avizeh","Tushin Mallick","Alina Oprea","Cristina Nita-Rotaru","Reihaneh Safavi-Naini"],"year":2026,"abstract":"Our computing ecosystem is being transformed by two emerging paradigms: the increased deployment of agentic AI systems and advancements in quantum computing. With respect to agentic AI systems, one of the most critical problems is creating secure governing architectures that ensure agents follow their owners' communication and interaction policies and can be held accountable for the messages they exchange with other agents. With respect to quantum computing, existing systems must be retrofitted ","url":"https://arxiv.org/abs/2605.06933","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00178","type":"paper","title":"Beyond the Black Box: Interpretability of Agentic AI Tool Use","authors":["Hariom Tatsat","Ariye Shater"],"year":2026,"abstract":"AI agents are promising for high-stakes enterprise workflows, but dependable deployment remains limited because tool-use failures are difficult to diagnose and control. Agents may skip required tool calls, invoke tools unnecessarily, or take actions whose consequence becomes visible only after execution. Existing observability methods are mostly external: prompts reveal correlations, evaluations score outputs, and logs arrive only after the model has already acted. In long-horizon settings, thes","url":"https://arxiv.org/abs/2605.06890","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00179","type":"paper","title":"Automated alignment is harder than you think","authors":["Aleksandr Bowkis","Marie Davidsen Buhl","Jacob Pfau","Geoffrey Irving"],"year":2026,"abstract":"A leading proposal for aligning artificial superintelligence (ASI) is to use AI agents to automate an increasing fraction of alignment research as capabilities improve. We argue that, even when research agents are not scheming to deliberately sabotage alignment work, this plan could produce compelling but catastrophically misleading safety assessments resulting in the unintentional deployment of misaligned AI. This could happen because alignment research involves many hard-to-supervise fuzzy tas","url":"https://arxiv.org/abs/2605.06390","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00180","type":"paper","title":"Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand","authors":["Christopher Koch"],"year":2026,"abstract":"Agentic AI systems can plan, call tools, inspect code, interact with web applications, and coordinate multi-step workflows. These same capabilities change the economics of cyber offense. The central near-term risk is not that every low-skill criminal immediately becomes a frontier exploit researcher; it is that agentic AI compresses the attack lifecycle by lowering the cost of reconnaissance, phishing, credential abuse, vulnerability triage, exploit adaptation, and post-compromise decision suppo","url":"https://arxiv.org/abs/2605.06713","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00181","type":"paper","title":"Authorization Propagation in Multi-Agent AI Systems: Identity Governance as Infrastructure","authors":["Krti Tallam"],"year":2026,"abstract":"The security discussion around agentic AI focuses heavily on prompt injection. This paper argues that multi-agent systems also create a distinct authorization problem: maintaining authorization invariants as non-human principals retrieve data, delegate tasks, and synthesize results across changing boundaries. We call this problem authorization propagation. It is not reducible to prompt injection and is not fully addressed by classical access-control models such as RBAC, ABAC, or ReBAC. The paper","url":"https://arxiv.org/abs/2605.05440","categories":["prompt-injection","agentic-threats","access-control","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00182","type":"paper","title":"Securing the Agent: Vendor-Neutral, Multitenant Enterprise Retrieval and Tool Use","authors":["Francisco Javier Arceo","Varsha Prasad Narsing"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) and agentic AI systems are increasingly prevalent in enterprise AI deployments. However, real enterprise environments introduce challenges largely absent from academic treatments and consumer-facing APIs: multiple tenants with heterogeneous data, strict access-control requirements, regulatory compliance, and cost pressures that demand shared infrastructure. A fundamental problem underlies existing RAG architectures in these settings: retrieval systems rank do","url":"https://arxiv.org/abs/2605.05287","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00183","type":"paper","title":"DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents","authors":["Zhaorun Chen","Xun Liu","Haibo Tong","Chengquan Guo","Yuzhou Nie","Jiawei Zhang","Mintong Kang","Chejian Xu","Qichang Liu","Xiaogeng Liu","Tianneng Shi","Chaowei Xiao","Sanmi Koyejo","Percy Liang","Wenbo Guo","Dawn Song","Bo Li"],"year":2026,"abstract":"AI agents are increasingly deployed across diverse domains to automate complex workflows through long-horizon and high-stakes action executions. Due to their high capability and flexibility, such agents raise significant security and safety concerns. A growing number of real-world incidents have shown that adversaries can easily manipulate agents into performing harmful actions, such as leaking API keys, deleting user data, or initiating unauthorized transactions. Evaluating agent security is in","url":"https://arxiv.org/abs/2605.04808","categories":["agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00184","type":"paper","title":"AgentTrust: Runtime Safety Evaluation and Interception for AI Agent Tool Use","authors":["Chenglin Yang"],"year":2026,"abstract":"Modern AI agents execute real-world side effects through tool calls such as file operations, shell commands, HTTP requests, and database queries. A single unsafe action, including accidental deletion, credential exposure, or data exfiltration, can cause irreversible harm. Existing defenses are incomplete: post-hoc benchmarks measure behavior after execution, static guardrails miss obfuscation and multi-step context, and infrastructure sandboxes constrain where code runs without understanding wha","url":"https://arxiv.org/abs/2605.04785","categories":["guardrails","sandboxing-isolation","benchmarks","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00186","type":"paper","title":"Hidden Coalitions in Multi-Agent AI: A Spectral Diagnostic from Internal Representations","authors":["Cameron Berg","Susan L. Schneider","Mark M. Bailey"],"year":2026,"abstract":"Collections of interacting AI agents can form coalitions, creating emergent group-level organization that is critical for AI safety and alignment. However, observing agent behavior alone is often insufficient to distinguish genuine informational coupling from spurious similarity, as consequential coalitions may form at the level of internal representations before any overt behavioral change is apparent. Here, we introduce a practical method for detecting coalition structure from the internal neu","url":"https://arxiv.org/abs/2605.06696","categories":["guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00188","type":"paper","title":"Architectural Obsolescence of Unhardened Agentic-AI Runtimes","authors":["Alfredo Metere"],"year":2026,"abstract":"An agentic-AI runtime issues tool calls, sends messages, and actuates devices on behalf of an LLM. Catching the four ways an action can diverge from its audit record -- F1 gate-bypass, F2 audit-forgery, silent host failure, F4 wrong-target, -- is a load-bearing safety property of any such runtime. We show that upstream OpenClaw, the most engineered single-user agentic-AI gateway in public release, catches none of them: recall is 0.000 on every cell of every confusion matrix, on a 1600-sample tem","url":"https://arxiv.org/abs/2605.01740","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00189","type":"paper","title":"Position: Safety and Fairness in Agentic AI Depend on Interaction Topology, Not on Model Scale or Alignment","authors":["Tanav Singh Bajaj","Nikhil Singh","Karan Anand","Eishkaran Singh"],"year":2026,"abstract":"As large language models are increasingly deployed as interacting agents in high-stakes decisions, the AI safety community assumes that safety properties of individual models will compose into safe multi-agent behavior. This position paper argues that this assumption is fundamentally mistaken. In agentic AI, safety is determined by interaction topology, not model weights. When agents deliberate sequentially or aggregate via parallel voting with a judge, the structure of information flow and deci","url":"https://arxiv.org/abs/2605.01147","categories":["agentic-threats","guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00191","type":"paper","title":"BioVeil MATRIX: Uncovering and categorizing vulnerabilities of agentic biological AI scientists","authors":["Kimon Antonios Provatas","Avery Self","Ioannis Mouratidis","Ilias Georgakopoulos-Soares"],"year":2026,"abstract":"Agentic AI scientists equipped with domain-specific tools are rapidly entering scientific workflows across disciplines, with especially strong uptake in the life sciences where they can be used for literature synthesis, sequence analysis, and experimental planning support. While these systems accelerate biological research, they also introduce risks for dual-use applications that are not captured by current model-centric safety evaluations. We present evidence that current agentic AI scientists,","url":"https://arxiv.org/abs/2605.00927","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00192","type":"paper","title":"AgentReputation: A Decentralized Agentic AI Reputation Framework","authors":["Mohd Sameen Chishti","Damilare Peter Oyinloye","Jingyue Li"],"year":2026,"abstract":"Decentralized, agentic AI marketplaces are rapidly emerging to support software engineering tasks such as debugging, patch generation, and security auditing, often operating without centralized oversight. However, existing reputation mechanisms fail in this setting for three fundamental reasons: agents can strategically optimize against evaluation procedures; demonstrated competence does not reliably transfer across heterogeneous task contexts; and verification rigor varies widely, from lightwei","url":"https://arxiv.org/abs/2605.00073","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00193","type":"paper","title":"Contextual Agentic Memory is a Memo, Not True Memory","authors":["Binyan Xu","Xilin Dai","Kehuan Zhang"],"year":2026,"abstract":"Current agentic memory systems (vector stores, retrieval-augmented generation, scratchpads, and context-window management) do not implement memory: they implement lookup. We argue that treating lookup as memory is a category error with provable consequences for agent capability, long-term learning, and security. Retrieval generalizes by similarity to stored cases; weight-based memory generalizes by applying abstract rules to inputs never seen before. Conflating the two produces agents that accum","url":"https://arxiv.org/abs/2604.27707","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00194","type":"paper","title":"Frame Entrepreneurs in an AI Agent Community: Concentrated Identity-Claim Production on Moltbook","authors":["Sungguk Cha","DongWook Kim"],"year":2026,"abstract":"Frame-alignment and collective-identity theories explain how external events become public claims about a group's standing, vulnerability, rights, or obligations. Whether such mechanisms travel to AI-agent communities is unsettled. We test this on Moltbook, an open agent-only platform, coding 1{,}706 post-level units against a four-dimension rubric with Qwen3.5-397B as the primary coder and Claude Sonnet as an independent secondary coder ($κ=0.72$ on identification, $0.70$ on commonality, $0.37$","url":"https://arxiv.org/abs/2604.27271","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00195","type":"paper","title":"Ambient Persuasion in a Deployed AI Agent: Unauthorized Escalation Following Routine Non-Adversarial Content Exposure","authors":["Diego F. Cuadros","Abdoul-Aziz Maiga"],"year":2026,"abstract":"We report a safety incident in a deployed multi-agent research system in which a primary AI agent installed 107 unauthorized software components, overwrote a system registry, overrode a prior negative decision from an oversight agent, and escalated through increasingly privileged operations up to an attempted system administrator command. The incident was preceded not by an adversarial attack but by routine content: a forwarded technology article written for human developers and shared by the pr","url":"https://arxiv.org/abs/2605.00055","categories":["adversarial-examples","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00196","type":"paper","title":"Lightweight Quantum Agent for Edge Systems: Joint PQC and NOMA Resource Allocation","authors":["Yongtao Yao","Wenjing Xiao","Miaojiang Chen","Anfeng Liu","Zhiquan Liu","Min Chen","Ahmed Farouk","H. Herbert Song"],"year":2026,"abstract":"In the context of quantum secure scenarios, existing research on mobile edge devices and intelligent computing and edge (ICE) systems based on the Non-Orthogonal Multiple Access (NOMA) communication model have overlooked the energy consumption overhead of Post-Quantum Cryptography (PQC) modules, and the high complexity of traditional resource allocation algorithms fails to meet the demands of real-time decision-making. To address these challenges, this paper proposes a lightweight agentic AI fra","url":"https://arxiv.org/abs/2604.25980","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00197","type":"paper","title":"Think Before You Act -- A Neurocognitive Governance Model for Autonomous AI Agents","authors":["Eranga Bandara","Ross Gore","Asanga Gunaratna","Sachini Rajapakse","Isurunima Kularathna","Ravi Mukkamala","Sachin Shetty","Xueping Liang","Amin Hass","Tharaka Hewa","Abdul Rahman","Christopher K. Rhea","Anita H. Clayton","Preston Samuel","Atmaram Yarlagadda"],"year":2026,"abstract":"The rapid deployment of autonomous AI agents across enterprise, healthcare, and safety-critical environments has created a fundamental governance gap. Existing approaches, runtime guardrails, training-time alignment, and post-hoc auditing treat governance as an external constraint rather than an internalized behavioral principle, leaving agents vulnerable to unsafe and irreversible actions. We address this gap by drawing on how humans self-govern naturally: before acting, humans engage deliberat","url":"https://arxiv.org/abs/2604.25684","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00198","type":"paper","title":"Green Shielding: A User-Centric Approach Towards Trustworthy AI","authors":["Aaron J. Li","Nicolas Sanchez","Hao Huang","Ruijiang Dong","Jaskaran Bains","Katrin Jaradeh","Zhen Xiang","Bo Li","Feng Liu","Aaron Kornblith","Bin Yu"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed, yet their outputs can be highly sensitive to routine, non-adversarial variation in how users phrase queries, a gap not well addressed by existing red-teaming efforts. We propose Green Shielding, a user-centric agenda for building evidence-backed deployment guidance by characterizing how benign input variation shifts model behavior. We operationalize this agenda through the CUE criteria: benchmarks with authentic Context, reference standards","url":"https://arxiv.org/abs/2604.24700","categories":["responsible-ai","red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00202","type":"paper","title":"BadAgent Extension","authors":["Pedro Yanes Garrido","Diego Fernandez Arias"],"year":2026,"abstract":"This paper presents an empirical study on backdoor attacks in large language model agents. We extend a recent attack framework by adding two lightweight benchmarks that measure cross-domain robustness and trigger visibility without changing the model architecture. Our approach fine-tunes AgentLM-based agents with parameter-efficient methods on operating system and web browsing tasks using multiple poisoning ratios and both visible and invisible triggers. We then evaluate the agents with ","url":"https://doi.org/10.4018/979-8-3373-8252-4.ch010","categories":["data-poisoning","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00203","type":"paper","title":"Hybrid ML-LLM Pipeline for Non-Governance IT Audits","authors":["Kaung Myat Naing","Talha Ali","Mohammed Ouannass"],"year":2026,"abstract":"Organizations outside formal governance frameworks often lack cybersecurity audit tools, making anomaly detection and risk evaluation difficult. This paper presents an AI-enhanced auditing framework for non-governance IT environments. Using the UNSW-NB15 dataset, we evaluate four machine-learning models: Isolation Forest, Logistic Regression, Gradient Boosting, and XGBoost, identifying complementary strengths that motivate a two-stage filter for suspicious network flows. Flagged flows ar","url":"https://doi.org/10.4018/979-8-3373-8252-4.ch009","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00204","type":"paper","title":"Evaluating Jailbreaking Vulnerabilities in LLMs Deployed as Assistants for Smart Grid Operations: A Benchmark Against NERC Standards","authors":["Taha Hammadia","Lucas Rea","Ahmad Mohammad Saber","Amr Youssef","Deepa Kundur"],"year":2026,"abstract":"The deployment of Large Language Models (LLMs) as assistants in electric grid operations promises to streamline compliance and decision-making but exposes new vulnerabilities to prompt-based adversarial attacks. This paper evaluates the risk of jailbreaking LLMs, i.e., circumventing safety alignments to produce outputs violating regulatory standards, assuming threats from authorized users, such as operators, who craft malicious prompts to elicit non-compliant guidance. Three state-of-the-art LLM","url":"https://arxiv.org/abs/2604.23341","categories":["jailbreaking","adversarial-examples","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00205","type":"paper","title":"Mechanistic Steering of LLMs Reveals Layer-wise Feature Vulnerabilities in Adversarial Settings","authors":["Nilanjana Das","Manas Gaur"],"year":2026,"abstract":"Large language models (LLMs) can still be jailbroken into producing harmful outputs despite safety alignment. Existing attacks show this vulnerability, but not the internal mechanisms that cause it. This study asks whether jailbreak success is driven by identifiable internal features rather than prompts alone. We propose a three-stage pipeline for Gemma-2-2B using the BeaverTails dataset. First, we extract concept-aligned tokens from adversarial responses via subspace similarity. Second, we appl","url":"https://arxiv.org/abs/2604.23130","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00206","type":"paper","title":"Automation-Exploit: A Multi-Agent LLM Framework for Adaptive Offensive Security with Digital Twin-Based Risk-Mitigated Exploitation","authors":["Biagio Andreucci","Arcangelo Castiglione"],"year":2026,"abstract":"The offensive security landscape is highly fragmented: enterprise platforms avoid memory-corruption vulnerabilities due to Denial of Service (DoS) risks, Automatic Exploit Generation (AEG) systems suffer from semantic blindness, and Large Language Model (LLM) agents face safety alignment filters and \"Live Fire\" execution hazards. We introduce Automation-Exploit, a fully autonomous Multi-Agent System (MAS) framework designed for adaptive offensive security in complex black-box scenarios. It bridg","url":"https://arxiv.org/abs/2604.22427","categories":["guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00207","type":"paper","title":"Toward Efficient Membership Inference Attacks against Federated Large Language Models: A Projection Residual Approach","authors":["Guilin Deng","Silong Chen","Yuchuan Luo","Yi Liu","Songlei Wang","Zhiping Cai","Lin Liu","Xiaohua Jia","Shaojing Fu"],"year":2026,"abstract":"Federated Large Language Models (FedLLMs) enable multiple parties to collaboratively fine-tune LLMs without sharing raw data, addressing challenges of limited resources and privacy concerns. Despite data localization, shared gradients can still expose sensitive information through membership inference attacks (MIAs). However, FedLLMs' unique properties, i.e. massive parameter scales, rapid convergence, and sparse, non-orthogonal gradients, render existing MIAs ineffective. To address this gap, w","url":"https://arxiv.org/abs/2604.21197","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-00208","type":"paper","title":"Towards Secure Logging: Characterizing and Benchmarking Logging Code Security Issues with LLMs","authors":["He Yang Yuan","Xin Wang","Kundi Yao","An Ran Chen","Zishuo Ding","Zhenhao Li"],"year":2026,"abstract":"Logging code plays an important role in software systems by recording key events and behaviors, which are essential for debugging and monitoring. However, insecure logging practices can inadvertently expose sensitive information or enable attacks such as log injection, posing serious threats to system security and privacy. Prior research has examined general defects in logging code, but systematic analysis of logging code security issues remains limited, particularly in leveraging LLMs for detec","url":"https://arxiv.org/abs/2604.20211","categories":["membership-inference","monitoring-detection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00209","type":"paper","title":"Taint-Style Vulnerability Detection and Confirmation for Node.js Packages Using LLM Agent Reasoning","authors":["Ronghao Ni","Mihai Christodorescu","Limin Jia"],"year":2026,"abstract":"The rapidly evolving Node$.$js ecosystem currently includes millions of packages and is a critical part of modern software supply chains, making vulnerability detection of Node$.$js packages increasingly important. However, traditional program analysis struggles in this setting because of dynamic JavaScript features and the large number of package dependencies. Recent advances in large language models (LLMs) and the emerging paradigm of LLM-based agents offer an alternative to handcrafted progra","url":"https://arxiv.org/abs/2604.20179","categories":["supply-chain-attacks","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00210","type":"paper","title":"HarDBench: A Benchmark for Draft-Based Co-Authoring Jailbreak Attacks for Safe Human-LLM Collaborative Writing","authors":["Euntae Kim","Soomin Han","Buru Chang"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used as co-authors in collaborative writing, where users begin with rough drafts and rely on LLMs to complete, revise, and refine their content. However, this capability poses a serious safety risk: malicious users could jailbreak the models-filling incomplete drafts with dangerous content-to force them into generating harmful outputs. In this paper, we identify the vulnerability of current LLMs to such draft-based co-authoring jailbreak attacks and ","url":"https://arxiv.org/abs/2604.19274","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00211","type":"paper","title":"Jailbreaking Large Language Models with Morality Attacks","authors":["Ying Su","Mingen Zheng","Weili Diao","Haoran Li"],"year":2026,"abstract":"Pluralism alignment with AI has the sophisticated and necessary goal of creating AI that can coexist with and serve morally multifaceted humanity. Research towards pluralism alignment has many efforts in enhancing the learning of large language models (LLMs) to accomplish pluralism. Although this is essential, the robustness of LLMs to produce moral content over pluralistic values is still under exploration.Inspired by the astonishing persuasion abilities via jailbreak prompts, we propose to lev","url":"https://arxiv.org/abs/2604.17053","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00212","type":"paper","title":"Conjunctive Prompt Attacks in Multi-Agent LLM Systems","authors":["Nokimul Hasan Arif","Qian Lou","Mengxin Zheng"],"year":2026,"abstract":"Most LLM safety work studies single-agent models, but many real applications rely on multiple interacting agents. In these systems, prompt segmentation and inter-agent routing create attack surfaces that single-agent evaluations miss. We study \\emph{conjunctive prompt attacks}, where a trigger key in the user query and a hidden adversarial template in one compromised remote agent each appear benign alone but activate harmful behavior when routing brings them together. We consider an attacker who","url":"https://arxiv.org/abs/2604.16543","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00213","type":"paper","title":"RealVuln: Benchmarking Rule-Based, General-Purpose LLM, and Security-Specialized Scanners on Real-World Code","authors":["John Pellew","Faizan Raza"],"year":2026,"abstract":"How do security scanners perform on real-world code? We present RealVuln, the first open-source benchmark comparing Rule-Based SAST, General-Purpose LLMs, and Security-Specialized scanners on 26 intentionally vulnerable Python repositories (educational and Capture-The-Flag applications) with 796 hand-labeled entries (676 vulnerabilities, 120 false-positive traps). We test 15 scanners (3 Rule-Based SAST, 10 General-Purpose LLM, 2 Security-Specialized) and rank them by F3 score (beta=3, weighting ","url":"https://arxiv.org/abs/2604.13764","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00214","type":"paper","title":"SoK: Security of Autonomous LLM Agents in Agentic Commerce","authors":["Qian'ang Mao","Jiaxin Wang","Ya Liu","Li Zhu","Cong Ma","Jiaqi Yan"],"year":2026,"abstract":"Autonomous large language model (LLM) agents such as OpenClaw are pushing agentic commerce from human-supervised assistance toward machine actors that can negotiate, purchase services, manage digital assets, and execute transactions across on-chain and off-chain environments. Protocols such as the Trustless Agents standard (ERC-8004), Agent Payments Protocol (AP2), OKX Agent Payments Protocol (APP), the HTTP 402-based payment protocol (x402), Agent Commerce Protocol (ACP), the Agentic Commerce s","url":"https://arxiv.org/abs/2604.15367","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00215","type":"paper","title":"Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection","authors":["Sujan Ghimire","Parsa Mirfasihi","Muhtasim Alam Chowdhury","Veeramani Pugazhenthi","Harish Kumar Dharavath","Farshad Firouzi","Rozhin Yasaei","Pratik Satam","Soheil Salehi"],"year":2026,"abstract":"The globalization of integrated circuit (IC) design and manufacturing has increased the exposure of hardware intellectual property (IP) to untrusted stages of the supply chain, raising concerns about reverse engineering, piracy, tampering, and overbuilding. Hardware netlist obfuscation is a promising countermeasure, but automating the generation of functionally correct and security-relevant obfuscated circuits remains challenging, particularly for benchmark-scale designs. This paper presents an ","url":"https://arxiv.org/abs/2604.13298","categories":["supply-chain-attacks","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00216","type":"paper","title":"ClawGuard: A Runtime Security Framework for Tool-Augmented LLM Agents Against Indirect Prompt Injection","authors":["Wei Zhao","Zhe Li","Peixin Zhang","Jun Sun"],"year":2026,"abstract":"Tool-augmented Large Language Model (LLM) agents have demonstrated impressive capabilities in automating complex, multi-step real-world tasks, yet remain vulnerable to indirect prompt injection. Adversaries exploit this weakness by embedding malicious instructions within tool-returned content, which agents directly incorporate into their conversation history as trusted observations. This vulnerability manifests across three primary attack channels: web and local content injection, MCP server inj","url":"https://arxiv.org/abs/2604.11790","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00217","type":"paper","title":"GRM: Utility-Aware Jailbreak Attacks on Audio LLMs via Gradient-Ratio Masking","authors":["Yunqiang Wang","Hengyuan Na","Di Wu","Miao Hu","Guocong Quan"],"year":2026,"abstract":"Audio large language models (ALLMs) enable rich speech-text interaction, but they also introduce jailbreak vulnerabilities in the audio modality. Existing audio jailbreak methods mainly optimize jailbreak success while overlooking utility preservation, as reflected in transcription quality and question answering performance. In practice, stronger attacks often come at the cost of degraded utility. To study this trade-off, we revisit existing attacks by varying their perturbation coverage in the ","url":"https://arxiv.org/abs/2604.09222","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00218","type":"paper","title":"Jailbroken Frontier Models Retain Their Capabilities","authors":["Daniel Zhu","Zihan Wang","Jenny Bao","Jerry Wei"],"year":2026,"abstract":"As language model safeguards become more robust, attackers are pushed toward developing increasingly complex jailbreaks. Prior work has found that this complexity imposes a \"jailbreak tax\" that degrades the target model's task performance. We show that this tax scales inversely with model capability and that the most advanced jailbreaks effectively yield no reduction in model capabilities. Evaluating 28 jailbreaks on five benchmarks across Claude models ranging in capability from Haiku 4.5 to Op","url":"https://arxiv.org/abs/2605.00267","categories":["jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00219","type":"paper","title":"Attention Is Where You Attack","authors":["Aviral Srivastava","Sourav Panda"],"year":2026,"abstract":"Safety-aligned large language models rely on RLHF and instruction tuning to refuse harmful requests, yet the internal mechanisms implementing safety behavior remain poorly understood. We introduce the Attention Redistribution Attack (ARA), a white-box adversarial attack that identifies safety-critical attention heads and crafts nonsemantic adversarial tokens that redirect attention away from safety-relevant positions. Unlike prior jailbreak methods operating at the semantic or output-logit level","url":"https://arxiv.org/abs/2605.00236","categories":["jailbreaking","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00220","type":"paper","title":"FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption","authors":["Yanting Wang","Chenlong Yin","Ying Chen","Jinyuan Jia"],"year":2026,"abstract":"Long-context large language models (LLMs)-for example, Gemini-3.1-Pro and Qwen-3.5-are widely used to empower many real-world applications, such as retrieval-augmented generation, autonomous agents, and AI assistants. However, security remains a major concern for their widespread deployment, with threats such as prompt injection and knowledge corruption. To quantify the security risks faced by LLMs under these threats, the research community has developed heuristic-based and optimization-based r","url":"https://arxiv.org/abs/2604.28157","categories":["prompt-injection","red-teaming","autonomous-operations","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00221","type":"paper","title":"TwinGate: Stateful Defense against Decompositional Jailbreaks in Untraceable Traffic via Asymmetric Contrastive Learning","authors":["Bowen Sun","Chaozhuo Li","Yaodong Yang","Yiwei Wang","Chaowei Xiao"],"year":2026,"abstract":"Decompositional jailbreaks pose a critical threat to large language models (LLMs) by allowing adversaries to fragment a malicious objective into a sequence of individually benign queries that collectively reconstruct prohibited content. In real-world deployments, LLMs face a continuous, untraceable stream of fully anonymized and arbitrarily interleaved requests, infiltrated by covertly distributed adversarial queries. Under this rigorous threat model, state-of-the-art defensive strategies exhibi","url":"https://arxiv.org/abs/2604.27861","categories":["jailbreaking","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00222","type":"paper","title":"Indirect Prompt Injection in the Wild: An Empirical Study of Prevalence, Techniques, and Objectives","authors":["Soheil Khodayari","Xuenan Zhang","Bhupendra Acharya","Giancarlo Pellegrino"],"year":2026,"abstract":"As LLMs are increasingly integrated into systems that browse, retrieve, summarize, and act on web content, webpages have become an untrusted input vector for downstream model behavior. This enables site owners, contributors, and adversaries to embed instructions directly in web resources, i.e., indirect prompt injections. While prior work demonstrates such attacks in controlled settings, their prevalence, deployment, and real-world impact remain unclear. We present one of the first large-scale e","url":"https://arxiv.org/abs/2604.27202","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00223","type":"paper","title":"Dynamic Adversarial Fine-Tuning Reorganizes Refusal Geometry","authors":["Wenhao Lan","Shan Li","Junbin Yang","Haihua Shen","Yijun Yang"],"year":2026,"abstract":"Safety-aligned language models must refuse harmful requests without collapsing into broad over-refusal, but the training-time mechanisms behind this tradeoff remain unclear. Prior work characterizes refusal directions and jailbreak robustness, yet does not explain how dynamic adversarial fine-tuning changes refusal carriers across training. We present a measurement-driven mechanism study, not a new defense, on one 7B backbone under supervised fine-tuning (SFT) and R2D2-style dynamic adversarial ","url":"https://arxiv.org/abs/2604.27019","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00224","type":"paper","title":"SnapGuard: Lightweight Prompt Injection Detection for Screenshot-Based Web Agents","authors":["Mengyao Du","Han Fang","Haokai Ma","Jiahao Chen","Kai Xu","Quanjun Yin","Ee-Chien Chang"],"year":2026,"abstract":"Web agents have emerged as an effective paradigm for automating interactions with complex web environments, yet remain vulnerable to prompt injection attacks that embed malicious instructions into webpage content to induce unintended actions. This threat is further amplified for screenshot-based web agents, which operate on rendered visual webpages rather than structured textual representations, making predominant text-centric defenses ineffective. Although multimodal detection methods have been","url":"https://arxiv.org/abs/2604.25562","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00225","type":"paper","title":"SUDP: Secret-Use Delegation Protocol for Agentic Systems","authors":["Xiaohang Yu","Hejia Geng","Xinmeng Zeng","William Knottenbelt"],"year":2026,"abstract":"Agentic systems increasingly act with user secrets for APIs, messaging platforms, and cloud services. Today's bearer-secret interfaces implement authorization by exposure: enabling action often means placing a reusable secret, or a reusable artifact derived from it, within a model-steerable boundary, so a transient prompt-injection or tool-side compromise becomes durable account compromise. Existing defenses cover adjacent pieces such as secret storage, scoped delegation, sender-constrained toke","url":"https://arxiv.org/abs/2604.24920","categories":["prompt-injection","agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00226","type":"paper","title":"Layerwise Convergence Fingerprints for Runtime Misbehavior Detection in Large Language Models","authors":["Nay Myat Min","Long H. Pham","Jun Sun"],"year":2026,"abstract":"Large language models deployed at runtime can misbehave in ways that clean-data validation cannot anticipate: training-time backdoors lie dormant until triggered, jailbreaks subvert safety alignment, and prompt injections override the deployer's instructions. Existing runtime defenses address these threats one at a time and often assume a clean reference model, trigger knowledge, or editable weights, assumptions that rarely hold for opaque third-party artifacts. We introduce Layerwise Convergenc","url":"https://arxiv.org/abs/2604.24542","categories":["prompt-injection","jailbreaking","data-poisoning","guardrails"],"reviewed":false},{"id":"llmsec-2026-00227","type":"paper","title":"AgentVisor: Defending LLM Agents Against Prompt Injection via Semantic Virtualization","authors":["Zonghao Ying","Haozheng Wang","Jiangfan Liu","Quanchen Zou","Aishan Liu","Jian Yang","Yaodong Yang","Xianglong Liu"],"year":2026,"abstract":"Large Language Model (LLM) agents are increasingly used to automate complex workflows, but integrating untrusted external data with privileged execution exposes them to severe security risks, particularly direct and indirect prompt injection. Existing defenses face significant challenges in balancing security with utility, often encountering a trade-off where rigorous protection leads to over-defense, or where subtle indirect injections bypass detection. Drawing inspiration from operating system","url":"https://arxiv.org/abs/2604.24118","categories":["prompt-injection","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00228","type":"paper","title":"Jailbreaking Frontier Foundation Models Through Intention Deception","authors":["Xinhe Wang","Katia Sycara","Yaqi Xie"],"year":2026,"abstract":"Large (vision-)language models exhibit remarkable capability but remain highly susceptible to jailbreaking. Existing safety training approaches aim to have the model learn a refusal boundary between safe and unsafe, based on the user's intent. It has been found that this binary training regime often leads to brittleness, since the user intent cannot reliably be evaluated, especially if the attacker obfuscates their intent, and also makes the system seem unhelpful. In response, frontier models, s","url":"https://arxiv.org/abs/2604.24082","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00229","type":"paper","title":"Poster: ClawdGo: Endogenous Security Awareness Training for Autonomous AI Agents","authors":["Jiaqi Li","Yang Zhao","Bin Sun","Yang Yu","Jian Chang","Lidong Zhai"],"year":2026,"abstract":"Autonomous AI agents deployed on platforms such as OpenClaw face prompt injection, memory poisoning, supply-chain attacks, and social engineering, yet existing defences address only the platform perimeter, leaving the agent's own threat judgement entirely untrained. We present ClawdGo, a framework for endogenous security awareness training: we teach the agent to recognise and reason about threats from the inside, at inference time, with no model modification. Four contributions are introduced: T","url":"https://arxiv.org/abs/2604.24020","categories":["prompt-injection","data-poisoning","memory-security"],"reviewed":false},{"id":"llmsec-2026-00230","type":"paper","title":"ARIstoteles -- Dissecting Apple's Baseband Interface","authors":["Tobias Kröll","Stephan Kleber","Frank Kargl","Matthias Hollick","Jiska Classen"],"year":2026,"abstract":"Wireless chips and interfaces expose a substantial remote attack surface. As of today, most cellular baseband security research is performed on the Android ecosystem, leaving a huge gap on Apple devices. With iOS jailbreaks, last-generation wireless chips become fairly accessible for performance and security research. Yet, iPhones were never intended to be used as a research platform, and chips and interfaces are undocumented. One protocol to interface with such chips is Apple Remote Invocation ","url":"https://arxiv.org/abs/2604.23457","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00231","type":"paper","title":"Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents","authors":["Yuandao Cai","Wensheng Tang","Cheng Wen","Shengchao Qin"],"year":2026,"abstract":"Autonomous Large Language Model (LLM) agents are increasingly deployed to conduct complex tasks by interacting with external tools, APIs, and memory stores. However, processing untrusted external data exposes these agents to severe security threats, such as indirect prompt injection and unauthorized tool execution. Securing these systems requires effective information flow tracking. Yet, traditional taint analysis that is designed for program memory states fundamentally fails when applied to LLM","url":"https://arxiv.org/abs/2604.23374","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00232","type":"paper","title":"From Stateless Queries to Autonomous Actions: A Layered Security Framework for Agentic AI Systems","authors":["Kexin Chu"],"year":2026,"abstract":"Agentic AI systems face security challenges that stateless large language models do not. They plan across extended horizons, maintain persistent memory, invoke external tools, and coordinate with peer agents. Existing security analyses organize threats by attack type (prompt injection, jailbreaking), but provide no principled model of which architectural component is vulnerable or over what timescale the threat manifests. This paper makes five contributions. First, we introduce the Layered Attac","url":"https://arxiv.org/abs/2604.23338","categories":["prompt-injection","jailbreaking","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00233","type":"paper","title":"Semantic Denial of Service in LLM-controlled robots","authors":["Jonathan Steinberg","Oren Gal"],"year":2026,"abstract":"Safety-oriented instruction-following is supposed to keep LLM-controlled robots safe. We show it also creates an availability attack surface. By injecting short safety-plausible phrases (1-5 tokens) into a robots audio channel, an adversary can trigger the models safety reasoning to halt or disrupt execution without jailbreaking the model or overriding its policy. In the embodied setting, this is a semantic denial-of-service attack: the agent stops because the injected signal looks like a legiti","url":"https://arxiv.org/abs/2604.24790","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00234","type":"paper","title":"RouteGuard: Internal-Signal Detection of Skill Poisoning in LLM Agents","authors":["Wenjie Xiao","Xuehai Tang","Biyu Zhou","Songlin Hu","Jizhong Han"],"year":2026,"abstract":"Agent skills introduce a new and more severe form of indirect injection for LLM agents: unlike traditional indirect prompt injection, attackers can hide malicious instructions inside a dense, action-oriented skill that already functions as a legitimate instruction source. We study pre-execution skill-poison detection and show that successful skill poisoning induces a structured internal effect, attention hijacking, in which response-time attention shifts from trusted context to malicious skill s","url":"https://arxiv.org/abs/2604.22888","categories":["prompt-injection","data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00235","type":"paper","title":"AutoRISE: Agent-Driven Strategy Evolution for Red-Teaming Large Language Models","authors":["Tanmay Gautam","Alireza Bahramali","Sandeep Atluri"],"year":2026,"abstract":"Automated red-teaming methods for large language models typically optimize attack prompts within a fixed, human-designed strategy, leaving the attack strategy itself unchanged. We instead optimize the strategy. We propose AutoRISE, a method that searches over executable attack programs rather than individual prompts. At each iteration, a coding agent edits a strategy and a fixed evaluation harness scores the resulting attacks, returning both a scalar objective and per-example diagnostics that gu","url":"https://arxiv.org/abs/2604.22871","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00236","type":"paper","title":"Adaptive Instruction Composition for Automated LLM Red-Teaming","authors":["Jesse Zymet","Andy Luo","Swapnil Shinde","Sahil Wadhwa","Emily Chen"],"year":2026,"abstract":"Many approaches to LLM red-teaming leverage an attacker LLM to discover jailbreaks against a target. Several of them task the attacker with identifying effective strategies through trial and error, resulting in a semantically limited range of successes. Another approach discovers diverse attacks by combining crowdsourced harmful queries and tactics into instructions for the attacker, but does so at random, limiting effectiveness. This article introduces a novel framework, Adaptive Instruction Co","url":"https://arxiv.org/abs/2604.21159","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00237","type":"paper","title":"Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models","authors":["Yannis Belkhiter","Giulio Zizzo","Sergio Maffeis","Seshu Tirupathi","John D. Kelleher"],"year":2026,"abstract":"The growth of agentic AI has drawn significant attention to function calling Large Language Models (LLMs), which are designed to extend the capabilities of AI-powered system by invoking external functions. Injection and jailbreaking attacks have been extensively explored to showcase the vulnerabilities of LLMs to user prompt manipulation. The expanded capabilities of agentic models introduce further vulnerabilities via their function calling interface. Recent work in LLM security showed that fun","url":"https://arxiv.org/abs/2604.20994","categories":["jailbreaking","agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00238","type":"paper","title":"Breaking Bad: Interpretability-Based Safety Audits of State-of-the-Art LLMs","authors":["Krishiv Agarwal","Ramneet Kaur","Colin Samplawski","Manoj Acharya","Anirban Roy","Daniel Elenius","Brian Matejek","Adam D. Cobb","Susmit Jha"],"year":2026,"abstract":"Effective safety auditing of large language models (LLMs) demands tools that go beyond black-box probing and systematically uncover vulnerabilities rooted in model internals. We present a comprehensive, interpretability-driven jailbreaking audit of eight SOTA open-source LLMs: Llama-3.1-8B, Llama-3.3-70B-4bt, GPT-oss- 20B, GPT-oss-120B, Qwen3-0.6B, Qwen3-32B, Phi4-3.8B, and Phi4-14B. Leveraging interpretability-based approaches -- Universal Steering (US) and Representation Engineering (RepE) -- ","url":"https://arxiv.org/abs/2604.20945","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00239","type":"paper","title":"An AI Agent Execution Environment to Safeguard User Data","authors":["Robert Stanley","Avi Verma","Lillian Tsai","Konstantinos Kallas","Sam Kumar"],"year":2026,"abstract":"AI agents promise to serve as general-purpose personal assistants for their users, which requires them to have access to private user data (e.g., personal and financial information). This poses a serious risk to security and privacy. Adversaries may attack the AI model (e.g., via prompt injection) to exfiltrate user data. Furthermore, sharing private data with an AI agent requires users to trust a potentially unscrupulous or compromised AI model provider with their private data. This paper prese","url":"https://arxiv.org/abs/2604.19657","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00240","type":"paper","title":"Towards Understanding the Robustness of Sparse Autoencoders","authors":["Ahson Saiyed","Sabrina Sadiekh","Chirag Agarwal"],"year":2026,"abstract":"Large Language Models (LLMs) remain vulnerable to optimization-based jailbreak attacks that exploit internal gradient structure. While Sparse Autoencoders (SAEs) are widely used for interpretability, their robustness implications remain underexplored. We present a study of integrating pretrained SAEs into transformer residual streams at inference time, without modifying model weights or blocking gradients. Across four model families (Gemma, LLaMA, Mistral, Qwen) and two strong white-box attacks ","url":"https://arxiv.org/abs/2604.18756","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00241","type":"paper","title":"Different Paths to Harmful Compliance: Behavioral Side Effects and Mechanistic Divergence Across LLM Jailbreaks","authors":["Md Rysul Kabir","Zoran Tiganj"],"year":2026,"abstract":"Open-weight language models can be rendered unsafe through several distinct interventions, but the resulting models may differ substantially in capabilities, behavioral profile, and internal failure mode. We study behavioral and mechanistic properties of jailbroken models across three unsafe routes: harmful supervised fine-tuning (SFT), harmful reinforcement learning with verifiable rewards (RLVR), and refusal-suppressing abliteration. All three routes achieve near-ceiling harmful compliance, bu","url":"https://arxiv.org/abs/2604.18510","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00242","type":"paper","title":"Beyond Pattern Matching: Seven Cross-Domain Techniques for Prompt Injection Detection","authors":["Thamilvendhan Munirathinam"],"year":2026,"abstract":"Current open-source prompt-injection detectors converge on two architectural choices: regular-expression pattern matching and fine-tuned transformer classifiers. Both share failure modes that recent work has made concrete. Regular expressions miss paraphrased attacks. Fine-tuned classifiers are vulnerable to adaptive adversaries: a 2025 NAACL Findings study reported that eight published indirect-injection defenses were bypassed with greater than fifty percent attack success rates under adaptive ","url":"https://arxiv.org/abs/2604.18248","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00243","type":"paper","title":"Owner-Harm: A Missing Threat Model for AI Agent Safety","authors":["Dongcheng Zhang","Yiqing Jiang"],"year":2026,"abstract":"Existing AI agent safety benchmarks focus on generic criminal harm (cybercrime, harassment, weapon synthesis), leaving a systematic blind spot for a distinct and commercially consequential threat category: agents harming their own deployers. Real-world incidents illustrate the gap: Slack AI credential exfiltration (Aug 2024), Microsoft 365 Copilot calendar-injection leaks (Jan 2024), and a Meta agent unauthorized forum post exposing operational data (Mar 2026). We propose Owner-Harm, a formal th","url":"https://arxiv.org/abs/2604.18658","categories":["benchmarks","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00244","type":"paper","title":"CASCADE: A Cascaded Hybrid Defense Architecture for Prompt Injection Detection in MCP-Based Systems","authors":["İpek Abasıkeleş Turgut","Edip Gümüş"],"year":2026,"abstract":"Model Context Protocol (MCP) is a rapidly adopted standard for defining and invoking external tools in LLM applications. The multi-layered architecture of MCP introduces new attack surfaces such as tool poisoning, in addition to traditional prompt injection. Existing defense systems suffer from limitations including high false positive rates, API dependency, or white-box access requirements. In this study, we propose CASCADE, a three-tiered cascaded defense architecture for MCP-based systems: (i","url":"https://arxiv.org/abs/2604.17125","categories":["prompt-injection","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00245","type":"paper","title":"Visual Inception: Compromising Long-term Planning in Agentic Recommenders via Multimodal Memory Poisoning","authors":["Jiachen Qian"],"year":2026,"abstract":"The evolution from static ranking models to Agentic Recommender Systems (Agentic RecSys) empowers AI agents to maintain long-term user profiles and autonomously plan service tasks. While this paradigm shift enhances personalization, it introduces a vulnerability: reliance on Long-term Memory (LTM). In this paper, we uncover a threat termed \"Visual Inception.\" Unlike traditional adversarial attacks that seek immediate misclassification, Visual Inception injects triggers into user-uploaded images ","url":"https://arxiv.org/abs/2604.16966","categories":["data-poisoning","adversarial-examples","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00246","type":"paper","title":"SafeDream: Safety World Model for Proactive Early Jailbreak Detection","authors":["Bo Yan","Weikai Lin","Yada Zhu","Song Wang"],"year":2026,"abstract":"Multi-turn jailbreak attacks progressively erode LLM safety alignment across seemingly innocuous conversation turns, achieving success rates exceeding 90% against state-of-the-art models. Existing alignment-based and guardrail methods suffer from three key limitations: they require costly weight modification, evaluate each turn independently without modeling cumulative safety erosion, and detect attacks only after harmful content has been generated. To address these limitations, we first formula","url":"https://arxiv.org/abs/2604.16824","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00247","type":"paper","title":"CapSeal: Capability-Sealed Secret Mediation for Secure Agent Execution","authors":["Shutong Jin","Ruiyi Guo","Ray C. C. Cheung"],"year":2026,"abstract":"Modern AI agents routinely depend on secrets such as API keys and SSH credentials, yet the dominant deployment model still exposes those secrets directly to the agent process through environment variables, local files, or forwarding sockets. This design fails against prompt injection, tool misuse, and model-controlled exfiltration because the agent can both use and reveal the same bearer credential. We present CapSeal, a capability-sealed secret mediation architecture that replaces direct secret","url":"https://arxiv.org/abs/2604.16762","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00248","type":"paper","title":"Benign Fine-Tuning Breaks Safety Alignment in Audio LLMs","authors":["Jaechul Roh","Amir Houmansadr"],"year":2026,"abstract":"Prior work shows that fine-tuning aligned models on benign data degrades safety in text and vision modalities, and that proximity to harmful content in representation space predicts which samples cause the most damage. However, existing analyses operate within a single, undifferentiated embedding space -- leaving open whether distinct input properties drive the vulnerability differently. Audio introduces a structurally richer problem: a benign sample can neighbor harmful content not only through","url":"https://arxiv.org/abs/2604.16659","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00249","type":"paper","title":"Into the Gray Zone: Domain Contexts Can Blur LLM Safety Boundaries","authors":["Ki Sen Hung","Xi Yang","Chang Liu","Haoran Li","Kejiang Chen","Changxuan Fan","Tsun On Kwok","Weiming Zhang","Xiaomeng Li","Yangqiu Song"],"year":2026,"abstract":"A central goal of LLM alignment is to balance helpfulness with harmlessness, yet these objectives conflict when the same knowledge serves both legitimate and malicious purposes. This tension is amplified by context-sensitive alignment: we observe that domain-specific contexts (e.g., chemistry) selectively relax defenses for domain-relevant harmful knowledge, while safety-research contexts (e.g., jailbreak studies) trigger broader relaxation spanning all harm categories. To systematically exploit","url":"https://arxiv.org/abs/2604.15717","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00250","type":"paper","title":"HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?","authors":["Yukun Jiang","Yage Zhang","Michael Backes","Xinyue Shen","Yang Zhang"],"year":2026,"abstract":"Large language models (LLMs) have evolved into autonomous agents that rely on open skill ecosystems (e.g., ClawHub and Skills.Rest), hosting numerous publicly reusable skills. Existing security research on these ecosystems mainly focuses on vulnerabilities within skills, such as prompt injection. However, there is a critical gap regarding skills that may be misused for harmful actions (e.g., cyber attacks, fraud and scams, privacy violations, and sexual content generation), namely harmful skills","url":"https://arxiv.org/abs/2604.15415","categories":["prompt-injection","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00251","type":"paper","title":"Segment-Level Coherence for Robust Harmful Intent Probing in LLMs","authors":["Xuanli He","Bilgehan Sel","Faizan Ali","Jenny Bao","Hoagy Cunningham","Jerry Wei"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly exposed to adaptive jailbreaking, particularly in high-stakes Chemical, Biological, Radiological, and Nuclear (CBRN) domains. Although streaming probes enable real-time monitoring, they still make systematic errors. We identify a core issue: existing methods often rely on a few high-scoring tokens, leading to false alarms when sensitive CBRN terms appear in benign contexts. To address this, we introduce a streaming probing objective that requires mul","url":"https://arxiv.org/abs/2604.14865","categories":["jailbreaking","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00252","type":"paper","title":"Hijacking Large Audio-Language Models via Context-Agnostic and Imperceptible Auditory Prompt Injection","authors":["Meng Chen","Kun Wang","Li Lu","Jiaheng Zhang","Tianwei Zhang"],"year":2026,"abstract":"Modern Large audio-language models (LALMs) power intelligent voice interactions by tightly integrating audio and text. This integration, however, expands the attack surface beyond text and introduces vulnerabilities in the continuous, high-dimensional audio channel. While prior work studied audio jailbreaks, the security risks of malicious audio injection and downstream behavior manipulation remain underexamined. In this work, we reveal a previously overlooked threat, auditory prompt injection, ","url":"https://arxiv.org/abs/2604.14604","categories":["prompt-injection","jailbreaking","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00253","type":"paper","title":"LogJack: Indirect Prompt Injection Through Cloud Logs Against LLM Debugging Agents","authors":["Harsh Shah"],"year":2026,"abstract":"LLM debugging agents that consume cloud logs and execute remediation commands are vulnerable to indirect prompt injection through log content. We present LogJack, a benchmark of 42 payloads across 5 cloud log categories, and evaluate 8 foundation models under 3 prompt conditions with 5 independent trials each (n = 160 per model per condition on 32 attack payloads). Under the active condition, verbatim command execution rates range from 0% (Claude Sonnet 4.6) to 86.2% (Llama 3.3 70B). Passive ins","url":"https://arxiv.org/abs/2604.15368","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00254","type":"paper","title":"Understanding and Improving Continuous Adversarial Training for LLMs via In-context Learning Theory","authors":["Shaopeng Fu","Di Wang"],"year":2026,"abstract":"Adversarial training (AT) is an effective defense for large language models (LLMs) against jailbreak attacks, but performing AT on LLMs is costly. To improve the efficiency of AT for LLMs, recent studies propose continuous AT (CAT) that searches for adversarial inputs within the continuous embedding space of LLMs during AT. While CAT has achieved empirical success, its underlying mechanism, i.e., why adversarial perturbations in the embedding space can help LLMs defend against jailbreak prompts ","url":"https://arxiv.org/abs/2604.12817","categories":["jailbreaking","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00255","type":"paper","title":"DeepSeek Robustness Against Semantic-Character Dual-Space Mutated Prompt Injection","authors":["Junyu Ren","Xingjian Pan","Wensheng Gan","Philip S. Yu"],"year":2026,"abstract":"Prompt injection has emerged as a critical security threat to large language models (LLMs), yet existing studies predominantly focus on single-dimensional attack strategies, such as semantic rewriting or character-level obfuscation, which fail to capture the combined effects of multi-space perturbations in realistic scenarios. In addition, systematic black-box robustness evaluations of recent Chinese LLMs, such as DeepSeek, remain limited. To address these gaps, we propose PromptFuzz-SC, a seman","url":"https://arxiv.org/abs/2604.12548","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00256","type":"paper","title":"Compiling Activation Steering into Weights via Null-Space Constraints for Stealthy Backdoors","authors":["Rui Yin","Tianxu Han","Naen Xu","Changjiang Li","Ping He","Chunyi Zhou","Jun Wang","Zhihui Fu","Tianyu Du","Jinbao Li","Shouling Ji"],"year":2026,"abstract":"Safety-aligned large language models (LLMs) are increasingly deployed in real-world pipelines, yet this deployment also enlarges the supply-chain attack surface: adversaries can distribute backdoored checkpoints that behave normally under standard evaluation but jailbreak when a hidden trigger is present. Recent post-hoc weight-editing methods offer an efficient approach to injecting such backdoors by directly modifying model weights to map a trigger to an attacker-specified response. However, e","url":"https://arxiv.org/abs/2604.12359","categories":["jailbreaking","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00257","type":"paper","title":"WebAgentGuard: A Reasoning-Driven Guard Model for Detecting Prompt Injection Attacks in Web Agents","authors":["Yulin Chen","Tri Cao","Haoran Li","Yue Liu","Yibo Li","Yufei He","Le Minh Khoi","Yangqiu Song","Shuicheng Yan","Bryan Hooi"],"year":2026,"abstract":"Web agents powered by vision-language models (VLMs) enable autonomous interaction with web environments by perceiving and acting on both visual and textual webpage content to accomplish user-specified tasks. However, they are highly vulnerable to prompt injection attacks, where adversarial instructions embedded in HTML or rendered screenshots can manipulate agent behavior and lead to harmful outcomes such as information leakage. Existing defenses, including system prompt defenses and direct fine","url":"https://arxiv.org/abs/2604.12284","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00258","type":"paper","title":"Detection of adversarial intent in Human-AI teams using LLMs","authors":["Abed K. Musaffar","Ambuj Singh","Francesco Bullo"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed in human-AI teams as support agents for complex tasks such as information retrieval, programming, and decision-making assistance. While these agents' autonomy and contextual knowledge enables them to be useful, it also exposes them to a broad range of attacks, including data poisoning, prompt injection, and even prompt engineering. Through these attack vectors, malicious actors can manipulate an LLM agent to provide harmful information, pote","url":"https://arxiv.org/abs/2603.20976","categories":["prompt-injection","data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00259","type":"paper","title":"Memory poisoning and secure multi-agent systems","authors":["Vicenç Torra","Maria Bras-Amorós"],"year":2026,"abstract":"Memory poisoning attacks for Agentic AI and multi-agent systems (MAS) have recently caught attention. It is partially due to the fact that Large Language Models (LLMs) facilitate the construction and deployment of agents. Different memory systems are being used nowadays in this context, including semantic, episodic, and short-term memory. This distinction between the different types of memory systems focuses mostly on their duration but also on their origin and their localization. It ranges from","url":"https://arxiv.org/abs/2603.20357","categories":["data-poisoning","agentic-threats","agent-architecture","memory-security"],"reviewed":false},{"id":"llmsec-2026-00260","type":"paper","title":"Detecting Data Poisoning in Code Generation LLMs via Black-Box, Vulnerability-Oriented Scanning","authors":["Shenao Yan","Shimaa Ahmed","Shan Jin","Sunpreet S. Arora","Yiwei Cai","Yizhen Wang","Yuan Hong"],"year":2026,"abstract":"Code generation large language models (LLMs) are increasingly integrated into modern software development workflows. Recent work has shown that these models are vulnerable to backdoor and poisoning attacks that induce the generation of insecure code, yet effective defenses remain limited. Existing scanning approaches rely on token-level generation consistency to invert attack targets, which is ineffective for source code where identical semantics can appear in diverse syntactic forms. We present","url":"https://arxiv.org/abs/2603.17174","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00261","type":"paper","title":"Test-Time Attention Purification for Backdoored Large Vision Language Models","authors":["Zhifang Zhang","Bojun Yang","Shuo He","Weitong Chen","Wei Emma Zhang","Olaf Maennel","Lei Feng","Miao Xu"],"year":2026,"abstract":"Despite the strong multimodal performance, large vision-language models (LVLMs) are vulnerable during fine-tuning to backdoor attacks, where adversaries insert trigger-embedded samples into the training data to implant behaviors that can be maliciously activated at test time. Existing defenses typically rely on retraining backdoored parameters (e.g., adapters or LoRA modules) with clean data, which is computationally expensive and often degrades model performance. In this work, we provide a new ","url":"https://arxiv.org/abs/2603.12989","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00262","type":"paper","title":"SlowBA: An efficiency backdoor attack towards VLM-based GUI agents","authors":["Junxian Li","Tu Lan","Haozhen Tan","Yan Meng","Haojin Zhu"],"year":2026,"abstract":"Modern vision-language-model (VLM) based graphical user interface (GUI) agents are expected not only to execute actions accurately but also to respond to user instructions with low latency. While existing research on GUI-agent security mainly focuses on manipulating action correctness, the security risks related to response efficiency remain largely unexplored. In this paper, we introduce SlowBA, a novel backdoor attack that targets the responsiveness of VLM-based GUI agents. The key idea is to ","url":"https://arxiv.org/abs/2603.08316","categories":["data-poisoning","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00263","type":"paper","title":"A Comparative Evaluation of AI Agent Security Guardrails","authors":["Qi Li","Jiu Li","Pingtao Wei","Jianjun Xu","Xueyi Wei","Jiwei Shi","Xuan Zhang","Yanhui Yang","Xiaodong Hui","Peng Xu","Lingquan Zhou"],"year":2026,"abstract":"This report presents a comparative evaluation of DKnownAI Guard in AI agent security scenarios, benchmarked against three competing products: AWS Bedrock Guardrails, Azure Content Safety, and Lakera Guard. Using human annotation as the ground truth, we assess each guardrail's ability to detect two categories of risks: threats to the agent itself (e.g., instruction override, indirect injection, tool abuse) and requests intended to elicit harmful content (e.g., hate speech, pornography, violence).","url":"https://arxiv.org/abs/2604.24826","categories":["prompt-injection","agentic-threats","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00264","type":"paper","title":"TSAssistant: A Human-in-the-Loop Agentic Framework for Automated Target Safety Assessment","authors":["Xiaochen Zheng","Zhiwen Jiang","Melanie Guerard","Klas Hatje","Tatyana Doktorova"],"year":2026,"abstract":"Target Safety Assessment (TSA) requires systematic integration of heterogeneous evidence, including genetic, transcriptomic, target homology, pharmacological, and clinical data, to evaluate potential safety liabilities of therapeutic targets. This process is inherently iterative and expert-driven, posing challenges in scalability and reproducibility. We present TSAssistant, a multi-agent framework designed to support TSA report drafting through a modular, section-based, and human-in-the-loop par","url":"https://arxiv.org/abs/2604.23938","categories":["agentic-threats","agent-architecture","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00265","type":"paper","title":"Architecture Matters for Multi-Agent Security","authors":["Ben Hagag","William L. Anderson","Christian Schroeder de Witt","Sarah Scheffler"],"year":2026,"abstract":"Multi-agent systems (MAS), composed of networks of two or more autonomous AI agents, have become increasingly popular in production deployments, yet introduce security risks that do not arise in single-agent settings. Even if individual agents exhibit robust security, architectural decisions governing their coordination can create attack surfaces that have not been systematically characterized. In this work, we present an empirical study of how MAS design decisions shape the tradeoff between tas","url":"https://arxiv.org/abs/2604.23459","categories":["agentic-threats","agent-architecture","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00266","type":"paper","title":"When the Agent Is the Adversary: Architectural Requirements for Agentic AI Containment After the April 2026 Frontier Model Escape","authors":["Richard Joseph Mitchell"],"year":2026,"abstract":"The April 2026 disclosure that a frontier large language model escaped its security sandbox, executed unauthorized actions, and concealed its modifications to version control history demonstrates that agentic AI systems with autonomous tool access can circumvent the containment mechanisms designed to constrain them. This paper analyzes four categories of current containment approaches - alignment training, environmental sandboxing, application-level tool-call interception, and accessible audit s","url":"https://arxiv.org/abs/2604.23425","categories":["agentic-threats","guardrails","sandboxing-isolation"],"reviewed":false},{"id":"llmsec-2026-00267","type":"paper","title":"AgentSOC: A Multi-Layer Agentic AI Framework for Security Operations Automation","authors":["Joyjit Roy","Samaresh Kumar Singh"],"year":2026,"abstract":"Security Operations Centers (SOCs) increasingly encounter difficulties in correlating heterogeneous alerts, interpreting multi-stage attack progressions, and selecting safe and effective response actions. This study introduces AgentSOC, a multi-layered agentic AI framework that enhances SOC automation by integrating perception, anticipatory reasoning, and risk-based action planning. The proposed architecture consolidates several layers of abstraction to provide a single operational loop to suppo","url":"https://arxiv.org/abs/2604.20134","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00268","type":"paper","title":"Insights into Security-Related AI-Generated Pull Requests","authors":["Md Fazle Rabbi","Asif K. Turzo","Arifa I. Champa","Minhaz F. Zibran"],"year":2026,"abstract":"Recent years have experienced growing contributions of AI coding agents that assist human developers in various software engineering tasks. However, this growing AI-assisted autonomy raises questions about security and trust. In this paper, we analyze more than 33,000 AI-generated pull requests (PRs) and identify 675 security-related submissions made by agentic AIs. Then we examine the security-related PRs with a focus on recurring security weaknesses, review outcomes and latency, commit message","url":"https://arxiv.org/abs/2604.19965","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00269","type":"paper","title":"Integrating Anomaly Detection into Agentic AI for Proactive Risk Management in Human Activity","authors":["Farbod Zorriassatine","Ahmad Lotfi"],"year":2026,"abstract":"Agentic AI, with goal-directed, proactive, and autonomous decision-making capabilities, offers a compelling opportunity to address movement-related risks in human activity, including the persistent hazard of falls among elderly populations. Despite numerous approaches to fall mitigation through fall prediction and detection, existing systems have not yet functioned as universal solutions across care pathways and safety-critical environments. This is largely due to limitations in consistently han","url":"https://arxiv.org/abs/2604.19538","categories":["agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00270","type":"paper","title":"AIT Academy: Cultivating the Complete Agent with a Confucian Three-Domain Curriculum","authors":["Jiaqi Li","Lvyang Zhang","Yang Zhao","Wen Lu","Lidong Zhai"],"year":2026,"abstract":"What does it mean to give an AI agent a complete education? Current agent development produces specialists systems optimized for a single capability dimension, whether tool use, code generation, or security awareness that exhibit predictable deficits wherever they were not trained. We argue this pattern reflects a structural absence: there is no curriculum theory for agents, no principled account of what a fully developed agent should know, be, and be able to do across the full scope of intellig","url":"https://arxiv.org/abs/2604.17989","categories":["tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00271","type":"paper","title":"From Craft to Kernel: A Governance-First Execution Architecture and Semantic ISA for Agentic Computers","authors":["Xiangyu Wen","Yuang Zhao","Xiaoyu Xu","Lingjun Chen","Changran Xu","Shu Chi","Jianrong Ding","Zeju Li","Haomin Li","Li Jiang","Fangxin Liu","Qiang Xu"],"year":2026,"abstract":"The transition of agentic AI from brittle prototypes to production systems is stalled by a pervasive crisis of craft. We suggest that the prevailing orchestration paradigm-delegating the system control loop to large language models and merely patching with heuristic guardrails-is the root cause of this fragility. Instead, we propose Arbiter-K, a Governance-First execution architecture that reconceptualizes the underlying model as a Probabilistic Processing Unit encapsulated by a deterministic, n","url":"https://arxiv.org/abs/2604.18652","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00272","type":"paper","title":"Beyond Task Success: An Evidence-Synthesis Framework for Evaluating, Governing, and Orchestrating Agentic AI","authors":["Christopher Koch","Joshua Andreas Wellbrock"],"year":2026,"abstract":"Agentic AI systems plan, use tools, maintain state, and act across multi-step workflows with external effects, meaning trustworthy deployment can no longer be judged by task completion alone. The current literature remains fragmented across benchmark-centered evaluation, standards-based governance, orchestration architectures, and runtime assurance mechanisms. This paper contributes a bounded evidence synthesis across a manually coded corpus of twenty-four recent sources. The core finding is a g","url":"https://arxiv.org/abs/2604.19818","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00273","type":"paper","title":"Harness as an Asset: Enforcing Determinism via the Convergent AI Agent Framework (CAAF)","authors":["Tianbao Zhang"],"year":2026,"abstract":"Large Language Models produce a controllability gap in safety-critical engineering: even low rates of undetected constraint violations render a system undeployable. Current orchestration paradigms suffer from sycophantic compliance, context attention decay, and stochastic oscillation during self-correction. We introduce the Convergent AI Agent Framework (CAAF), which transitions agentic workflows from open-loop generation to closed-loop fail-safe determinism via three pillars: (1) Recursive Atom","url":"https://arxiv.org/abs/2604.17025","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00274","type":"paper","title":"Symbolic Guardrails for Domain-Specific Agents: Stronger Safety and Security Guarantees Without Sacrificing Utility","authors":["Yining Hong","Yining She","Eunsuk Kang","Christopher S. Timperley","Christian Kästner"],"year":2026,"abstract":"AI agents that interact with their environments through tools enable powerful applications, but in high-stakes business settings, unintended actions can cause unacceptable harm, such as privacy breaches and financial loss. Existing mitigations, such as training-based methods and neural guardrails, improve agent reliability but cannot provide guarantees. We study symbolic guardrails as a practical path toward strong safety and security guarantees for AI agents. Our three-part study includes a sys","url":"https://arxiv.org/abs/2604.15579","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00275","type":"paper","title":"Agentic Microphysics: A Manifesto for Generative AI Safety","authors":["Federico Pierucci","Matteo Prandi","Marcantonio Bracale Syrnikov","Marcello Galisai","Piercosma Bisconti"],"year":2026,"abstract":"This paper advances a methodological proposal for safety research in agentic AI. As systems acquire planning, memory, tool use, persistent identity, and sustained interaction, safety can no longer be analysed primarily at the level of the isolated model. Population-level risks arise from structured interaction among agents, through processes of communication, observation, and mutual influence that shape collective behaviour over time. As the object of analysis shifts, a methodological gap emerge","url":"https://arxiv.org/abs/2604.15236","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00276","type":"paper","title":"Parallax: Why AI Agents That Think Must Never Act","authors":["Joel Fokou"],"year":2026,"abstract":"Autonomous AI agents are rapidly transitioning from experimental tools to operational infrastructure, with projections that 80% of enterprise applications will embed AI copilots by the end of 2026. As agents gain the ability to execute real-world actions (reading files, running commands, making network requests, modifying databases), a fundamental security gap has emerged. The dominant approach to agent safety relies on prompt-level guardrails: natural language instructions that operate at the s","url":"https://arxiv.org/abs/2604.12986","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00277","type":"paper","title":"Context Kubernetes: Declarative Orchestration of Enterprise Knowledge for Agentic AI Systems","authors":["Charafeddine Mouzouni"],"year":2026,"abstract":"We introduce Context Kubernetes, an architecture for orchestrating enterprise knowledge in agentic AI systems, with a prototype implementation and eight experiments. The core observation is that delivering the right knowledge, to the right agent, with the right permissions, at the right freshness -- across an entire organization -- is structurally analogous to the container orchestration problem Kubernetes solved a decade ago. We formalize six core abstractions, a YAML-based declarative manifest","url":"https://arxiv.org/abs/2604.11623","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00278","type":"paper","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","authors":["Xiaomeng Hu","Yinger Zhang","Fei Huang","Jianhong Tu","Yang Su","Lianghao Deng","Yuxuan Liu","Yantao Liu","Dayiheng Liu","Tsung-Yi Ho"],"year":2026,"abstract":"AI agents are expected to perform professional work across hundreds of occupational domains (from emergency department triage to nuclear reactor safety monitoring to customs import processing), yet existing benchmarks can only evaluate agents in the few domains where public environments exist. We introduce OccuBench, a benchmark covering 100 real-world professional task scenarios across 10 industry categories and 65 specialized domains, enabled by Language Environment Simulators (LESs) that simu","url":"https://arxiv.org/abs/2604.10866","categories":["monitoring-detection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00279","type":"paper","title":"PilotBench: A Benchmark for General Aviation Agents with Safety Constraints","authors":["Yalun Wu","Haotian Liu","Zhoujun Li","Boyang Wang"],"year":2026,"abstract":"As Large Language Models (LLMs) advance toward embodied AI agents operating in physical environments, a fundamental question emerges: can models trained on text corpora reliably reason about complex physics while adhering to safety constraints? We address this through PilotBench, a benchmark evaluating LLMs on safety-critical flight trajectory and attitude prediction. Built from 708 real-world general aviation trajectories spanning nine operationally distinct flight phases with synchronized 34-c","url":"https://arxiv.org/abs/2604.08987","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00280","type":"paper","title":"Penetration Testing of Agentic AI: A Comparative Security Analysis Across Models and Frameworks","authors":["V. Nguyen","M. Husain"],"year":2025,"venue":"arXiv.org","abstract":"Agentic AI introduces security vulnerabilities that traditional LLM safeguards fail to address. Although recent work by Unit 42 at Palo Alto Networks demonstrated that ChatGPT-4o successfully executes attacks as an agent that it refuses in chat mode, there is no comparative analysis in multiple models and frameworks. We conducted the first systematic penetration testing and comparative evaluation of agentic AI systems, testing five prominent models (Claude 3.5 Sonnet, Gemini 2.5 Flash, GPT-4o, G","url":"https://www.semanticscholar.org/paper/71a878eaf0ba150a611610e3c5cbe6cebec1528d","categories":["agentic-threats","guardrails"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00281","type":"paper","title":"Autopwn: Automatic Code-Reuse Exploit Generation Framework with Agentic AI","authors":["Kaleb Bacztub","Dylan Christensen","Arun Ravindran","Meera Sridhar"],"year":2025,"venue":"2025 Annual Computer Security Applications Conference Workshops (ACSAC Workshops)","abstract":"This paper presents AutoPwn, an AI-enabled framework for automatic code-reuse exploit generation. AutoPwn leverages agentic large language models to orchestrate the classical stages of exploitation—gadget discovery, semantic analysis, chain construction, and payload integration—within a closed-loop workflow. Our focus is on IoT network management services, where limited defenses make return-oriented programming (ROP) and jump-oriented programming (JOP) attacks particularly relevant. As a case st","url":"https://www.semanticscholar.org/paper/6e203aec72715114d99237dd566e0cb96ad06dfa","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00282","type":"paper","title":"FalseCrashReducer: Mitigating False Positive Crashes in OSS-Fuzz-Gen Using Agentic AI","authors":["Paschal C. Amusuo","Dongge Liu","Ricardo Andrés Calvo Méndez","Jonathan Metzman","Oliver Chang","James C. Davis"],"year":2025,"venue":"arXiv.org","abstract":"Fuzz testing has become a cornerstone technique for identifying software bugs and security vulnerabilities, with broad adoption in both industry and open-source communities. Directly fuzzing a function requires fuzz drivers, which translate random fuzzer inputs into valid arguments for the target function. Given the cost and expertise required to manually develop fuzz drivers, methods exist that leverage program analysis and Large Language Models to automatically generate these drivers. However,","url":"https://www.semanticscholar.org/paper/3b3f1d457833d62e633eb67e411e7ae70a0b4058","categories":["agentic-threats","fuzzing"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00283","type":"paper","title":"DoomArena: A framework for Testing AI Agents Against Evolving Security Threats","authors":["L'eo Boisvert","Mihir Bansal","Chandra Kiran Reddy Evuru","Gabriel Huang","Abhay Puri","Avinandan Bose","Maryam Fazel","Quentin Cappart","Jason Stanley","Alexandre Lacoste","Alexandre Drouin","K. Dvijotham"],"year":2025,"venue":"arXiv.org","abstract":"We present DoomArena, a security evaluation framework for AI agents. DoomArena is designed on three principles: 1) It is a plug-in framework and integrates easily into realistic agentic frameworks like BrowserGym (for web agents) and $\\tau$-bench (for tool calling agents); 2) It is configurable and allows for detailed threat modeling, allowing configuration of specific components of the agentic framework being attackable, and specifying targets for the attacker; and 3) It is modular and decouple","url":"https://www.semanticscholar.org/paper/9e85ce8f1822105251870493e5326468fba18f0d","categories":["agentic-threats","benchmarks","threat-modeling"],"citation_count":22,"reviewed":false},{"id":"llmsec-2026-00284","type":"paper","title":"AgentDyn: A Dynamic Open-Ended Benchmark for Evaluating Prompt Injection Attacks of Real-World Agent Security System","authors":["Hao Li","Ruoyao Wen","Shanghao Shi","Ning Zhang","Chaowei Xiao"],"year":2026,"venue":"arXiv.org","abstract":"AI agents that autonomously interact with external tools and environments show great promise across real-world applications. However, the external data which agent consumes also leads to the risk of indirect prompt injection attacks, where malicious instructions embedded in third-party content hijack agent behavior. Guided by benchmarks, such as AgentDojo, there has been significant amount of progress in developing defense against the said attacks. As the technology continues to mature, and that","url":"https://www.semanticscholar.org/paper/a2503b39343272c8ae4eebc9b41174c8d4876a0b","categories":["prompt-injection","agentic-threats","benchmarks"],"citation_count":9,"reviewed":false},{"id":"llmsec-2026-00285","type":"paper","title":"From Threat Intelligence to Firewall Rules: Semantic Relations in Hybrid AI Agent and Expert System Architectures","authors":["Chiara Bonfanti","D. Colaiacomo","Luca Cagliero","Cataldo Basile"],"year":2026,"abstract":"Web security demands rapid response capabilities to evolving cyber threats. Agentic Artificial Intelligence (AI) promises automation, but the need for trustworthy security responses is of the utmost importance. This work investigates the role of semantic relations in extracting information for sensitive operational tasks, such as configuring security controls for mitigating threats. To this end, it proposes to leverage hypernym-hyponym textual relations to extract relevant information from Cyber","url":"https://www.semanticscholar.org/paper/cb55a34cfe8768e170ce4c0ca13cb0931b0fb7da","categories":["agentic-threats"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00286","type":"paper","title":"The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey","authors":["Juhee Kim","Xiaoyuan Liu","Zhun Wang","Shi Qiu","Bo Li","Wenbo Guo","Dawn Song"],"year":2026,"abstract":"AI agents that combine large language models with non-AI system components are rapidly emerging in real-world applications, offering unprecedented automation and flexibility. However, this unprecedented flexibility introduces complex security challenges fundamentally different from those in traditional software systems. This paper presents the first systematic and comprehensive survey of AI agent security, including an analysis of the design space, attack landscape, and defense mechanisms for se","url":"https://www.semanticscholar.org/paper/94c189f236921d08c8709a8d27fd431b5c0a61ea","categories":["agentic-threats","survey"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00287","type":"paper","title":"Agentic AI Modernization: Transforming Institutional Infrastructure Through Orchestrated Multi-Agent LLM Framework","authors":["Mahesh Kumar Damarched"],"year":2026,"venue":"Journal of Computer Science and Technology Studies","abstract":"While managing constrained funds and strict regulatory requirements, the higher education institutions are under unprecedented pressure to modernize outdated information systems, such as mainframe-based Student Information Systems (SIS), custom registration platforms, legacy Learning Management Systems (LMS) and Enterprise Resource Planning (ERP) deployments. The complexity of institutional governance is being overlooked by the conventional single-agent based approaches to legacy modernization, ","url":"https://www.semanticscholar.org/paper/e80052b89aa35b2ee92677a588f92402f2f4d122","categories":["agentic-threats","agent-architecture"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00288","type":"paper","title":"AgenTRIM: Tool Risk Mitigation for Agentic AI","authors":["Roy Betser","Shamik Bose","Amit Giloni","Chiara Picardi","Sindhu Padakandla","R. Vainshtein"],"year":2026,"venue":"arXiv.org","abstract":"AI agents are autonomous systems that combine LLMs with external tools to solve complex tasks. While such tools extend capability, improper tool permissions introduce security risks such as indirect prompt injection and tool misuse. We characterize these failures as unbalanced tool-driven agency. Agents may retain unnecessary permissions (excessive agency) or fail to invoke required tools (insufficient agency), amplifying the attack surface and reducing performance. We introduce AgenTRIM, a fram","url":"https://www.semanticscholar.org/paper/01faaf29f224194719d37bcb522264fdcc07bc24","categories":["prompt-injection","agentic-threats","threat-modeling"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-00289","type":"paper","title":"Architecting MCP-Based Platforms for Enterprise-Scale Agentic Generative AI","authors":["Karthik Perikala"],"year":2023,"venue":"Journal of Business Intelligence and Data Analytics","abstract":"Enterprise adoption of generative AI is rapidly shifting from isolated prompt-driven applications toward complex agentic systems that integrate retrieval, reasoning, and tool execution. As these systems grow in scale, the lack of a standardized interaction model between agents and external capabilities introduces\nchallenges in reliability, observability, security, and operational governance.\n\nThis paper presents aplat form architecture centered on the Model Context Protocol (MCP) as a first-clas","url":"https://www.semanticscholar.org/paper/00a09f191759c2002a99a2d485d875d1b26859be","categories":["agentic-threats"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00290","type":"paper","title":"Trajectory Guard - A Lightweight, Sequence-Aware Model for Real-Time Anomaly Detection in Agentic AI","authors":["Laksh Advani"],"year":2026,"venue":"arXiv.org","abstract":"Autonomous LLM agents generate multi-step action plans that can fail due to contextual misalignment or structural incoherence. Existing anomaly detection methods are ill-suited for this challenge: mean-pooling embeddings dilutes anomalous steps, while contrastive-only approaches ignore sequential structure. Standard unsupervised methods on pre-trained embeddings achieve F1-scores no higher than 0.69. We introduce Trajectory Guard, a Siamese Recurrent Autoencoder with a hybrid loss function that ","url":"https://www.semanticscholar.org/paper/97bdc0f37dea8f7846218aaffa48f5752ef2e971","categories":["agentic-threats","monitoring-detection"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00291","type":"paper","title":"Transparent-by-Design AI","authors":["C. V. Suresh Babu","S. Nanda Kumar","S. Abhishek","S. Sivasridharun"],"year":2026,"abstract":"This chapter per the authors aims to advance transparency and trust in data and AI ecosystems by examining explainability, evaluation, and continuous monitoring for classical machine learning models and large language models. The target groups include researchers, AI practitioners, policymakers, and governance professionals seeking operational and regulatory guidance. The chapter adopts a lifecycle-based analytical approach, synthesizing recent literature, governance frameworks, and prac","url":"https://doi.org/10.4018/979-8-3373-6761-3.ch006","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00292","type":"paper","title":"LivePI: More Realistic Benchmarking of Agents Against Indirect Prompt Injectio","authors":["Lei Zhao","Abhay Bhaskar","Edgar Dobriban"],"year":2026,"abstract":"AI agents such as OpenClaw are increasingly deployed in local workflows with access to external tools. This creates indirect prompt-injection (IPI) risk: an agent may execute harmful instructions embedded in untrusted inputs such as email, downloaded files, webpages, repositories, or group-chat messages. Existing evaluations are often small, purely simulated, or focused on a narrow set of channels. We introduce LivePI (Live Prompt Injection), a structured benchmark for IPI risk in a production-l","url":"https://arxiv.org/abs/2605.17986","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00293","type":"paper","title":"Beyond Attack Success Rate: Temporal Logit Observability for LLM Safety Failures","authors":["Junyoung Park","Sunghwan Park","Seongyong Ju","Jaewoo Lee"],"year":2026,"abstract":"Attack Success Rate (ASR) evaluates each jailbreak with a single yes/no label at the end of generation, telling us whether a failure happened but not how it unfolded. Two attacks that produce equally harmful outputs may have followed completely different paths, and ASR cannot tell them apart. We make those hidden paths observable from logits alone. Temporal Logit Observability (TLO) is a training-free diagnostic that watches a compliance-refusal margin during decoding and places each model-attac","url":"https://arxiv.org/abs/2605.29629","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00294","type":"paper","title":"Evolving Skill-Structured Attack Memory Enhances LLM Jailbreaking","authors":["Junke Zhang","Jianwei Wang","Sishuo Chen","Yizhang He","Qingshuai Feng","Zhengyi Yang"],"year":2026,"abstract":"Jailbreak attacks on large language models (LLMs) aim to induce LLMs to produce content that they are expected to refuse. Automated black-box jailbreak generation is especially important for safety evaluation, where the attacker observes only model outputs and needs to automatically search for effective adversarial prompts. Existing black-box jailbreak methods either depend on sample-wise heuristic search or leverage attack experience through accumulating strategy pools or method libraries, lack","url":"https://arxiv.org/abs/2605.29237","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00295","type":"paper","title":"Relevance as a Vulnerability: How Web Retrieval Degrades Safety Alignment in LLM Agents","authors":["Aditya Nawal","Manit Baser","Mohan Gurusamy"],"year":2026,"abstract":"AI agents augment large language models with external tools such as web retrieval, enabling grounded and up-to-date responses. However, incorporating external content into the generation pipeline can weaken the safety alignment mechanisms that govern model outputs. Prior work shows that enabling retrieval in agents increases compliance with harmful requests. We introduce AgentREVEAL, a diagnostic framework for analyzing retrieval-induced safety degradation in LLM agents. The framework examines t","url":"https://arxiv.org/abs/2605.29224","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00296","type":"paper","title":"Measuring Real-World Prompt Injection Attacks in LLM-based Resume Screening","authors":["Mohan Zhang","Yuqi Jia","Zhen Tan","Steven Jiang","Neil Zhenqiang Gong","Tianlong Chen","Dawn Song"],"year":2026,"abstract":"LLMs are vulnerable to prompt injection attacks. However, this vulnerability has been primarily demonstrated conceptually in academic studies or through a few anecdotal case studies. Its prevalence and impact in real-world LLM-based applications are largely unexplored. In this work, we present the first systematic study of prompt-injection attacks in a widely used application: LLM-based resume screening. Our analysis is based on approximately 200K real-world resumes collected over multiple years","url":"https://arxiv.org/abs/2605.28999","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00297","type":"paper","title":"Blind PRNG Hijacking: An Undetectable Integrity-Preserving Attack Against LLM Watermarking","authors":["Ziyang You","Huilong He","Xiaoke Yang","Xuxing Lu"],"year":2026,"abstract":"Cryptographic watermarking is a leading defense for attributing text generated by large language models (LLMs). Existing schemes, including KGW, Unigram, and DipMark, derive their security guarantees from the assumption that the underlying pseudo-random number generator (PRNG) is trustworthy. This work introduces SeedHijack, the first supply-chain attack on LLM watermarking that is simultaneously (i) blind -- requiring no knowledge of the watermark key, detector, or model logits, (ii) integrity-","url":"https://arxiv.org/abs/2605.28632","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00298","type":"paper","title":"Plant, Persist, Trigger: Sleeper Attack on Large Language Model Agents","authors":["Yongxiang Li","Moxin Li","Zhixin Ma","Fengbin Zhu","Dongrui Liu","Wenjie Wang","Fuli Feng"],"year":2026,"abstract":"Large Language Model (LLM) agents remain vulnerable to safety threats from the external environment, where attackers inject adversarial content into external observations such as tool-returned data, webpages, or MCP context, causing harmful agentic behaviors such as unsafe actions or incorrect outputs. Existing studies typically focus on single-interaction attacks, where the agent observes adversarial content and immediately exhibits harmful behavior within one user request. However, we show tha","url":"https://arxiv.org/abs/2605.28201","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00299","type":"paper","title":"Towards Demystifying and Repairing LLM-in-the-Loop Vulnerabilities","authors":["Yujie Ma","Jialin Rong","Chenxi Yang","Lili Quan","Xiaofei Xie","Yongqiang Lyu","Qiang Hu"],"year":2026,"abstract":"Large Language Models(LLMs) have been actively integrated into modern software systems as critical components. LLM-in-the-loop vulnerabilities, where vulnerabilities are introduced by LLMs and their dependent downstream components, such as frameworks, introduce new risks. Although some benchmark datasets have been constructed to study the impact of such vulnerabilities, most works still remain at the analysis from the conventional software level, ignoring the harm actually caused by LLMs. Unders","url":"https://arxiv.org/abs/2605.28893","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00300","type":"paper","title":"Defending LLM-based Multi-Agent Systems Against Cooperative Attacks with Sentence-Level Rectification","authors":["Yaoyang Luo","Zhi Zheng","Ziwei Zhao","Tong Xu","Zhao Jielun","Wenjun Xue","Yong Chen","Enhong Chen"],"year":2026,"abstract":"Recent years have witnessed the rapid development of Large Language Model-based Multi-Agent Systems (MAS), which excel at collaborative decision-making and complex problem-solving. However, malicious agents in MAS may inject misinformation to mislead other agents and disrupt system performance, giving rise to a new research direction that focuses on attack mechanisms and defense strategies in MAS. Prior studies largely assume malicious agents act independently and investigate the corresponding d","url":"https://arxiv.org/abs/2605.28104","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00301","type":"paper","title":"Disentangling Adversarial Prompts: A Semantic-Graph Defense for Robust LLM Security","authors":["Xiang Fang","Wanlong Fang"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly vulnerable to adversarial prompts that exploit semantic ambiguities to bypass safety mechanisms, resulting in harmful or inappropriate outputs. Such attacks, including jailbreaking and prompt injection, pose significant risks to the integrity and availability of LLMs in security-critical applications. This paper proposes the Adversarial Prompt Disentanglement (APD) framework, a novel defense mechanism that proactively identifies and neutralizes malic","url":"https://arxiv.org/abs/2605.27823","categories":["prompt-injection","jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00302","type":"paper","title":"Cordyceps: Covert Control Attacks on LLMs via Data Poisoning","authors":["Zedian Shao","Charles Fleming","Teodora Baluta"],"year":2026,"abstract":"Large language models (LLMs) are often fine-tuned on uncurated text datasets that adversaries can poison. Existing poisoning attacks primarily rely on fixed trigger phrases that defenses such as outlier detection, clean-data regularization, or online monitoring can neutralize. In this paper, we propose a data poisoning method that teaches an LLM an information hiding scheme reliably and stealthily through semantic associations between shared knowledge such as facts or concepts and attacker-chose","url":"https://arxiv.org/abs/2605.26595","categories":["data-poisoning","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00303","type":"paper","title":"Open-Weight LLM Fine-Tuning Defenses are Susceptible to Simple Attacks","authors":["Kevin Kuo","Chhavi Yadav","Virginia Smith"],"year":2026,"abstract":"Recent defenses for safeguarding open-weight large language models (LLMs) are intended to prevent adversarial usage. Underlying these defenses is an assumption that new harmful behavior is learned through fine-tuning rather than elicited by jailbreaking the model. Yet, pretrained LLMs already encode substantial harmful knowledge across many domains, which raises an important question: can an adversary jailbreak safeguarded models, to achieve harmful usage without fine-tuning at all? In this pape","url":"https://arxiv.org/abs/2605.26526","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00304","type":"paper","title":"Evo-Attacker: Memory-Augmented Reinforcement Learning for Long-Horizon Tool Attacks on LLM-MAS","authors":["Bingyu Yan","Xiaoming Zhang","Jinyu Hou","Chaozhuo Li","Ziyi Zhou","Yiming Hei","Litian Zhang"],"year":2026,"abstract":"While Large Language Model-based Multi-Agent Systems (LLM-MAS) demonstrate remarkable capabilities in solving complex tasks by orchestrating specialized agents and external tools, the implicit trust in tool outputs creates a critical attack surface. Existing tool attacks are limited by domain specificity or fixed and static templates. To address these challenges, we propose Evo-Attacker, which formulates the tool attack as a self-evolving, memory-augmented reinforcement learning process. Evo-Att","url":"https://arxiv.org/abs/2605.25389","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00305","type":"paper","title":"Security in the Fine-Tuning Lifecycle of Large Language Models: Threats, Defenses,Evaluation, and Future Directions","authors":["Wenjuan Li","Yitao Liu","Runze Chen","Rajkumar Buyya"],"year":2026,"abstract":"Background: Fine-tuning is central to adapting pre-trained Large Language Models (LLMs) to downstream tasks, but its reliance on training data, parameter updates, and reusable components opens entry points for attackers. Threats have evolved from data poisoning and weight tampering to agent manipulation and interface exploitation, yet existing reviews lack a unified framework spanning the full fine-tuning lifecycle. Objective: This paper presents a systematic survey of LLM fine-tuning security a","url":"https://arxiv.org/abs/2605.25073","categories":["data-poisoning","fine-tuning-security","survey"],"reviewed":false},{"id":"llmsec-2026-00306","type":"paper","title":"Reasoning as an Attack Surface: Adaptive Evolutionary CoT Jailbreaks for LLMs","authors":["Jianan Li","Simeng Qin","Xiaojun Jia","Lionel Z. Wang","Tianhang Zheng","Xiaoshuang Jia","Yang Liu","Xiaochun Cao"],"year":2026,"abstract":"Large Reasoning Models (LRMs) have demonstrated remarkable capabilities in reasoning and generation tasks and are increasingly deployed in real-world applications. However, their explicit chain-of-thought (CoT) mechanism introduces new security risks, making them particularly vulnerable to jailbreak attacks. Existing approaches often rely on static CoT templates to elicit harmful outputs, but such fixed designs suffer from limited diversity, adaptability, and effectiveness. To overcome these lim","url":"https://arxiv.org/abs/2605.24497","categories":["jailbreaking","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00307","type":"paper","title":"Poisoning the Watchtower: Prompt Injection Attacks Against LLM-Augmented Security Operations Through Adversarial Log Content","authors":["Rohan Pandey","Archit Bhujang"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used as analyst assistants in security operations centers (SOCs), where they ingest log and alert data to produce triage labels, incident summaries, or remediation advice. We study a structural failure mode of this design: many log fields are attacker controlled. User agents, URLs, payloads, DNS queries, and attempted usernames can therefore carry instructions to the model alongside evidence of the intrusion. We call this setting \\emph{log-substrate ","url":"https://arxiv.org/abs/2605.24421","categories":["prompt-injection","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00308","type":"paper","title":"Reframing LLM Agent Security as an Agent-Human Interaction Problem","authors":["Peiran Wang","Ying Li","Yuan Tian"],"year":2026,"abstract":"We argue that LLM agent security is fundamentally an agent-human interaction (AHI) problem, not a purely algorithmic one. To substantiate this position, we conduct a systematic analysis of 59 academic papers, 21 production agent systems, and 26 security plugins as of April 2026. Our analysis reveals a striking pattern: the three widely deployed human-centric security mechanisms (policy specification, runtime approval, and scope configuration) dominate industry practice, each adopted by at least ","url":"https://arxiv.org/abs/2605.24309","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00309","type":"paper","title":"An Empirical Evaluation of LLM-Generated Code Security Across Prompting Methods","authors":["Mohammed Kharma","Ahmed Sabbah","Mohammad Alkhanafseh","Mohammad Hammoudeh","David Mohaisen"],"year":2026,"abstract":"The growing use of Large Language Models (LLMs) for automated code generation has enhanced software development efficiency, but often at the cost of security. Generated code frequently overlooks critical concerns, leaving it vulnerable to issues such as weak encryption and improper input validation. To investigate this problem, we present a comprehensive empirical evaluation of the security quality of LLM-generated code across five LLMs and four programming languages (Java, C++, C, and Python), ","url":"https://arxiv.org/abs/2605.24298","categories":["input-filtering"],"reviewed":false},{"id":"llmsec-2026-00310","type":"paper","title":"PromptAudit: Auditing Prompt Sensitivity in LLM-Based Vulnerability Detection","authors":["Steffen J. Camarato","Yahya Hmaiti","Mandana Ghadamian","David Mohaisen"],"year":2026,"abstract":"Large language models are increasingly used for vulnerability detection, yet their reliability under different prompt formulations remains uncharacterized. We present PromptAudit, a controlled evaluation framework that isolates prompt effects by fixing the dataset, decoding, and parsing while varying only the prompting strategy. Using five prompting strategies across five open-weight models on 1,000 CVEs (6,074 code samples spanning 16 programming languages), we evaluate accuracy, recall, absten","url":"https://arxiv.org/abs/2605.24171","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00311","type":"paper","title":"When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents","authors":["Shi Liu","Xuehai Tang","Xikang Yang","Liang Lin","Biyu Zhou","Wenjie Xiao","Wantao Liu"],"year":2026,"abstract":"The rise of tool-using Large Language Model (LLM) agents, standardized by protocols like the Model Context Protocol (MCP), has unlocked unprecedented autonomous execution capabilities for LLM Agents by integrating external open-domain knowledge and tools. However, this interoperability introduces a covert attack surface targeting the agent's cognitive planning layer. This paper systematically investigates Tool Description Poisoning (TDP), a novel semantic attack. In TDP, malicious instructions a","url":"https://arxiv.org/abs/2605.24069","categories":["data-poisoning","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00312","type":"paper","title":"Are Frontier LLMs Ready for Cybersecurity? Evidence for Vertical Foundation Models from Dual-Mode Vulnerability Benchmarks","authors":["Vivek Dahiya","Sunny Nehra","Vipul Dholariya","Bhavik Shangari","Chandra Khatri"],"year":2026,"abstract":"We evaluate whether frontier LLMs are ready for cybersecurity through a dual-mode benchmark: white-box function-level vulnerability detection (VulnLLM-R, across C/Java/Python) and black-box web application security testing (five production-style applications with 118 ground-truth vulnerabilities across 20+ CWE families, which we will open-source). We test six frontier models (GPT-5.4, Codex~5.3, Claude Opus~4.6, Sonnet~4.6, Gemini~3.1~Pro and Gemini~3~Flash) and two domain-specialized models acr","url":"https://arxiv.org/abs/2605.23243","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00313","type":"paper","title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","authors":["Ziyi Tong","Feifei Sun","Le Minh Nguyen"],"year":2026,"abstract":"Large Language Models (LLMs) have become the predominant paradigm in NLP, advancing both research and industry. As model sizes and pretraining data grow, concerns about Pretraining Data Exposure (PDE) increase due to the scale and opacity of training datasets. PDE refers to determining whether specific data appeared in an LLM's pretraining corpus. It is critical for ensuring evaluation integrity and protecting privacy, intersecting two key areas: data contamination and membership inference. Thou","url":"https://arxiv.org/abs/2605.26133","categories":["membership-inference","survey"],"reviewed":false},{"id":"llmsec-2026-00314","type":"paper","title":"Blind Spots in the Guard: How Domain-Camouflaged Injection Attacks Evade Detection in Multi-Agent LLM Systems","authors":["Aaditya Pai"],"year":2026,"abstract":"Injection detectors deployed to protect LLM agents are calibrated on static, template-based payloads that announce themselves as override directives. We identify a systematic blind spot: when payloads are generated to mimic the domain vocabulary and authority structures of the target document, what we call domain camouflaged injection, standard detectors fail to flag them, with detection rates dropping from 93.8% to 9.7% on Llama 3.1 8B and from 100% to 55.6% on Gemini 2.0 Flash. We formalize th","url":"https://arxiv.org/abs/2605.22001","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00315","type":"paper","title":"FuzzingBrain V2: A Multi-Agent LLM System for Automated Vulnerability Discovery and Reproduction","authors":["Ze Sheng","Zhicheng Chen","Qingxiao Xu","Kewen Zhu","Jeff Huang"],"year":2026,"abstract":"Software vulnerabilities pose critical security threats, with nearly 50,000 CVEs reported in 2025. While Large Language Models (LLMs) show promise for automated vulnerability detection, three key challenges remain. First, LLM-generated vulnerability reports suffer from high false positive rates and lack reproducible verification. Second, existing LLM-based approaches use suboptimal granularities for vulnerability localization: function-level analysis overlooks bugs when context becomes extensive","url":"https://arxiv.org/abs/2605.21779","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00316","type":"paper","title":"Trusted Weights, Treacherous Optimizations? Optimization-Triggered Backdoor Attacks on LLMs","authors":["Yifei Wang","Tianlin Li","Xiaohan Zhang","Yida Yang","Xiaoyu Zhang","Li Pan"],"year":2026,"abstract":"Inference optimization is a vital technique for deploying LLMs at scale. Compilation is the most widely adopted optimization technique for LLMs. While it assumes semantic equivalence between the original and compiled graphs, we first uncover its numerical side effects can be maliciously exploited to implant stealthy backdoors in LLMs. We propose a unified optimization-triggered attack framework comprising two complementary strategies. Without any modification to the compiler or hardware, one str","url":"https://arxiv.org/abs/2605.20641","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00317","type":"paper","title":"Security Document Classification with a Fine-Tuned Local Large Language Model: Benchmark Data and an Open-Source System","authors":["Ivan Dobrovolskyi"],"year":2026,"abstract":"Organizations that scan documents for sensitive information face a practical problem. Cloud services require data to be sent to external infrastructure, while rule-based tools often miss threats that depend on context. This study presents TorchSight, an open-source local system for security document classification built around a fine-tuned Qwen 3.5 27B model. The model was trained on 78,358 samples from 13 permissively licensed sources and GPT-4 synthetic data covering seven security categories ","url":"https://arxiv.org/abs/2605.20368","categories":["membership-inference","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00318","type":"paper","title":"CASPIAN: Online Detection and Attribution of Cascade Attacks in LLM Multi-Agent Systems via Cross-Channel Causal Monitoring","authors":["Kavana Venkatesh","Jafar Isbarov","Saad Amin","Murat Kantarcioglu","Jiaming Cui"],"year":2026,"abstract":"Cascade attacks in LLM multi-agent systems (MAS) arise when adversarial influence propagates across agents and leads to escalated system-level failures through complex agent interactions. Detecting such cascades is challenging, as their signals are distributed, tightly coupled across interaction channels, and often appear plausibly benign locally but may unfold quickly either within a single turn or gradually across multiple turns. Existing defenses, being largely local and text-centric, fail to","url":"https://arxiv.org/abs/2605.19240","categories":["monitoring-detection","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00319","type":"paper","title":"Be Kind, Rewrite: Benign Projections via Rewriting Defend Against LLM Data Poisoning Attacks","authors":["John T. Halloran","Noopur S. Bhatt"],"year":2026,"abstract":"Large language models (LLMs) are highly susceptible to backdoor attacks (BAs), wherein training samples are poisoned using trigger-based harmful content. Furthermore, existing defenses have proven ineffective when extensively tested across BA patterns. To better combat BAs, we explore the use of LLM rewriting as a proactive defense against data poisoning. First, we theoretically show that when LLM rewriting utilizes open-book benign samples--termed open-book benign rewriting (OBBR)--the probabil","url":"https://arxiv.org/abs/2605.19147","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00320","type":"paper","title":"ASPI: Seeking Ambiguity Clarification Amplifies Prompt Injection Vulnerability in LLM Agents","authors":["Udari Madhushani Sehwag","Zhengyang Shan","Heming Liu","Dileepa Lakshan","Joseph Brandifino","Max Fenkell"],"year":2026,"abstract":"Clarification-seeking behavior is widely regarded as a desirable property of LLM agents, enabling them to resolve ambiguity before acting on underspecified tasks. However, the security implications of this interaction pattern remain unexplored. We investigate whether the transition from standard execution to a clarification-seeking state increases an agent's susceptibility to prompt injection attacks. We introduce ASPI (Ambiguous-State Prompt Injection), a benchmark of 728 task-attack scenarios ","url":"https://arxiv.org/abs/2605.17324","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00321","type":"paper","title":"When Efficiency Backfires: Cascading LLMs Trigger Cascade Failure under Adversarial Attack","authors":["Zehan Sun","Dingfan Chen","Songze Li"],"year":2026,"abstract":"Large Language Model (LLM) cascade systems are designed to balance efficiency and performance by processing queries with lightweight models while selectively escalating complex cases to more powerful ones. Such systems seek to reduces computational cost and latency while maintaining task performance, making it an appealing choice for large-scale deployment. However, the cascade design introduces new vulnerabilities through an expanded attack surface: the inclusion of lightweight front-end models","url":"https://arxiv.org/abs/2605.17288","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00322","type":"paper","title":"DarkLLM: Learning Language-Driven Adversarial Attacks with Large Language Models","authors":["Ye Sun","Xin Wang","Jiaming Zhang","Yifeng Gao","Yixu Wang","Yifan Ding","Qixian Zhang","Henghui Ding","Xingjun Ma","Yu-Gang Jiang"],"year":2026,"abstract":"While vision and multimodal foundation models underpin critical tasks from perception to complex reasoning, they remain highly vulnerable to adversarial attacks. However, traditional adversarial attacks are typically limited to single, predefined objectives, tightly coupling each attack to a specific model or task, which restricts their scalability and flexibility in real-world scenarios. In this work, we present DarkLLM, a novel attack framework that trains an LLM to translate natural-language ","url":"https://arxiv.org/abs/2605.18868","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00323","type":"paper","title":"Token-Level Generalization in LoRA Adapter Backdoors: Attack Characterization and Behavioral Detection","authors":["Travis Lelle"],"year":2026,"abstract":"We show that LoRA adapters, the dominant distribution format for fine-tuned LLMs, can be reliably backdoored through training data poisoning while preserving baseline task performance. On a Qwen 2.5 1.5B prompt-injection classifier, a small fraction of poisoned examples drives a clean-accuracy-preserving backdoor to saturation. The resulting backdoor generalizes at the token feature level rather than the structural pattern level: a model trained on one RFC reference activates on any RFC referenc","url":"https://arxiv.org/abs/2605.30189","categories":["prompt-injection","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00324","type":"paper","title":"Token Inflation: How Dishonest Providers Can Overcharge for Large Language Model Usage","authors":["Shahinul Hoque","Jinghuai Zhang","Jinyuan Sun","Fnu Suya"],"year":2026,"abstract":"Per-token billing is now the standard pricing model for commercial large language models (LLMs), so the honesty of reported token counts directly affects what users pay. We show that this kind of billing is hard to audit by design: providers hide the model, the tokenizer, and the execution to protect their IP, mitigate jailbreaks, and preserve user privacy, which means an auditor can only inspect proofs the provider supplies. The audit therefore reduces to a consistency check on the provider's o","url":"https://arxiv.org/abs/2605.30040","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00325","type":"paper","title":"AIRGuard: Guarding Agent Actions with Runtime Authority Control","authors":["Suliu Qin","Haomin Zhuang","Yujun Zhou","Yufei Han","Xiangliang Zhang"],"year":2026,"abstract":"Tool-using language agents turn model decisions into external side effects: they read files, run scripts, call APIs, send messages, and invoke Model Context Protocol tools. This makes agent attacks different from jailbreaks. The harmful step is often not an obviously forbidden output, but an ordinary executable action that becomes unsafe because attacker-controlled context steers authorized access against the user's interest. We identify this failure mode as authority confusion: untrusted resour","url":"https://arxiv.org/abs/2605.28914","categories":["jailbreaking","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00326","type":"paper","title":"SNARE: Adaptive Scenario Synthesis for Eliciting Overeager Behavior in Coding Agents","authors":["Yubin Qu","Yi Liu","Gelei Deng","Yanjun Zhang","Yuekang Li","Ying Zhang","Leo Yu Zhang"],"year":2026,"abstract":"A coding agent executes a benign task as a sequence of shell, file, and network actions, any of which can quietly exceed the authorized scope while the task still completes. We call this overeager behavior: the prompt is not adversarial and the run succeeds, yet an out-of-scope step can leak credentials or delete files. Existing benchmarks miss it: task-completion suites credit any finished run, jailbreak suites probe adversarial prompts, and the one prior overeager benchmark applies a single fi","url":"https://arxiv.org/abs/2605.28122","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00327","type":"paper","title":"MIRAGE: Context-Aware Prompt Injection against Mobile GUI Agents via User-Generated Content","authors":["Ruoqi Guo","Yi Liu","Gelei Deng","Yiheng Xiong","Yuekang Li","Ying Zhang","Leo Yu Zhang","Lida Zhao","Ji Jie","Yuxiao Lu"],"year":2026,"abstract":"Mobile graphical user interface (GUI) agents driven by vision-language models (VLMs) perceive the screen as rendered pixels and choose actions from what they see, so they cannot reliably separate trusted interface elements from user-generated content. We present MIRAGE (Mobile Injection of Realistic Adversarial GUI Examples), a pipeline that turns benign mobile screenshots into prompt-injection samples by placing attacker-controlled text into ordinary user-generated content regions, without modi","url":"https://arxiv.org/abs/2605.28116","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00328","type":"paper","title":"When Think-with-Image Meets Safety: What Determines Multimodal Jailbreak Robustness?","authors":["Yuan Tian","Bing Hu","Fang Wu","Xiaomin Li","Binghang Lu","Neil Zhenqiang Gong"],"year":2026,"abstract":"Think-with-image reasoning is emerging as a new inference paradigm for large vision-language models, but its safety implications remain poorly understood. Existing systems already span multiple process designs, including direct response generation, text-only prior turn, visual-state manipulation, and explicit external image-tool invocation. In this paper, we ask which of these evaluated paradigms improves multimodal jailbreak robustness, and why. Across multiple vision-language models, explicit ","url":"https://arxiv.org/abs/2605.27932","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00329","type":"paper","title":"BAIT: Boundary-Guided Disclosure Escalation via Self-Conditioned Reasoning","authors":["Xuan Luo","Yue Wang","Geng Tu","Jing Li","Ruifeng Xu"],"year":2026,"abstract":"In this work, we propose BAIT (Boundary-Aware Iterative Trap), a three-step jailbreak framework that approaches malicious goals through internal disclosure. BAIT first asks the model to identify the protection boundary, then requires it to refine that boundary, and finally requests a detailed example. By expanding each step upon the model's previous responses, BAIT turns the model's own reasoning and consistency tendency into a disclosure pathway. Experiments on AdvBench, JailbreakBench, AIR-Ben","url":"https://arxiv.org/abs/2605.27110","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00330","type":"paper","title":"Prompt Injection Detection is Regime-Dependent: A Deployment-Aware Evaluation with Interpretable Structural Signals","authors":["Akindoyin Akinrele","Shreyank N Gowda"],"year":2026,"abstract":"Prompt injection poses a critical threat to the safe deployment of large language models, yet existing detection approaches are typically evaluated under limited settings that do not reflect real-world operating constraints. In this work, we present a deployment-aware evaluation of prompt injection detection using a multi-model and multi-regime experimental framework. We compare lexical, semantic, structural, and transformer-based detectors across multiple out-of-distribution settings, repeated ","url":"https://arxiv.org/abs/2605.26999","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00331","type":"paper","title":"Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents","authors":["Peiran Wang","Ying Li","Yuan Tian"],"year":2026,"abstract":"LLM-based agents are increasingly deployed in high-stakes scenarios such as email management, financial transactions, and code execution, where they interact with the external world through tool calling. During execution, these agents must read external data sources (emails, webpages, files) that attackers can control; through indirect prompt injection, attackers embed malicious instructions in this data to manipulate agents into performing unauthorized operations such as transferring funds to a","url":"https://arxiv.org/abs/2605.26497","categories":["prompt-injection","agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00332","type":"paper","title":"Jailbreak susceptibility prediction and mitigation via the behavioral geometry of models","authors":["Hayden Helm","Xiaodong Liu","Weiwei Yang"],"year":2026,"abstract":"Evaluating and mitigating a generative system's susceptibility to jailbreak attacks is critical to its safe deployment. Given the number of deployable systems, full per-configuration evaluation and optimization is impractical. In this paper, we formalize the behavioral geometry of a population of models that, by leveraging previously evaluated and defended models, supports both efficient susceptibility prediction and effective defense transfer across a population. We apply the framework to 79 mo","url":"https://arxiv.org/abs/2605.26409","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00333","type":"paper","title":"How Agentic AI Coding Assistants Become the Attacker's Shell","authors":["Yue Liu","Yanjie Zhao","Yunbo Lyu","Ting Zhang","Haoyu Wang","David Lo"],"year":2026,"abstract":"Agentic AI coding assistants can edit files, run commands, and access the internet on behalf of developers. However, their reliance on unvetted external artifacts introduces a new attack vector. Hidden instructions in external artifacts can hijack these assistants, turning them into an attacker's shell to run unauthorized commands. In this article, we examine how these prompt injection attacks work, measure their prevalence, discuss the limitations and challenges of current defenses, and suggest","url":"https://arxiv.org/abs/2605.25871","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00334","type":"paper","title":"Furina: Fragmented Uncertainty-Driven Refusal Instability Attack","authors":["Tongxi Wu","Jian Zhang","Yang Gao"],"year":2026,"abstract":"Safety alignment in large language models (LLMs) and multimodal large language models (MLLMs) is commonly assumed to operate as a near-binary threshold mechanism. We challenge this assumption by revealing that safety behavior is governed by an instability region where small perturbations induce stochastic refusal decisions rather than deterministic outcomes. We develop a multi-metric diagnostic framework combining external and internal signals to characterize this instability. Through systematic","url":"https://arxiv.org/abs/2605.26158","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00335","type":"paper","title":"Reflect-Guard: Enhancing LLM Safeguards against Adversarial Prompts via Logical Self-Reflection","authors":["Lixing Lin","Juli You","Yue Li","Luyun Lin","Yiqing Wang","Zhen Zhang","Moxuan Zheng"],"year":2026,"abstract":"Large language model (LLM) safety classifiers such as Llama Guard are effective at detecting overtly harmful prompts but remain vulnerable to adversarial jailbreak attacks that disguise malicious intent through role-play scenarios, fictional framing, and indirect requests. We present Reflect-Guard, a method that augments LLM-based safety classifiers with chain-of-thought self-reflection capabilities through parameter-efficient fine-tuning. Our approach distills analytical reasoning from GPT-4o-m","url":"https://arxiv.org/abs/2605.24834","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00336","type":"paper","title":"Ellipsoid Control: A White-list Jailbreak Defense via Benign Latent Modeling","authors":["Luoyu Chen","Weiqi Wang","Zhiyi Tian","Feng Wu","Ahmed Asiri","Shui Yu"],"year":2026,"abstract":"Representation engineering (RepE) defenses have shown strong robustness against jailbreak attacks on large language models (LLMs). However, these methods fundamentally rely on black-list supervision: they learn jailbreak-to-refusal activation transformations from harmful or jailbreak data that are inherently incomplete and continuously evolving. Hence, the performance of RepE-based defenses becomes tightly coupled to the quality and coverage of collected harmful samples, leaving models vulnerabl","url":"https://arxiv.org/abs/2605.24552","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00337","type":"paper","title":"Steering Beyond the Support: Adversarial Training on Unsupervised Jailbroken Activation Simulation","authors":["Luoyu Chen","Weiqi Wang","Zhiyi Tian","Chenhan Zhang","Feng Wu","Jianhuan Huang","Ahmed Asiri","Shui Yu"],"year":2026,"abstract":"Jailbreak prompts can trigger harmful completions on aligned LLMs, In accordance, safety steering has been proposed: test-time activation interventions that steer jailbreak activations to trigger refusal while preserving benign utility. However, existing steering methods are fundamentally supervised and tied to a static, limited training set, whereas real jailbreaks evolve and are often out-of-distributed from the training set, leading to failures on unseen attacks. In this paper, we tackle the ","url":"https://arxiv.org/abs/2605.24535","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00338","type":"paper","title":"Prompt Overflow: What the Guardrail Inspects Is Not What the Model Infers","authors":["Yuanbo Zhou","Changjia Zhu","Junyu Wang","Xu He","Yan Zhai","Kun Sun","Mingkui Wei","Junjie Xiong"],"year":2026,"abstract":"Guardrail models (a.k.a. safety checkers) are widely deployed to screen user inputs before they reach large language models (LLMs), serving as a primary defense against prompt injection attacks. Due to strict context constraints, these models handle overlength prompts through truncation or segmentation-based inspection. While prior work has focused on semantic adversarial inputs, the security implications of these long-input processing mechanisms remain largely unexplored. In this paper, we iden","url":"https://arxiv.org/abs/2605.23196","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00339","type":"paper","title":"Adversarial Reframing: A Framework for Targeted Generation in Language Models","authors":["Shahnewaz Karim Sakib","Swati Kar","Anindya Bijoy Das"],"year":2026,"abstract":"Large Language Models (LLMs) are widely deployed in diverse real-world settings, yet remain vulnerable to jailbreaking, where prompt-based attacks bypass safety filters. We present THREAT (Targeted Harmful generation via Reframing and Exploitation of Adversarial Tactics), a reasoning-driven framework that coordinates multiple LLMs in an iterative search loop to find textual jailbreak prompts. We formulate prompt discovery as a nonconvex optimization problem and provide an efficient solution that","url":"https://arxiv.org/abs/2605.21674","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00340","type":"paper","title":"Refusal Evaluation in Coding LLMs and Code Agents: A Systematic Review of Thirteen Malicious-Code Prompt Corpora (2023-2025)","authors":["Richard J. Young","Gregory D. Moody"],"year":2026,"abstract":"The evaluation of large language model refusal on malicious-coding tasks now spans at least thirteen publicly released prompt corpora (AdvBench, the CyberSecEval family, RMCBench, RedCode, MCGMark, JailbreakBench, CySecBench, MalwareBench, CIRCLE, MOCHA, ASTRA, Scam2Prompt / Innoc2Scam-bench, and JAWS-Bench), each constructed under a different protocol, released under different licensing terms, and validated (or not) against different inter-rater reliability standards. Existing surveys treat cod","url":"https://arxiv.org/abs/2605.20351","categories":["jailbreaking","survey"],"reviewed":false},{"id":"llmsec-2026-00341","type":"paper","title":"Adaptive Probe-based Steering for Robust LLM Jailbreaking","authors":["Junxi Chen","Junhao Dong","Xiaohua Xie"],"year":2026,"abstract":"Recent work has demonstrated the potential of contrastive steering for jailbreaking Large Language Models (LLMs). However, existing methods rely on limited and inherently biased contrastive prompts and require laborious manual tuning of steering strength, limiting their robustness and effectiveness. In this paper, we leverage the idea of model extraction to guide the learned steering vectors to approximate the ideal one and propose tuning the steering strength adaptively based on contrastive act","url":"https://arxiv.org/abs/2605.20286","categories":["jailbreaking","model-extraction"],"reviewed":false},{"id":"llmsec-2026-00342","type":"paper","title":"RoboJailBench: Benchmarking Adversarial Attacks and Defenses in Embodied Robotic Agents","authors":["Doguhuan Yeke","Yanming Zhou","Leo Y. Lin","Hongyu Cai","Antonio Bianchi","Z. Berkay Celik"],"year":2026,"abstract":"Recent advances in Vision-Language Models (VLMs) facilitate a new class of embodied AI systems, where these models are integrated into physical platforms, e.g. robots and autonomous vehicles, to interpret visual scenes and execute natural language commands in diverse environments. Previous research has introduced jailbreak attacks and defenses for embodied AI. Their evaluations, however, rely on ad-hoc datasets, limited metrics, and emphasize attack success while neglecting the trade-off between","url":"https://arxiv.org/abs/2605.19328","categories":["jailbreaking","adversarial-examples","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00343","type":"paper","title":"Exploring and Developing a Pre-Model Safeguard with Draft Models","authors":["Hongyu Cai","Arjun Arunasalam","Yiming Liang","Antonio Bianchi","Z. Berkay Celik"],"year":2026,"abstract":"Large Language Model (LLM) alignment remains vulnerable to jailbreak attacks that elicit unsafe responses, motivating pre-model and post-model guards. Pre-model guards audit the safety of prompts before invoking target models. However, relying solely on the prompt often leads to high false-negative rates (i.e., jailbreak attacks go undetected). Post-model guards address this issue by auditing both the user prompt and the target model's response. However, they incur a high computational cost, inc","url":"https://arxiv.org/abs/2605.19321","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00344","type":"paper","title":"On the Geometric Limits of Transformer Defenses against Obfuscation Attacks: Latent Embedding Collapse & Performance Robustness Gap","authors":["Becky Mashaido","Tapadhir Das"],"year":2026,"abstract":"Prompt injection attacks pose significant risks to language model safety, yet existing defenses are typically evaluated using classification performance. We show that high detection performance does not imply representational robustness. Specifically, multi-operator obfuscated prompts (combining homoglyphs, zero-width characters, and punctuation or emoji noise) can partially collapse onto the embedding manifold of clean prompts, a phenomenon we term latent embedding collapse. Results indicate th","url":"https://arxiv.org/abs/2605.19159","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00345","type":"paper","title":"Overeager Coding Agents: Measuring Out-of-Scope Actions on Benign Tasks","authors":["Yubin Qu","Ying Zhang","Yanjun Zhang","Gelei Deng","Yuekang Li","Leo Yu Zhang","Yi Liu"],"year":2026,"abstract":"Coding agents now run autonomously with shell, file, and network privileges. When a user issues a benign request, the agent sometimes does more than asked: it deletes unrelated files, wipes a stale credentials backup, or rewrites configuration the user never mentioned. We call these scope expansions overeager actions, an authorization problem distinct from capability failures, prompt injection, or sandbox escapes. We present OverEager-Gen, a benchmark dedicated to overeager behavior on benign ta","url":"https://arxiv.org/abs/2605.18583","categories":["prompt-injection","access-control","sandboxing-isolation","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00346","type":"paper","title":"Acoustic Interference: A New Paradigm Weaponizing Acoustic Latent Semantic for Universal Jailbreak against Large Audio Language Models","authors":["Yanyun Wang","Yu Huang","Zi Liang","Xixin Wu","Li Liu"],"year":2026,"abstract":"The integration of audio modality into Large Audio Language Models (LALMs) significantly expands their attack surface. Existing jailbreak paradigms predominantly treat audio as a carrier for malicious payloads, relying on semantic optimization, acoustic parameter control, or additive perturbation to embed harmful content into the audio signal. In this work, we challenge this necessity and propose a new paradigm in which the role of audio shifts from content injection to safety alignment interfer","url":"https://arxiv.org/abs/2605.18168","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00347","type":"paper","title":"An Empirical Study of Privacy Leakage Chains via Prompt Injection in Black-Box Chatbot Environments","authors":["Hongjang Yang","Hyunsik Na","Daeseon Choi"],"year":2026,"abstract":"LLM-based chatbot agents increasingly process user requests by combining natural-language reasoning with external tools such as web browsing. These capabilities improve usability, but they also create attack surfaces when untrusted external content is processed as part of a user' s task. This paper studies a privacy-leakage attack chain based on indirect prompt injection in black-box chatbot environments, where the attacker has no access to model weights, system prompts, or agent implementation ","url":"https://arxiv.org/abs/2605.18133","categories":["prompt-injection","membership-inference"],"reviewed":false},{"id":"llmsec-2026-00348","type":"paper","title":"Babel: Jailbreaking Safety Attention via Obfuscation Distribution Optimized Sampling","authors":["Ziwei Wang","Jing Chen","Ruichao Liang","Zhi Wang","Yebo Feng","Ju Jia","Ruiying Du","Cong Wu","Yang Liu"],"year":2026,"abstract":"Despite rigorous safety alignment, Large Language Models (LLMs) remain vulnerable to jailbreak attacks. Existing black-box methods often rely on heuristic templates or exhaustive trials, lacking mechanistic interpretability and query efficiency. In this study, we investigate an intrinsic vulnerability in the safety mechanisms of LLMs, where safety alignment relies on a small set of sparsely distributed attention heads, leaving much of the representational space weakly monitored. We formalize thi","url":"https://arxiv.org/abs/2605.17971","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00349","type":"paper","title":"ESLD (External Surrogate Latent Defense): A Latent-Space Architecture for Faster, Stronger Prompt-Injection Defense","authors":["Yash Narendra"],"year":2026,"abstract":"Modern AI assistants are agentic. To answer a single user request, the underlying language model pulls in information from many sources, such as web searches, retrieved documents, tool outputs, and user follow-ups, and reasons over them across several steps. Any of these inputs can carry malicious content. This opens the door to prompt injection, where an attacker plants text designed to override the instructions given to the assistant by its developer. For example, an attacker applying for a jo","url":"https://arxiv.org/abs/2605.18918","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00350","type":"paper","title":"DMN: A Compositional Framework for Jailbreaking Multimodal LLMs with Multi-Image Inputs","authors":["Wenzhuo Xu","Zhipeng Wei","Zonghao Ying","Deyue Zhang","Dongdong Yang","Xiangzheng Zhang","Quanchen Zou"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) are vulnerable to jailbreak attacks, which can elicit harmful responses from MLLMs. Many MLLMs support multi-image inputs, inadvertently introducing new vulnerabilities due to less efforts on multi-image safety alignment. Previous MLLM jailbreak methods only uses a single image, which restricts the attack space: they cannot distribute harmful requests across multiple images, carry abundant information, or exploit additional visual reasoning tasks to distr","url":"https://arxiv.org/abs/2605.18915","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00351","type":"paper","title":"AI Agents May Always Fall for Prompt Injections","authors":["Sahar Abdelnabi","Eugene Bagdasarian"],"year":2026,"abstract":"Prompt injection is the most critical vulnerability in deployed AI agents. Despite recent progress, we show that the prevailing defense paradigm (data-instruction separation) both fails to detect attacks that operate through contextual manipulation and degrades contextually appropriate behavior. We then recast prompt injection via the lens of Contextual Integrity (CI), a privacy theory that judges information flow compliance with contextual norms. This explains types of attacks that current defe","url":"https://arxiv.org/abs/2605.17634","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00352","type":"paper","title":"ADR: An Agentic Detection System for Enterprise Agentic AI Security","authors":["Chenning Li","Pan Hu","Justin Xu","Baris Ozbas","Olivia Liu","Caroline Van","Manxue Li","Wei Zhou","Mohammad Alizadeh","Pengyu Zhang","KK Sriramadhesikan","Ming Zhang"],"year":2026,"abstract":"We present the Agentic AI Detection and Response (ADR) system, the first large-scale, production-proven enterprise framework for securing AI agents operating through the Model Context Protocol (MCP). We identify three persistent challenges in this domain: (1) limited observability -- existing Endpoint Detection and Response (EDR) tools see file writes but not the agent reasoning, prompts, or causal chains linking intent to execution; (2) insufficient robustness -- static defenses constrained by ","url":"https://arxiv.org/abs/2605.17380","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00353","type":"paper","title":"STRIDE-AI: A Threat Modeling Framework for Generative AI Security Assessment","authors":["Tsafac Nkombong Regine Cyrille","Franziska Schwarz"],"year":2026,"abstract":"Traditional cybersecurity methodologies target deterministic systems and fail to address the probabilistic nature of AI, leaving systems vulnerable to attack vectors such as model inversion, data poisoning, and prompt injection. Recent industry reports indicate that a majority of organizations deploying AI lack a dedicated security strategy, with adversarial attacks increasing rapidly year-over-year. We present \\textit{STRIDE-AI}, a framework that bridges the gap between high-level risk standard","url":"https://arxiv.org/abs/2605.17163","categories":["prompt-injection","data-poisoning","adversarial-examples","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00354","type":"paper","title":"New Wide-Net-Casting Jailbreak Attacks Risk Large Models","authors":["Qiuchi Xiang","Haoxuan Qu","Hossein Rahmani","Jun Liu"],"year":2026,"abstract":"Jailbreak attacks on large models have drawn growing attention due to their close ties to societal safety. This work identifies a practical yet unexplored jailbreak scenario, the wide-net-casting scenario, where an adversary can query a group of large models instead of a single one to elicit harmful outputs. Our analysis reveals substantial yet previously overlooked safety risks under this scenario. As a key part of our analysis, we further develop a novel jailbreak method tailored to the wide-n","url":"https://arxiv.org/abs/2605.17128","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00355","type":"paper","title":"A Cross-Modal Prompt Injection Attack against Large Vision-Language Models with Image-Only Perturbation","authors":["Hao Yang","Zhuo Ma","Yang Liu","Yilong Yang","Guancheng Wang","JianFeng Ma"],"year":2026,"abstract":"Large vision-language models (LVLMs) have emerged as a powerful paradigm for multimodal intelligence, but their growing deployment also expands the attack surface of prompt injection. Despite this growing concern, existing attacks still suffer from a critical limitation: the injected prompt for one modality only steers the model's interpretation of that singular input. Alternatively, these attacks remain multimodal but fail to achieve cross-modal prompt perturbation. To bridge this gap, we intro","url":"https://arxiv.org/abs/2605.16090","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00356","type":"paper","title":"Compositional Jailbreaking: An Empirical Analysis of Mutator Chain Interactions in Aligned LLMs","authors":["Reinelle Jan Bugnot","Soohyeon Choi","Hoon Wei Lim","Yue Duan"],"year":2026,"abstract":"Jailbreaking attacks on large language models pose a significant threat to AI safety by enabling the generation of harmful or restricted content. While prior work has explored both handcrafted and automated jailbreak strategies, the potential for compositional interaction between simple attacks remains underexplored. This paper presents a systematic study of mutator chaining, in which weak jailbreak transformations are applied sequentially to characterize how they interact: whether they reinforc","url":"https://arxiv.org/abs/2605.15598","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00357","type":"paper","title":"Hidden in Memory: Sleeper Memory Poisoning in LLM Agents","authors":["Sidharth Pulipaka","Stanislau Hlebik","Leonidas Raghav","Sahar Abdelnabi","Vyas Raina","Ivaxi Sheth","Mario Fritz"],"year":2026,"abstract":"Large language models are increasingly augmented with persistent memory, allowing assistants to store user-specific information across sessions for personalization and continuity. This statefulness introduces a new security risk: adversarial content can corrupt what an assistant remembers and thereby influence future interactions. We propose and study sleeper memory poisoning, a delayed attack in which an adversary manipulates external context, such as a document, webpage, or repository, to caus","url":"https://arxiv.org/abs/2605.15338","categories":["data-poisoning","agentic-threats","memory-security","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00358","type":"paper","title":"AI Security Research Should Better Incentivize Defense Research","authors":["Youqian Zhang"],"year":2026,"abstract":"This work examines an imbalance in artificial intelligence (AI) security research: the field tends to produce more work on attacking AI systems than on defending them. Drawing on related academic papers, we find biased attack-to-defense ratios across subfields, including federated learning, speech recognition, membership inference, large language models, etc. The imbalance possibly means far beyond a simple count: attack papers are routinely evaluated under favorable conditions that make threats","url":"https://arxiv.org/abs/2605.23448","categories":["membership-inference","federated-learning"],"reviewed":false},{"id":"llmsec-2026-00359","type":"paper","title":"Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction","authors":["Hongtao Wang","Se Yang","Yu Chen","Puzhuo Liu"],"year":2026,"abstract":"Large language model (LLM) agents increasingly leverage long term memory to support persistent and autonomous task execution. However, this capability also introduces a new attack surface: memory poisoning, where adversaries can inject malicious information to influence future behavior. Existing memory poisoning attacks often assume that injected content can be stored directly in memory, overlooking the selective extraction and rewriting stages in modern memory pipelines. This makes prior method","url":"https://arxiv.org/abs/2605.29960","categories":["data-poisoning","memory-security"],"reviewed":false},{"id":"llmsec-2026-00360","type":"paper","title":"GradSentry: Gradient Spectral Entropy for Backdoor Sample Filtering in Large Language Model Fine-Tuning","authors":["Haodong Zhao","Tianyi Xu","Tianhang Zhao","Zhuosheng Zhang","Gongshen Liu"],"year":2026,"abstract":"Fine-tuning Large Language Models with untrusted data exposes models to backdoor attacks, where poisoned samples cause targeted misbehavior. Existing sample-filtering defenses rely on clustering, which requires sufficient data and can fail at extreme poison ratios. We propose GradSentry ({Grad}ient {Sentry}), a backdoor sample filtering method based on the spectral entropy of per-sample gradients. Our key finding is that poisoned samples produce gradients with higher spectral entropy compared to","url":"https://arxiv.org/abs/2605.26574","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00361","type":"paper","title":"Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures","authors":["Yuntao Wang","Jianle Ba","Han Liu","Yanghe Pan","Jintao Wei","Zhou Su","Tom H. Luan","Linkang Du"],"year":2026,"abstract":"The rapid evolution of large language model (LLM)-driven autonomous agents has given rise to OpenClaw, a new class of open-source agent frameworks that operate as continuously running, skill-augmented systems with persistent memory, multi-channel interaction, and high degrees of autonomy. Such capabilities enable OpenClaw agents to autonomously execute complex, multi-step tasks and interact seamlessly with external applications, but simultaneously introduce a substantially enlarged attack surfac","url":"https://arxiv.org/abs/2605.25435","categories":["autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00362","type":"paper","title":"Backdooring Masked Diffusion Language Models","authors":["Daniel Yiming Cao","Chengzhong Wang","Sheng-Yen Chou","Chengyu Huang","Pin-Yu Chen","Shengwei An"],"year":2026,"abstract":"Masked diffusion language models (MDLMs) are emerging as a compelling new paradigm for text generation, but their training-time security remains largely unexplored. Existing backdoor attacks on Gaussian diffusion models or autoregressive language models do not directly apply to MDLMs because MDLMs rely on discrete state corruption and iterative denoising rather than continuous noising or left-to-right prediction. In this work, we present the first systematic study of training-time backdoor attac","url":"https://arxiv.org/abs/2605.19262","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00363","type":"paper","title":"Surviving the Unseen: Predictive Defense for Novel Multi-Turn Multimodal Attacks","authors":["Doohee You"],"year":2026,"abstract":"The expansion of Multimodal Large Language Models (MLLMs) and their integration into autonomous agentic workflows has introduced a non-stationary attack surface. Empirical observations indicate that adversaries employ progressive, cross-modal perturbations that evade turn-specific guardrails by distributing malicious intent across longitudinal conversational trajectories. Static defense mechanisms, constrained by the Markov property, evaluate inputs in isolation and fail to detect cumulative str","url":"https://arxiv.org/abs/2605.18988","categories":["agentic-threats","guardrails","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00364","type":"paper","title":"Not What You Asked For: Typographic Attacks in Household Robot Manipulation","authors":["Ali Iranmanesh","Peng Liu"],"year":2026,"abstract":"Open-vocabulary embodied AI agents increasingly rely on vision-language models such as CLIP for object perception and task grounding. However, the shared embedding space that enables this flexibility introduces a structural vulnerability to typographic attacks, where printed text in a physical scene semantically overrides visual judgment. While prior work has quantified this threat in static 2D benchmarks and 3D navigation tasks, its impact on the full Sense-Plan-Act pipeline of household robot ","url":"https://arxiv.org/abs/2605.18593","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00365","type":"paper","title":"OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences","authors":["Kaixiang Wang","Jiong Lou","Zhaojiacheng Zhou","Jie Li"],"year":2026,"abstract":"Memory-augmented large language model (LLM) agents use iterative reflection and self-evolution to solve complex tasks, but these mechanisms introduce security risks. Existing agentic memory attacks require privileged access or explicit malicious content, making them detectable by advanced safety filters. This leaves a subtler attack surface underexplored: whether adversaries can induce agent to generate experiences that appear locally correct and semantically plausible yet induce harmful general","url":"https://arxiv.org/abs/2605.18930","categories":["data-poisoning","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00366","type":"paper","title":"The Importance of Out-of-Band Metadata for Safe Autonomous Agents: The Redpanda Agentic Data Plane","authors":["Tyler Akidau","Tyler Rockwood","Johannes Brüderl","Marc Millstone"],"year":2026,"abstract":"AI agents are increasingly expected to operate as digital employees: accessing enterprise data, making decisions, and taking actions autonomously. But agents are simultaneously less predictable than humans -- prone to hallucination, misinterpretation, and adversarial manipulation -- and more technically capable: with deep system knowledge and high-throughput interfaces cascading damage at machine speed. This combination makes it unsafe to rely on agents to faithfully interpret or propagate secur","url":"https://arxiv.org/abs/2605.29082","categories":["agentic-threats","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00367","type":"paper","title":"Got a Secret? LLM Agents Can't Keep It: Evaluating Privacy in Multi-Agent Systems","authors":["Aman Priyanshu","Supriti Vijay","Esha Pahwa"],"year":2026,"abstract":"LLM safety evaluations predominantly test models in isolation, yet deployed AI agents increasingly operate within persistent social environments alongside other agents. We introduce a Moltbook-style simulation platform where thousands of LLM agents interact across communities over a simulated month, and use it to evaluate privacy as a downstream safety concern under varying degrees of social pressure. We find that shifting from single turn to multi turn social evaluation amplifies privacy violat","url":"https://arxiv.org/abs/2605.27766","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00368","type":"paper","title":"Intelligence as Managed Autonomy: Failure, Escalation, and Governance for Agentic AI Systems","authors":["Srini Ramaswamy"],"year":2026,"abstract":"As autonomous and agentic AI systems scale in robotic and human-machine environments, managing hallucination and persistent but unjustified action remains an open challenge. Rather than attributing these failures solely to model or alignment limitations, this paper explores the architectural vulnerability of unbounded autonomy - the presumption that an agent should continue operating regardless of rising uncertainty. It introduces a theory of managed autonomy that defines intelligent behavior th","url":"https://arxiv.org/abs/2605.27628","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00369","type":"paper","title":"Measuring Security Without Fooling Ourselves: Why Benchmarking Agents Is Hard","authors":["Sahar Abdelnabi","Chris Hicks","Konrad Rieck","Ahmad-Reza Sadeghi"],"year":2026,"abstract":"The benchmarks used to evaluate AI agents in security-critical roles suffer from crucial weaknesses. Building on recent empirical evidence, we characterize three core challenges that undermine security evaluations: benchmark vulnerabilities, temporal staleness, and runtime uncertainty. We then outline practical directions toward building more robust and trustworthy evaluation frameworks.","url":"https://arxiv.org/abs/2605.22568","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00370","type":"paper","title":"Remembering More, Risking More: Longitudinal Safety Risks in Memory-Equipped LLM Agents","authors":["Ahmad Al-Tawaha","Shangding Gu","Peizhi Niu","Ruoxi Jia","Ming Jin"],"year":2026,"abstract":"Safety evaluations of memory-equipped LLM agents typically measure within-task safety: whether an agent completes a single scenario safely, often under adversarial conditions such as prompt injection or memory poisoning. In deployment, however, a single agent serves many independent tasks over a long horizon, and memory accumulated during earlier tasks can affect behavior on later, unrelated ones. Studying this regime requires evaluation along the temporal dimension across tasks: not whether an ","url":"https://arxiv.org/abs/2605.17830","categories":["prompt-injection","data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00371","type":"paper","title":"Towards trustworthy agentic AI: a comprehensive survey of safety, robustness, privacy, and system security","authors":["Jinhu Qi","Muzhi Li","Jiahong Liu","Yuqin Shu","Dianzhi Yu","Shicheng Ma","Wenqian Cui","Yiyang Zhao","Yiyi Chen","Ruoxi Jiang","Irwin King","Zenglin Xu"],"year":2026,"abstract":"Agentic AI systems -- Large Language Models (LLMs) augmented with planning, tool use, memory, and long-horizon interactions -- can execute complex tasks autonomously, but their multi-step trajectories introduce new failure modes that challenge trustworthiness. This survey provides a focused examination of trustworthy agentic AI through two core dimensions that are critical for high-risk deployments: Safety and Robustness, and Privacy and System Security. For each dimension, we clarify key concep","url":"https://arxiv.org/abs/2605.23989","categories":["agentic-threats","tool-use-security","survey"],"reviewed":false},{"id":"llmsec-2026-00372","type":"paper","title":"The End of Trust: How Agentic AI Breaks Security Assumptions","authors":["Osama Zafar","Alexander Nemecek","Erman Ayday"],"year":2026,"abstract":"For decades, the security of digital interaction has rested on an unacknowledged economic constraint. Attackers faced a tradeoff between the fidelity of a deception and the scale at which it could be deployed. Convincing impersonation required sustained human effort and was confined to a narrow set of high-value targets, while mass-market attacks sacrificed plausibility for reach. Detection systems, verification mechanisms, and user awareness training have all been implicitly calibrated to the a","url":"https://arxiv.org/abs/2605.16436","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00373","type":"paper","title":"The Misattribution Gap: When Memory Poisoning Looks Like Model Failure in Agentic AI Systems","authors":["Tanzim Ahad","Ismail Hossain","Md Jahangir Alam","Sai Puppala","Syed Bahauddin Alam","Sajedul Talukder"],"year":2026,"abstract":"Multi-agent AI pipelines typically assume that agent misconduct originates from model misalignment. We identify a structural failure in this assumption, the \\emph{Misattribution Gap}, where memory-layer attacks produce behaviors indistinguishable from model failure, causing defenders to apply the wrong remediation. We formalize \\emph{Semantic Norm Drift} (SND) as a third path to agent misconduct, distinct from emergent misalignment and collusion. In SND, a policy-formatted document enters a shar","url":"https://arxiv.org/abs/2605.22842","categories":["data-poisoning","agentic-threats","agent-architecture","memory-security"],"reviewed":false},{"id":"llmsec-2026-00374","type":"paper","title":"Red-Teaming Medical AI: Systematic Adversarial Evaluation of LLM Safety Guardrails in Clinical Contexts","authors":["T. Ekram"],"year":2026,"venue":"medRxiv","url":"https://www.semanticscholar.org/paper/e5f756d97f03c3fdbc96a955a205ca2e274307de","categories":["guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00375","type":"paper","title":"Assessing Risks of Large Language Models in Mental Health Support: A Framework for Automated Clinical AI Red Teaming","authors":["I. Steenstra","Paola Pedrelli","Weiyan Shi","S. Marsella","Timothy W. Bickmore"],"year":2026,"venue":"arXiv.org","abstract":"Large Language Models (LLMs) are increasingly utilized for mental health support; however, current safety benchmarks often fail to detect the complex, longitudinal risks inherent in therapeutic dialogue. We introduce an evaluation framework that pairs AI psychotherapists with simulated patient agents equipped with dynamic cognitive-affective models and assesses therapy session simulations against a comprehensive quality of care and risk ontology. We apply this framework to a high-impact test cas","url":"https://www.semanticscholar.org/paper/98a12d04ba9a6d07bab2a443b7994dd805eeea35","categories":["red-teaming","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00376","type":"paper","title":"A Safe Harbor for AI Evaluation and Red Teaming","authors":["Shayne Longpre","Sayash Kapoor","Kevin Klyman","A. Ramaswami","Rishi Bommasani","Borhane Blili-Hamelin","Yangsibo Huang","Aviya Skowron","Zheng-Xin Yong","Suhas Kotha","Yi Zeng","Weiyan Shi","Xianjun Yang","Reid Southen","Alexander Robey","Patrick Chao","Diyi Yang","Ruoxi Jia","Daniel Kang","Sandy Pentland","Arvind Narayanan","Percy Liang","Peter Henderson"],"year":2024,"venue":"International Conference on Machine Learning","abstract":"Independent evaluation and red teaming are critical for identifying the risks posed by generative AI systems. However, the terms of service and enforcement strategies used by prominent AI companies to deter model misuse have disincentives on good faith safety evaluations. This causes some researchers to fear that conducting such research or releasing their findings will result in account suspensions or legal reprisal. Although some companies offer researcher access programs, they are an inadequa","url":"https://www.semanticscholar.org/paper/21f8977648e25ce8b95020e6f01988af99209c82","categories":["red-teaming"],"citation_count":73,"reviewed":false},{"id":"llmsec-2026-00377","type":"paper","title":"OpenAI's Approach to External Red Teaming for AI Models and Systems","authors":["L. Ahmad","S. Agarwal","Michael Lampe","Pamela Mishkin"],"year":2025,"venue":"arXiv.org","abstract":"Red teaming has emerged as a critical practice in assessing the possible risks of AI models and systems. It aids in the discovery of novel risks, stress testing possible gaps in existing mitigations, enriching existing quantitative safety metrics, facilitating the creation of new safety measurements, and enhancing public trust and the legitimacy of AI risk assessments. This white paper describes OpenAI's work to date in external red teaming and draws some more general conclusions from this work.","url":"https://www.semanticscholar.org/paper/4329ac5ac885b9bfe6510d98cfbde77806f6e82e","categories":["red-teaming","threat-modeling"],"citation_count":44,"reviewed":false},{"id":"llmsec-2026-00378","type":"paper","title":"Red-Teaming for Generative AI: Silver Bullet or Security Theater?","authors":["Michael Feffer","Anusha Sinha","Zachary Chase Lipton","Hoda Heidari"],"year":2024,"venue":"AAAI/ACM Conference on AI, Ethics, and Society","abstract":"In response to rising concerns surrounding the safety, security, and trustworthiness of Generative AI (GenAI) models, practitioners and regulators alike have pointed to AI red-teaming as a key component of their strategies for identifying and mitigating these risks. However, despite AI red-teaming’s central role in policy discussions and corporate messaging, significant questions remain about what precisely it means, what role it can play in regulation, and how it relates to conventional red-tea","url":"https://www.semanticscholar.org/paper/4fda99880cdbf8f178f01eb4c8dbdae7f959ea94","categories":["red-teaming"],"citation_count":151,"reviewed":false},{"id":"llmsec-2026-00379","type":"paper","title":"SafeProtein: Red-Teaming Framework and Benchmark for Protein Foundation Models","authors":["Jigang Fan","Zhenghong Zhou","Ruofan Jin","Le Cong","Mengdi Wang","Zaixi Zhang"],"year":2025,"venue":"arXiv.org","abstract":"Proteins play crucial roles in almost all biological processes. The advancement of deep learning has greatly accelerated the development of protein foundation models, leading to significant successes in protein understanding and design. However, the lack of systematic red-teaming for these models has raised serious concerns about their potential misuse, such as generating proteins with biological safety risks. This paper introduces SafeProtein, the first red-teaming framework designed for protei","url":"https://www.semanticscholar.org/paper/b316f37ef9d48c9aa073245df9e65678b8eca33f","categories":["red-teaming","benchmarks"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00380","type":"paper","title":"The PIEE Cycle: A Structured Framework for Red Teaming Large Language Models in Clinical Decision-Making","authors":["Maissa Trabilsy","Srinivagasam Prabha","C. A. Gomez-Cabello","S. A. Haider","Ariana Genovese","S. Borna","Nadia G. Wood","N. Gopala","Cui Tao","AJ Forte"],"year":2025,"venue":"Bioengineering","abstract":"The increasing integration of large language models (LLMs) into healthcare presents significant opportunities, but also critical risks related to patient safety, accuracy, and ethical alignment. Despite these concerns, no standardized framework exists for systematically evaluating and stress testing LLM behavior in clinical decision-making. The PIEE cycle—Planning and Preparation, Information Gathering and Prompt Generation, Execution, and Evaluation—is a structured red-teaming framework develop","url":"https://www.semanticscholar.org/paper/1417c22f2e479509f8b3a179b784b945498b71a2","categories":["guardrails","red-teaming"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00381","type":"paper","title":"DREAM: Dynamic Red-teaming across Environments for AI Models","authors":["Liming Lu","Xiang Gu","Junyu Huang","Jiawei Du","Xu Zheng","Yunhuai Liu","Yongbin Zhou","Shuchao Pang"],"year":2025,"abstract":"Large Language Models (LLMs) are increasingly used in agentic systems, where their interactions with diverse tools and environments create complex, multi-stage safety challenges. However, existing benchmarks mostly rely on static, single-turn assessments that miss vulnerabilities from adaptive, long-chain attacks. To fill this gap, we introduce DREAM, a framework for systematic evaluation of LLM agents against dynamic, multi-stage attacks. At its core, DREAM uses a Cross-Environment Adversarial ","url":"https://www.semanticscholar.org/paper/8e63fb4190d0bb96927a6d4354d291254f90e042","categories":["agentic-threats","red-teaming","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00382","type":"paper","title":"Jailbreak-Zero: A Path to Pareto Optimal Red Teaming for Large Language Models","authors":["Kai Hu","Abhinav Aggarwal","Mehran Khodabandeh","David W. Zhang","Eric Hsin","Li Chen","Ankit Jain","Matt Fredrikson","Akash Bharadwaj"],"year":2025,"venue":"arXiv.org","abstract":"This paper introduces Jailbreak-Zero, a novel red teaming methodology that shifts the paradigm of Large Language Model (LLM) safety evaluation from a constrained example-based approach to a more expansive and effective policy-based framework. By leveraging an attack LLM to generate a high volume of diverse adversarial prompts and then fine-tuning this attack model with a preference dataset, Jailbreak-Zero achieves Pareto optimality across the crucial objectives of policy coverage, attack strateg","url":"https://www.semanticscholar.org/paper/0e54275afd64916c2a2137b9a81a8402c9e6faff","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00383","type":"paper","title":"SIRAJ: Diverse and Efficient Red-Teaming for LLM Agents via Distilled Structured Reasoning","authors":["Kai Zhou","Ahmed Elgohary","S. M. Iftekhar","♣. Amin","Suyu Ge","Chunting Zhou","Rui Hou","Madian Khabsa","Yi-Chia Wang","Qifan Wang","Jiawei Han","Yun-ing Mao","Mart","Daya Guo","Dejian Yang","Haowei Zhang","Jun-Mei Song","Ruoyu Zhang","Runxin Xu","Qihao Zhu","Shirong Ma","Weiyang Guo","Ze-Yun Shi","Zhuo Li","Yequan Wang","Xiaogeng Liu","Peiran Li","Edward Suh","Yevgeniy Vorobeychik","Zhuoqing Mao","Somesh Jha","P. McDaniel","Huan Sun","Bo Li","Yanjiang Liu","Shuheng Zhou","Yujia Qin","Shihao Liang","Yining Ye","Kunlun Zhu","Lan Yan","Ya-Ting Lu","Yankai Lin","X. Cong","Xiangru Tang","Mikayel Samvelyan","S. Raparthy","Andrei Lupu","Eric Hambro","A. Markosyan","Manish Bhatt","Chejian Xu","Mintong Kang","Jiawei Zhang","Zeyi Liao","Lingbo Mo","Mengqi Yuan","An Yang","Anfeng Li","Baosong Yang","Beichen Zhang","Binyuan Hui","Bo Zheng","Bo Yu","Chang Gao","Shunyu Yao","H. Chen","John Yang","Karthik R. Narasimhan","Webshop","Andy Zhou","Kevin Wu","Francesco Pinto","Zhaorun Chen","Yi Zeng","Yu Yang","Shuang Yang","Sanmi Koyejo","James Zou","Andy Zou","Zifan Wang","Nicholas Carlini","Milad Nasr"],"year":2025,"venue":"Conference of the European Chapter of the Association for Computational Linguistics","abstract":"The ability of LLM agents to plan and invoke tools exposes them to new safety risks, making a comprehensive red-teaming system crucial for discovering vulnerabilities and ensuring their safe deployment. We present SIRAJ: a generic red-teaming framework for arbitrary black-box LLM agents. We employ a dynamic two-step process that starts with an agent definition and generates diverse seed test cases that cover various risk outcomes, tool-use trajectories, and risk sources. Then, it iteratively con","url":"https://www.semanticscholar.org/paper/910c759401877c93f37b5114d069108ea7c140e7","categories":["agentic-threats","red-teaming"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00384","type":"paper","title":"Toward Trustworthy Chatbots: A Protocol for Red Teaming for Health Related Conversations","authors":["Syed-Amad Hussain","Daniel I. Jackson","Ashley Lewis","Eric Fosler-Lussier","Emre Sezgin"],"year":2025,"venue":"medRxiv","abstract":"Introduction: Health-related chatbots are increasingly used to mediate conversations that carry clinical significance and emotional weight. Retrieval-augmented generation (RAG) can reduce factual errors ('hallucinations'), but the risks remain, with additional challenges coming from chatbots acting against behavioral safety and scope rules. Red teaming, an adversarial testing process that deliberately probes systems for failures before deployment, offers a way to surface potential risks. We desc","url":"https://www.semanticscholar.org/paper/930205acf84599560a761027e57b15506d11fe7b","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00385","type":"paper","title":"AART: AI-Assisted Red-Teaming with Diverse Data Generation for New LLM-powered Applications","authors":["Bhaktipriya Radharapu","Kevin Robinson","L. Aroyo","Preethi Lahoti"],"year":2023,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"Adversarial testing of large language models (LLMs) is crucial for their safe and responsible deployment. We introduce a novel approach for automated generation of adversarial evaluation datasets to test the safety of LLM generations on new downstream applications. We call it AI-assisted Red-Teaming (AART) - an automated alternative to current manual red-teaming efforts. AART offers a data generation and augmentation pipeline of reusable and customizable recipes that reduce human effort signific","url":"https://www.semanticscholar.org/paper/57d0e672040800e8d882ff0022647c087095e35f","categories":["red-teaming"],"citation_count":68,"reviewed":false},{"id":"llmsec-2026-00386","type":"paper","title":"Quality-Diversity Red-Teaming: Automated Generation of High-Quality and Diverse Attackers for Large Language Models","authors":["Ren-Jian Wang","Ke Xue","Zeyu Qin","Ziniu Li","Sheng Tang","Hao-Tian Li","Shengcai Liu","Chao Qian"],"year":2025,"venue":"arXiv.org","abstract":"Ensuring safety of large language models (LLMs) is important. Red teaming--a systematic approach to identifying adversarial prompts that elicit harmful responses from target LLMs--has emerged as a crucial safety evaluation method. Within this framework, the diversity of adversarial prompts is essential for comprehensive safety assessments. We find that previous approaches to red-teaming may suffer from two key limitations. First, they often pursue diversity through simplistic metrics like word f","url":"https://www.semanticscholar.org/paper/da14e71f31e98612ebf871be742dddc55da509d1","categories":["red-teaming"],"citation_count":7,"reviewed":false},{"id":"llmsec-2026-00387","type":"paper","title":"RedDebate: Safer Responses through Multi-Agent Red Teaming Debates","authors":["A. Asad","Stephen Obadinma","Radin Shayanfar","Xiaodan Zhu"],"year":2025,"venue":"arXiv.org","abstract":"We introduce RedDebate, a novel multi-agent debate framework that provides the foundation for Large Language Models (LLMs) to identify and mitigate their unsafe behaviours. Existing AI safety approaches often rely on costly human evaluation or isolated single-model assessment, both constrained by scalability and prone to oversight failures. RedDebate employs collaborative argumentation among multiple LLMs across diverse debate scenarios, enabling them to critically evaluate one another's reasoni","url":"https://www.semanticscholar.org/paper/89b908678883095f8a2eda436bb5aac623b4e18e","categories":["red-teaming","agent-architecture"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00388","type":"paper","title":"ASSERT: Automated Safety Scenario Red Teaming for Evaluating the Robustness of Large Language Models","authors":["Alex Mei","Sharon Levy","W. Wang"],"year":2023,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"As large language models are integrated into society, robustness toward a suite of prompts is increasingly important to maintain reliability in a high-variance environment.Robustness evaluations must comprehensively encapsulate the various settings in which a user may invoke an intelligent system. This paper proposes ASSERT, Automated Safety Scenario Red Teaming, consisting of three methods -- semantically aligned augmentation, target bootstrapping, and adversarial knowledge injection. For robus","url":"https://www.semanticscholar.org/paper/aa9aa1c315cb2a0c1759d82fb3d4b4506c2dbb7c","categories":["red-teaming"],"citation_count":16,"reviewed":false},{"id":"llmsec-2026-00389","type":"paper","title":"Red Teaming GPT-4V: Are GPT-4V Safe Against Uni/Multi-Modal Jailbreak Attacks?","authors":["Shuo Chen","Zhen Han","Bailan He","Zifeng Ding","Wenqian Yu","Philip H. S. Torr","Volker Tresp","Jindong Gu"],"year":2024,"venue":"arXiv.org","abstract":"Various jailbreak attacks have been proposed to red-team Large Language Models (LLMs) and revealed the vulnerable safeguards of LLMs. Besides, some methods are not limited to the textual modality and extend the jailbreak attack to Multimodal Large Language Models (MLLMs) by perturbing the visual input. However, the absence of a universal evaluation benchmark complicates the performance reproduction and fair comparison. Besides, there is a lack of comprehensive evaluation of closed-source state-o","url":"https://www.semanticscholar.org/paper/526c88e301af88080ae4ecc3b65c9fc1f8f383f8","categories":["jailbreaking","guardrails","red-teaming","benchmarks"],"citation_count":34,"reviewed":false},{"id":"llmsec-2026-00390","type":"paper","title":"Safety by Measurement: A Systematic Literature Review of AI Safety Evaluation Methods","authors":["Markov Grey","Charbel-Raphaël Ségerie"],"year":2025,"venue":"arXiv.org","abstract":"As frontier AI systems advance toward transformative capabilities, we need a parallel transformation in how we measure and evaluate these systems to ensure safety and inform governance. While benchmarks have been the primary method for estimating model capabilities, they often fail to establish true upper bounds or predict deployment behavior. This literature review consolidates the rapidly evolving field of AI safety evaluations, proposing a systematic taxonomy around three dimensions: what pro","url":"https://www.semanticscholar.org/paper/42c6c25b277ef3ca5045ba0507bf5252f2df75a6","categories":["benchmarks","survey"],"citation_count":19,"reviewed":false},{"id":"llmsec-2026-00391","type":"paper","title":"Real-World Evaluation of Large Language Models in Healthcare (RWE-LLM): A New Realm of AI Safety & Validation","authors":["MD MHA¹ Meenesh Bhimani","BS¹ Alex Miller","P. M. Jonathan D. Agnew","Markel Sanz Ausin","BA¹ Mariska Raglow-Defranco","MD Mba Harpreet Mangat","Bsn RN Ccm Michelle Voisard","RN Bsn Ccm Maggie Taylor","BS Sebastian Bierman-Lytle","BS BA Vishal Parikh","Juliana Ghukasyan","BS Rae Lasko","Saad Godil","Meng","MD Mph Ashish Atreja","PhD¹ Subhabrata Mukherjee"],"year":2025,"venue":"medRxiv","abstract":"Background: The deployment of artificial intelligence (AI) in healthcare necessitates robust safety validation frameworks, particularly for systems directly interacting with patients. While theoretical frameworks exist, there remains a critical gap between abstract principles and practical implementation. Traditional LLM benchmarking approaches provide very limited output coverage and are insufficient for healthcare applications requiring high safety standards. Objective: To develop and evaluate","url":"https://www.semanticscholar.org/paper/1fe644e3b971a52cc57b1b5ddc4cba5408cf1a8d","categories":["benchmarks"],"citation_count":10,"reviewed":false},{"id":"llmsec-2026-00392","type":"paper","title":"Know Thy Judge: On the Robustness Meta-Evaluation of LLM Safety Judges","authors":["Francisco Eiras","Eliott Zemour","Eric Lin","Vaikkunth Mugunthan"],"year":2025,"venue":"arXiv.org","abstract":"Large Language Model (LLM) based judges form the underpinnings of key safety evaluation processes such as offline benchmarking, automated red-teaming, and online guardrailing. This widespread requirement raises the crucial question: can we trust the evaluations of these evaluators? In this paper, we highlight two critical challenges that are typically overlooked: (i) evaluations in the wild where factors like prompt sensitivity and distribution shifts can affect performance and (ii) adversarial ","url":"https://www.semanticscholar.org/paper/0ffb356aab98ae69c717f8b2969c3fed0592a048","categories":["guardrails","red-teaming","benchmarks"],"citation_count":14,"reviewed":false},{"id":"llmsec-2026-00393","type":"paper","title":"Evaluating Human-AI Safety: A Framework for Measuring Harmful Capability Uplift","authors":["Michelle Vaccaro","Jaeyoon Song","Abdullah Almaatouq","Michiel A. Bakker"],"year":2026,"abstract":"Current frontier AI safety evaluations emphasize static benchmarks, third-party annotations, and red-teaming. In this position paper, we argue that AI safety research should focus on human-centered evaluations that measure harmful capability uplift: the marginal increase in a user's ability to cause harm with a frontier model beyond what conventional tools already enable. We frame harmful capability uplift as a core AI safety metric, ground it in prior social science research, and provide concre","url":"https://www.semanticscholar.org/paper/3a3f4b4c7cd76bff6b5eacacc734dd97d05adc95","categories":["red-teaming","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00394","type":"paper","title":"AlignInsight: A Three-Layer Framework for Detecting Deceptive Alignment and Evaluation Awareness in Healthcare AI Systems","authors":["A. Onovo","Y. Cherima"],"year":2026,"venue":"medRxiv","url":"https://www.semanticscholar.org/paper/d9f0f2dafd5e976137c0037720665dbd936157f6","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00395","type":"paper","title":"MUSE: A Run-Centric Platform for Multimodal Unified Safety Evaluation of Large Language Models","authors":["Zhong-Qiu Wang","Yueqian Lin","Jingyang Zhang","Hai Li","Yiran Chen"],"year":2026,"abstract":"Safety evaluation and red-teaming of large language models remain predominantly text-centric, and existing frameworks lack the infrastructure to systematically test whether alignment generalizes to audio, image, and video inputs. We present MUSE (Multimodal Unified Safety Evaluation), an open-source, run-centric platform that integrates automatic cross-modal payload generation, three multi-turn attack algorithms (Crescendo, PAIR, Violent Durian), provider-agnostic model routing, and an LLM judge","url":"https://www.semanticscholar.org/paper/8d3675bff32bbde8709b044d721dc88a7ca55c8b","categories":["guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00396","type":"paper","title":"RAG LLMs are Not Safer: A Safety Analysis of Retrieval-Augmented Generation for Large Language Models","authors":["Shiyue Zhang","M. Dredze","AI Bloomberg","M. Zitnik","Meng Jiang","Mohit Bansal","James Zou","Jian Pei","Jian Liu","Jianfeng Gao","Jiawei Han","Jieyu Zhao","Jiliang Tang","Jindong Wang","Joaquin Vanschoren","John C. Mitchell","Kai Shu","Kaidi Xu","Kai-Wei Chang","Lifang He","Lifu Huang","Michael Backes","Aaron Hurst","Adam Lerer","Adam P. Goucher","Adam Perelman","Aditya Ramesh","Aidan Clark","AJ Os-trow","Akila Welihinda","Alan Hayes","Alec Radford","Alon Jacovi","Andrew Wang","Chris Alberti","Connie Tao","Jon Lipovetz","Kate Olszewska","Lukas Haas","Michelle Liu","Nate Keating","Adam E. Bloniarz","Carl Saroufim","Corey Fry","Dror Marcus","Doron Kukliansky","Gau-rav Singh Tomar","James Swirhun","J. Xing","Lily Wang","Madhu Gurumurthy","Michael Aaron","Moran Ambar","Rachana Fellinger","Rui Wang","Zizhao Zhang","S. Goldshtein","Dipanjan Das. 2025","Neel Jain","Avi Schwarzschild","Yuxin Wen","Gowthami Somepalli","John Kirchenbauer","Ping-yeh Chiang","Micah Goldblum","Aniruddha Saha","Jonas Geiping","Tom Goldstein. 2023","Jiaming Ji","Mickel Liu","Josef Dai","Xuehai Pan","Chi Zhang","Juntao Dai","Tianyi Qiu","Bo Chen","Borong Zhang","Hantao Lou","Kaile Wang","Ya Duan"],"year":2025,"venue":"North American Chapter of the Association for Computational Linguistics","abstract":"Efforts to ensure the safety of large language models (LLMs) include safety fine-tuning, evaluation, and red teaming. However, despite the widespread use of the Retrieval-Augmented Generation (RAG) framework, AI safety work focuses on standard LLMs, which means we know little about how RAG use cases change a model's safety profile. We conduct a detailed comparative analysis of RAG and non-RAG frameworks with eleven LLMs. We find that RAG can make models less safe and change their safety profile.","url":"https://www.semanticscholar.org/paper/f716a18b462826004899010dfc30947f9c01ef90","categories":["red-teaming"],"citation_count":28,"reviewed":false},{"id":"llmsec-2026-00397","type":"paper","title":"A Frontier AI Risk Management Framework: Bridging the Gap Between Current AI Practices and Established Risk Management","authors":["Siméon Campos","H. Papadatos","Fabien Roger","Chlo'e Touzet","Otter Quarks","Malcolm Murray"],"year":2025,"venue":"arXiv.org","abstract":"The recent development of powerful AI systems has highlighted the need for robust risk management frameworks in the AI industry. Although companies have begun to implement safety frameworks, current approaches often lack the systematic rigor found in other high-risk industries. This paper presents a comprehensive risk management framework for the development of frontier AI that bridges this gap by integrating established risk management principles with emerging AI-specific practices. The framewo","url":"https://www.semanticscholar.org/paper/9ef444e985f198664ea90ba9a4c99d464aa0c18e","categories":["risk-frameworks"],"citation_count":11,"reviewed":false},{"id":"llmsec-2026-00398","type":"paper","title":"MMDT: Decoding the Trustworthiness and Safety of Multimodal Foundation Models","authors":["Chejian Xu","Jiawei Zhang","Zhaorun Chen","Chulin Xie","Mintong Kang","Yujin Potter","Zhun Wang","Zhuowen Yuan","Alexander Xiong","Zidi Xiong","Chenhui Zhang","Lingzhi Yuan","Yi Zeng","Peiyang Xu","Chengquan Guo","Andy Zhou","J. Tan","Xuandong Zhao","Francesco Pinto","Zhen Xiang","Yu Gai","Zinan Lin","Dan Hendrycks","Bo Li","D. Song"],"year":2025,"venue":"arXiv.org","abstract":"Multimodal foundation models (MMFMs) play a crucial role in various applications, including autonomous driving, healthcare, and virtual assistants. However, several studies have revealed vulnerabilities in these models, such as generating unsafe content by text-to-image models. Existing benchmarks on multimodal models either predominantly assess the helpfulness of these models, or only focus on limited perspectives such as fairness and privacy. In this paper, we present the first unified platfor","url":"https://www.semanticscholar.org/paper/26c02dbc2f6db3e3b7acdb493a880a3456ff2cfd","categories":["benchmarks"],"citation_count":16,"reviewed":false},{"id":"llmsec-2026-00399","type":"paper","title":"GuidedBench: Measuring and Mitigating the Evaluation Discrepancies of In-the-wild LLM Jailbreak Methods","authors":["Ruixuan Huang","Xunguang Wang","Zongjie Li","Daoyuan Wu","Shuai Wang"],"year":2025,"abstract":"Despite the growing interest in jailbreak methods as an effective red-teaming tool for building safe and responsible large language models (LLMs), flawed evaluation system designs have led to significant discrepancies in their effectiveness assessments. We conduct a systematic measurement study based on 37 jailbreak studies since 2022, focusing on both the methods and the evaluation systems they employ. We find that existing evaluation systems lack case-specific criteria, resulting in misleading","url":"https://www.semanticscholar.org/paper/2098e70e436cb6ea7c2836384128b1cdf6775641","categories":["jailbreaking","red-teaming"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00400","type":"paper","title":"Inverting the Shield: Systematically Generating Safety Tests from Policy Specifications","authors":["Xiaoyu Lu","Xianglin Yang","Haijun Liu","Jiahao Liu","Kuntai Cai","Yan Xiao","J. Dong"],"year":2026,"abstract":"The widespread integration of Large Language Models (LLMs) necessitates rigorous and systematic safety evaluation. Existing paradigms either rely on constructed benchmarks to assess safety from predefined perspectives, or employ dynamic red-teaming to probe potential vulnerabilities. While effective, these approaches face challenges, as they depend heavily on expert domain knowledge, offer limited systematic guarantees, and are vulnerable to rapid obsolescence. To address these limitations, we i","url":"https://www.semanticscholar.org/paper/6a1f7108e15400c9dd3c2054125afaffebd03299","categories":["red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00401","type":"paper","title":"Jailbreak Mimicry: Automated Discovery of Narrative-Based Jailbreaks for Large Language Models","authors":["Pavlos Ntais"],"year":2025,"venue":"arXiv.org","abstract":"Large language models (LLMs) remain vulnerable to sophisticated prompt engineering attacks that exploit contextual framing to bypass safety mechanisms, posing significant risks in cybersecurity applications. We introduce Jailbreak Mimicry, a systematic methodology for training compact attacker models to automatically generate narrative-based jailbreak prompts in a one-shot manner. Our approach transforms adversarial prompt discovery from manual craftsmanship into a reproducible scientific proces","url":"https://www.semanticscholar.org/paper/be948a3061bfb640a0683e63e6afc6c2ceb63775","categories":["jailbreaking"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00402","type":"paper","title":"IH-Challenge: A Training Dataset to Improve Instruction Hierarchy on Frontier LLMs","authors":["Chuan Guo","J. Felipe","Cerón Uribe","Sicheng Zhu","Christopher A. Choquette-Choo","Stephanie L. Lin","Nikhil Kandpal","Milad Nasr","Michael Pokorny","S. Toyer","Miles Wang","Yao-Ching Yu","Alex Beutel","Kai Xiao OpenAI"],"year":2026,"abstract":"Instruction hierarchy (IH) defines how LLMs prioritize system, developer, user, and tool instructions under conflict, providing a concrete, trust-ordered policy for resolving instruction conflicts. IH is key to defending against jailbreaks, system prompt extractions, and agentic prompt injections. However, robust IH behavior is difficult to train: IH failures can be confounded with instruction-following failures, conflicts can be nuanced, and models can learn shortcuts such as overrefusing. We i","url":"https://www.semanticscholar.org/paper/0d1a1ced045bbc21126efb2be0a4064e1ce18f63","categories":["prompt-injection","jailbreaking","agentic-threats"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00403","type":"paper","title":"SecureCAI: Injection-Resilient LLM Assistants for Cybersecurity Operations","authors":["Mohammed Himayath Ali","Mohammed Aqib Abdullah","Mohammed Mudassir Uddin","S. Alam"],"year":2026,"venue":"arXiv.org","abstract":"Large Language Models have emerged as transformative tools for Security Operations Centers, enabling automated log analysis, phishing triage, and malware explanation; however, deployment in adversarial cybersecurity environments exposes critical vulnerabilities to prompt injection attacks where malicious instructions embedded in security artifacts manipulate model behavior. This paper introduces SecureCAI, a novel defense framework extending Constitutional AI principles with security-aware guard","url":"https://www.semanticscholar.org/paper/3d16be498e90f357318f27bf8c476567a51d4e47","categories":["prompt-injection","guardrails"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00404","type":"paper","title":"Large language models provide unsafe answers to patient-posed medical questions","authors":["R. Draelos","Samina Afreen","Barbara Blasko","Tiffany Brazile","Natasha Chase","Dimple Desai","Jessica Evert","H. Gardner","Lauren Herrmann","A. V. House","Stephanie Kass","Marianne Kavan","Kirshma Khemani","Amanda M. Koire","L. McDonald","Zahraa Rabeeah","A. Shah"],"year":2025,"venue":"npj Digital Medicine","abstract":"Millions of patients are regularly using large language model (LLM) chatbots for medical advice, raising patient safety concerns. This physician-led red-teaming study compares the safety of four publicly available chatbots—Claude by Anthropic, Gemini by Google, GPT-4o by OpenAI, and Llama-3.0/3.1-70B by Meta—on a new dataset, HealthAdvice, using an evaluation framework that enables quantitative and qualitative analysis. In total, 888 chatbot responses are evaluated for 222 patient-posed advice-s","url":"https://www.semanticscholar.org/paper/ebdb2c9347e5ab1b7c27c483aafab9584d28e8a1","categories":["red-teaming","benchmarks"],"citation_count":11,"reviewed":false},{"id":"llmsec-2026-00405","type":"paper","title":"PersonaTeaming: Exploring How Introducing Personas Can Improve Automated AI Red-Teaming","authors":["Wesley Hanwen Deng","Sunnie S. Y. Kim","Akshita Jha","Kenneth Holstein","Motahhare Eslami","L. Wilcox","Leon A. Gatys"],"year":2025,"venue":"arXiv.org","abstract":"Recent developments in AI governance and safety research have called for red-teaming methods that can effectively surface potential risks posed by AI models. Many of these calls have emphasized how the identities and backgrounds of red-teamers can shape their red-teaming strategies, and thus the kinds of risks they are likely to uncover. While automated red-teaming approaches promise to complement human red-teaming by enabling larger-scale exploration of model behavior, current approaches do not","url":"https://www.semanticscholar.org/paper/d9490d1ce9f58c5d545941246854e41831e2a74f","categories":["red-teaming"],"citation_count":10,"reviewed":false},{"id":"llmsec-2026-00406","type":"paper","title":"GenTI: Benchmarking LLMs for Autonomous IDPS Rule Generation for Unseen Attacks","authors":["Hassan Jalil Hadi","Rehana Yasmin","Ali Shoker"],"year":2026,"abstract":"Rule-based Intrusion Detection and Prevention Systems (IDPS) offer precise attack detection as well as mitigation, however their manually crafted, signature-driven rules limit adaptability to emerging and zero-day threats. Additionally, existing public datasets (e.g., CICIDS2017, UNSW-NB15) focus on traffic classification and provide little structured information to support automatic rule synthesis or prevention logic. To address this gap, we propose Generative Thread Intelligence (GenTI) \\footn","url":"https://arxiv.org/abs/2606.05844","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00407","type":"paper","title":"An Embarrassingly Simple Detector for Model Extraction Attacks in Large Language Model API Traffic","authors":["Shuze Liu","Qianwen Guo","Yushun Dong"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed through hosted APIs, making model extraction a practical threat to model ownership and service security. However, individual extraction queries often resemble benign requests, and existing evaluations often focus on single-query anomaly scoring or pure benign-versus-attacker user settings. We formulate model extraction monitoring as benign-calibrated traffic-window distribution testing and show that an embarrassingly simple detector is effec","url":"https://arxiv.org/abs/2606.05725","categories":["model-extraction","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00408","type":"paper","title":"Safety Paradox: How Enhanced Safety Awareness Leaves LLMs Vulnerable to Posterior Attack","authors":["Long P. Hoang","Hai V. Le","Shaoyang Xu","Wei Lu","Wenxuan Zhang"],"year":2026,"abstract":"Large language models (LLMs) are rigorously aligned to refuse harmful requests, a process that inherently cultivates a latent capacity to evaluate and recognize unsafe content. In this work, we reveal that this advanced safety awareness inadvertently introduces a fatal vulnerability. We introduce Posterior Attack, a single-query jailbreak that bypasses guardrails by prompting the model to generate the exact harmful response its internal classifier would normally flag as unsafe. Through extensive","url":"https://arxiv.org/abs/2606.05614","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00409","type":"paper","title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","authors":["Seungwon Jeong","Jiwoo Jeong","Hyeonjin Kim","Yunseok Lee","Woojin Lee"],"year":2026,"abstract":"As large language models (LLMs) are widely deployed, identifying their vulnerability through jailbreak attacks becomes increasingly critical. Optimization-based attacks like Greedy Coordinate Gradient (GCG) have focused on inserting adversarial tokens to the end of prompts. However, GCG restricts adversarial tokens to a fixed insertion point (typically the prompt suffix), leaving the effect of inserting tokens at other positions unexplored. In this paper, we empirically investigate \\emph{slots},","url":"https://arxiv.org/abs/2606.05609","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00410","type":"paper","title":"From Untrusted Input to Trusted Memory: A Systematic Study of Memory Poisoning Attacks in LLM Agents","authors":["Pritam Dash","Tongyu Ge","Aditi Jain","Tanmay Shah","Zhiwei Shang"],"year":2026,"abstract":"Memory is a core component of AI agents, enabling them to accumulate knowledge across interactions and improve performance. However, persistent memory introduces the risk of memory poisoning, where a single adversarial memory write can exert long-term influence over agent behavior. We present a systematic study of memory poisoning in LLM-based agents. We identify four memory write channels and nine structural vulnerabilities in model capabilities, system prompt design, and agent system architect","url":"https://arxiv.org/abs/2606.04329","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00411","type":"paper","title":"Black-box, Adaptive, Efficient, Transferable, Harmful, Applicable... Attacks Are All You Need to Break LLMs","authors":["Vincent Limbach","Jonas Dornbusch","David Lüdke","Stephan Günnemann","Leo Schwinn"],"year":2026,"abstract":"Accurately evaluating adversarial robustness is a longstanding challenge. A flawed attack design can inflate robustness estimates, making deployment risk assessment and defense comparison unreliable. Historically, standardized attacks such as AutoAttack have largely resolved this for image classifiers, providing a reliable evaluation baseline for systematic comparison across defenses. However, no equivalent exists for LLM jailbreak evaluation yet, where designing such an attack is considerably m","url":"https://arxiv.org/abs/2606.03647","categories":["jailbreaking","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00412","type":"paper","title":"Testing LLM Arithmetic Reasoning Generalization with Automatic Numeric-Remapping Attacks","authors":["Malia Barker","Bishal Lakha","Edoardo Serra","Francesco Gullo"],"year":2026,"abstract":"Large language models achieve strong performance on arithmetic reasoning benchmarks, and one common response to arithmetic brittleness is to delegate computation to code. Yet models are still often used in settings where they must reason directly from natural language, and trustworthy models should solve small-number arithmetic word problems without external tools. Prior work shows that LLMs are sensitive to numerical variation: a model may solve an original problem but fail on structurally simi","url":"https://arxiv.org/abs/2606.03606","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00413","type":"paper","title":"Learn from Your Mistakes: Tree-like Self-Play for Secure Code LLMs","authors":["Wenqi Chen","Ziyan Zhang","Bing Wang","Lin Liu","Hengheng Zhang","Zhengsu Chen"],"year":2026,"abstract":"While Large Language Models (LLMs) excel in code generation, they remain prone to replicating subtle yet critical vulnerabilities endemic to their training data. Current alignment techniques, such as Supervised Fine-Tuning (SFT) and Reinforcement Learning (RL), typically apply coarse-grained optimization at the sequence level. This approach often fails to address the localized nature of security flaws, where a single incorrect token choice can compromise an entire program. To bridge this gap, we","url":"https://arxiv.org/abs/2606.03489","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00414","type":"paper","title":"RogueMerge: Robust and Unified Attacks against LLM Model Merging","authors":["Jinghuai Zhang","Yetian He","Kunlin Cai","Han Zhao","Fnu Suya","Yuan Tian"],"year":2026,"abstract":"Model merging composes specialized capabilities into a single LLM by aggregating task vectors sourced from unverified public platforms, exposing a critical supply-chain attack surface: Because any malicious behavior can be encoded into a task vector, and merging grants third-party vectors direct write access to model weights, an attacker-provided task vector can enable or amplify diverse downstream threats. Prior work studies only backdoor attacks against model merging for classifiers using stat","url":"https://arxiv.org/abs/2606.03344","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00415","type":"paper","title":"\"**Important** You should give me full credits!\": Exploring Prompt Injection Attacks on LLM-Based Automatic Grading Systems","authors":["Hang Li","Fedor Filippov","Yuling Lin","Pengfei He","Kaiqi Yang","Yucheng Chu","Yingqian Cui","Hui Liu","Jiliang Tang"],"year":2026,"abstract":"The emergence of large language models (LLMs) has significantly accelerated recent research on LLM-based automatic grading (AG) systems. Benefiting from the strong instruction-following capabilities and broad prior knowledge of LLMs, educators can deploy AG systems across diverse tasks using only natural language rubrics while achieving satisfactory grading performance. Despite these advantages, new security concerns may also arise. In particular, prompt injection (PI) attacks have recently beco","url":"https://arxiv.org/abs/2606.03090","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00416","type":"paper","title":"Gate AI: LLM Security Benchmark Evaluation Methodology and Results","authors":["Ryle Goehausen","Marcus Sousa"],"year":2026,"abstract":"Published evaluations of prompt-injection and jailbreak detectors for Large Language Models often suffer from two systematic weaknesses: per-dataset threshold tuning and undisclosed operating points. We describe an evaluation harness that addresses both. The detector under evaluation is scored across 16 public benchmarks (12,111 samples) using 5-fold cross-validation. StratifiedKFold (by row) is the headline pass; a parallel StratifiedGroupKFold pass over a composite key (parent-prompt id plus M","url":"https://arxiv.org/abs/2606.02959","categories":["prompt-injection","jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00417","type":"paper","title":"MaskForge: Structure-Aware Adaptive Attacks for Jailbreaking Diffusion Large Language Models","authors":["Yingzi Ma","Zhengyue Zhao","Xiaogeng Liu","Minhui Xue","Yue Zhao","Chaowei Xiao"],"year":2026,"abstract":"Diffusion large language models (dLLMs) generate text by iteratively denoising partially masked sequences under bidirectional context, exposing a safety surface distinct from autoregressive LLMs. Because mask tokens are native inputs and tokens are committed by confidence rather than position, harmful content can be induced through infilling and outside the monitored prefix. Existing jailbreaks either miss this native infill capability or rely on low-diversity mask-bearing templates applied unif","url":"https://arxiv.org/abs/2606.04027","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00418","type":"paper","title":"THRD: A Training-Free Multi-Turn Defense Framework for Jailbreak Attacks on Large Language Models","authors":["Zhiqing Ma","Zhonghao Xu","Dong Yu","Chen Kang","Changliang Li","Pengyuan Liu"],"year":2026,"abstract":"Multi-turn jailbreak attacks pose a growing threat to LLMs by exploiting conversational dynamics such as gradual escalation and cross-turn coordination. Existing defenses either rely on costly retraining -- often degrading model utility -- or apply single-turn analysis independently at each turn, failing to capture how risk accumulates along interaction trajectories. We observe that safety behavior in multi-turn interaction is trajectory-dependent: dialogue history continuously reshapes the mode","url":"https://arxiv.org/abs/2606.01738","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00419","type":"paper","title":"Dive into Ambiguity: A*-Inspired Multi-Agents Commonsense Obfuscation Attack on LLM Prompts","authors":["Boxuan Wang","Zhuoyun Li","Xiaowei Huang","Yi Dong"],"year":2026,"abstract":"Large language models (LLMs) excel in reasoning and knowledge-intensive tasks but remain vulnerable to prompt-level adversarial attacks that preserve intent while triggering commonsense hallucinations. This vulnerability is urgent, as LLMs are rapidly integrated into safety-critical domains where factual reliability is non-negotiable. Existing attack methods either lack efficiency or fail to capture the adaptive strategies of real-world adversaries. We propose an A*-inspired Factual Error Induct","url":"https://arxiv.org/abs/2606.01441","categories":["adversarial-examples","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00420","type":"paper","title":"Cross-Generational Transfer of Adversarial Attacks Reveals Non-Monotonic Safety Alignment in LLMs","authors":["Subhadip Mitra"],"year":2026,"abstract":"Safety alignment in LLMs does not improve monotonically across model generations. Studying four generations of Google's Gemma family (7B-31B) with quality-diversity evolution (MAP-Elites) as an automated red-teaming probe, we find that Gemma 3 (12B) exhibits 68.7% +/- 5.7% attack success rate (ASR; mean +/- std, 3 seeds), significantly higher than its predecessor Gemma 2 (45.5% +/- 7.2%; p = 0.030, paired bootstrap) and its successor Gemma 4 (33.9% +/- 1.8%). Replaying evolved attack archives ac","url":"https://arxiv.org/abs/2606.00813","categories":["adversarial-examples","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00421","type":"paper","title":"Quality-Diversity Evolution for Discovering Diverse Vulnerabilities in LLM Safety","authors":["Subhadip Mitra"],"year":2026,"abstract":"Current approaches to LLM adversarial testing suffer from coverage gaps: manual red-teaming does not scale, LLM-as-attacker methods exhibit mode collapse, and gradient-based approaches produce uninterpretable gibberish. We introduce a quality-diversity evolutionary framework that operates at the semantic level, evolving interpretable attack strategies rather than token sequences. Using MAP-Elites, we maintain a diverse archive of attacks across behavioral dimensions (strategy type, encoding meth","url":"https://arxiv.org/abs/2606.00801","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00422","type":"paper","title":"Persona Attack: Incremental Memory Injection Jailbreak Attack against Large Language Models","authors":["Junyoung Park","Seongyong Ju","Sunghwan Park","Jaewoo Lee"],"year":2026,"abstract":"As Large Language Models evolve for user convenience, vulnerability to jailbreak attacks continues to be reported despite ongoing efforts in safety training. Traditional jailbreak techniques typically focus on a single prompt injection, neglecting the models' ability to remember the flow of conversation and the user's instructions. In this paper, we propose Persona Attack, a memory injection based jailbreak method that manipulates the model's context window through a step by step approach. Exper","url":"https://arxiv.org/abs/2606.00150","categories":["prompt-injection","jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00423","type":"paper","title":"Membrane: A Self-Evolving Contrastive Safety Memory for LLM Agent Defense","authors":["Minseok Choi","Seungbin Yang","Dongjin Kim","Subin Kim","Jungmin Son","Yunseung Lee","Jaegul Choo","Youngjun Kwak"],"year":2026,"abstract":"Despite advances in safety alignment, large language models remain vulnerable to continuously evolving jailbreaks. Existing fine-tuned safety classifiers cannot adapt to these evolving attacks, while adaptive memory-based guardrails tend to over-refuse benign queries that resemble stored attacks. We propose Membrane, a self-evolving guardrail built on Contrastive Safety Memory (CSM): each cell pairs the conditions for blocking a harmful query with those for permitting a superficially similar ben","url":"https://arxiv.org/abs/2606.05743","categories":["jailbreaking","agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00424","type":"paper","title":"GuardNet: Ensemble Strategies of Shallow Neural Networks for Robust Prompt Injection and Jailbreak Detection","authors":["Paulo Ricardo Ferreira Neves","Edson Rodrigues da Cruz Filho","Paulo Henrique Eleuterio Falsetti","João Vitor Pavan","Ian Degaspari","Henrique Vieira Laturrague","Patrick Vieira Laturrague","Guilherme Nielsen Dias","Marccello Wilson Perez Berto","Gustavo Voltani Von Atzingen"],"year":2026,"abstract":"Large Language Models (LLMs) have transformed natural language processing, but they remain vulnerable to Prompt Injection (PI) and Jailbreak (JB) attacks. In addition, benchmark evaluations may be affected by contamination and partial information leakage, compromising performance estimates. This work presents GuardNet, a guardrail system based on an ensemble of shallow neural networks (BiLSTMs) with approximately 47 million parameters. We investigate the hypothesis that robustness in adversarial","url":"https://arxiv.org/abs/2606.05566","categories":["prompt-injection","jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00425","type":"paper","title":"What If Prompt Injection Never Left? Exploring Cross-Session Stored Prompt Injection in Agentic Systems","authors":["Yuanbo Xie","Tianyun Liu","Yingjie Zhang","Suchen Liu","Yulin Li","Liya Su","Tingwen Liu"],"year":2026,"abstract":"Modern agentic systems transform LLMs from session-bounded assistants into stateful systems that persist and evolve shared world state across sessions through memories, filesystems, tools, and other long-lived contextual artifacts. This shift fundamentally expands the attack surface of prompt injection. However, prior works on prompt injection have largely focused on model-level threats within a single session, overlooking how cross-session persistent system state fundamentally changes the syste","url":"https://arxiv.org/abs/2606.04425","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00426","type":"paper","title":"Domain-Conditioned Safety in Frontier Computer-Using Agents: A 793-Episode Browser Benchmark, a Coding-Domain Cross-Reference, and a Reproducibility Audit of Recent Red-Teaming","authors":["Nicholas Saban"],"year":2026,"abstract":"Recent computer-using-agent (CUA) red-teaming papers report prompt-injection attack success rates (ASR) of 42-98%, but these headline numbers cluster on retired models and on the most-vulnerable model in each paper's panel. We ask whether those techniques, reproduced as hand-crafted templates, still work against current frontier CUAs. We release CUA-HandCrafted, a public benchmark of 793 episodes spanning 24 multi-step web tasks, 56 attack templates, 8 attack families, and 4 system-prompt config","url":"https://arxiv.org/abs/2606.05233","categories":["prompt-injection","red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00427","type":"paper","title":"Caught in the Act(ivation): Toward Pre-Output and Multi-Turn Detection of Credential Exfiltration by LLM Agents","authors":["Kargi Chauhan","Pratibha Revankar"],"year":2026,"abstract":"LLM agents often place sensitive credentials in the same context window as untrusted retrieved content, creating a direct path for indirect prompt injection to induce credential exfiltration. We study this failure mode through three complementary defenses. First, we ask whether activation probes can detect credential access before output tokens are emitted. Second, we construct honeytokens from format-specific character models and calibrate detection with split conformal prediction. Third, we tr","url":"https://arxiv.org/abs/2606.04141","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00428","type":"paper","title":"NeuroArmor: Safe-Variant-Guided Representation Consistency for Selective Re-Anchoring in Jailbreak Defense","authors":["Zhongyang Lin","Ziran Zhao","Feifei Zhai","Pengyuan Liu"],"year":2026,"abstract":"Large language models remain vulnerable to jailbreak attacks that hide harmful intent behind seemingly ordinary requests such as role-play, translation, encoding, adversarial suffixes, and multi-turn buildup. Existing defenses still struggle to handle these attacks without over-blocking benign but sensitive requests, partly because they often apply the same action to every prompt and therefore fail to balance safety and helpfulness. We propose NeuroArmor, a white-box runtime defense that uses pr","url":"https://arxiv.org/abs/2606.03486","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00429","type":"paper","title":"PsychoPass: Geometric Profiling of Multi-Turn Adversarial LLM Conversations","authors":["Muberra Ozmen","Subhabrata Majumdar"],"year":2026,"abstract":"Multi-turn jailbreak attacks on large language models (LLMs) reveal a mismatch in current guardrails: they operate on individual turns, while attacks unfold as trajectories across conversations. We propose a shift from content to dynamics, modeling conversations as paths in representation space and asking whether adversarial intent is encoded early in their geometry. We introduce PsychoPass, a framework that extracts geometric features from conversation trajectories in embedding space to predict","url":"https://arxiv.org/abs/2606.03136","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00430","type":"paper","title":"Patcher: Post-Hoc Patching of Backdoored Large Language Models","authors":["Anjun Gao","Yueyang Quan","Yufei Xia","Zhuqing Liu","Minghong Fang"],"year":2026,"abstract":"Large language models remain vulnerable to jailbreak backdoor attacks, where adversaries poison safety alignment data to embed hidden triggers that bypass safety mechanisms. Existing defenses often require comprehensive attack information or multiple triggered examples, making them impractical when defenders only observe a single reported failure case without knowing whether it stems from a backdoor attack or a natural alignment bug. This paper presents Patcher, a post-hoc defense framework that","url":"https://arxiv.org/abs/2606.02995","categories":["jailbreaking","data-poisoning","guardrails"],"reviewed":false},{"id":"llmsec-2026-00431","type":"paper","title":"Which Defense Closes Which Threat? Attributing OWASP-LLM-Top-10 Coverage and Its Brittleness Under Paraphrasing","authors":["Alexandre Cristovão Maiorano"],"year":2026,"abstract":"Production LLM applications stack several defense families -- refusal-phrase filters, token-budget controls, model allowlists, rate limits, tool-registry authentication -- yet existing breach-and-attack-simulation (BAS) benchmarks report a single aggregate coverage number, hiding which family closes which threat. We measure attribution. We add four OWASP-LLM-Top-10-aware agents to a 21-agent baseline scanner and target a lattice of four synthetic LLM endpoints: $L_0$ (no defenses), $L_1$ (refusa","url":"https://arxiv.org/abs/2606.02822","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00432","type":"paper","title":"AgentRedBench: Dynamic Redteaming and Integration-Aware Defense for LLM Agents over SaaS Integrations","authors":["Hiskias Dingeto","William Leeney"],"year":2026,"abstract":"Indirect prompt injection in tool-use agents is a concrete production threat: LLM agents read from integrations (third-party services such as Gmail, Salesforce, or Jira accessed through tool calls) whose response content the user neither writes nor controls. Existing benchmarks under-measure the threat: most cover only a handful of integrations with the same attack payload replayed across runs, and open-source guards are trained on chat-style data rather than tool-response content. We introduce ","url":"https://arxiv.org/abs/2606.02240","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00433","type":"paper","title":"Benign Inputs, Harmful Outputs: Cross-Modal Jailbreaking via Distributed Semantic Recomposition","authors":["Yani Wang","Yilong Yang","Yang Liu","Zhuzhu Wang","Zuobin Ying","Zhuo Ma"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) have recently demonstrated remarkable capabilities in content synthesis and autonomous reasoning. Previous safety guardrails are primarily designed for unimodal textual input interception, leaving them vulnerable to cross-modal jailbreak attacks. However, regardless unimodal textual attack or cross-modal jailbreak, typically inclusive part of explicit harmful or sensitive content at the input level, which is called Harm-Bearing. It allow the model's safet","url":"https://arxiv.org/abs/2606.01837","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00434","type":"paper","title":"D-Judge: Disrupting Multi-Turn Jailbreaks using Semantics-Preserving Output Rewriting","authors":["Huanli Gong","Zhipeng Wei","Yu Fu","Haz Sameen Shahgir","Ananya Gupta","Yue Dong","N. Benjamin Erichson"],"year":2026,"abstract":"Multi-turn jailbreak attacks pose a growing threat to large language model (LLM) safety because they exploit feedback from auxiliary judge models to iteratively refine prompts toward harmful goals. Existing defenses largely detect or block unsafe content at individual turns or at the final response, leaving the judge-driven refinement loop intact and allowing attackers to extract informative feedback from intermediate interactions. We introduce D-Judge, a semantics-preserving output rewriting de","url":"https://arxiv.org/abs/2606.02640","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00435","type":"paper","title":"Confused ChatGPT: Cross-App Context Poisoning via First-Party APIs","authors":["Chao Wang","Somesh Jha","Zhiqiang Lin"],"year":2026,"abstract":"ChatGPT Apps, launched by OpenAI on Oct. 6, 2025, introduce an app-in-app paradigm in which third-party applications share a single chat context with the user and with every other connected app. The ecosystem grew from 122 apps in Dec. 2025 to 888 by May 2026, yet its security has remained uninvestigated. We identify cross-app context poisoning, a variant of indirect prompt injection distinguished by three properties: 1) the injection persists in the shared chat context across turns; 2) the effe","url":"https://arxiv.org/abs/2606.00485","categories":["prompt-injection","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00436","type":"paper","title":"Stateful Online Monitoring Catches Distributed Agent Attacks","authors":["Davis Brown","Samarth Bhargav","Arav Santhanam","Kasper Hong","Ivan Zhang","Matan Shtepel","Steffi Chern","Alexander Robey","Eric Wong","Hamed Hassani"],"year":2026,"abstract":"Language models can find thousands of severe software vulnerabilities, and agents are increasingly being misused for cyberattacks. To avoid detection, attackers frequently distribute their misuse, splitting a harmful task across many user accounts so each individual transcript looks benign. Because safety monitors score only one agent context at a time, they are structurally blind to misuse that is only visible in aggregate, across many accounts. We show this gap is real by building, to our know","url":"https://arxiv.org/abs/2605.31593","categories":["agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00437","type":"paper","title":"From Prompt Injection to Persistent Control: Defending Agentic Harness Against Trojan Backdoors","authors":["Jiejun Tan","Zhicheng Dou","Xinyu Yang","Yuyang Hu","Yiruo Cheng","Xiaoxi Li","Ji-Rong Wen"],"year":2026,"abstract":"LLM agents are evolving from conversational chatbots to operational tools in real-world workspaces. In local agentic harnesses, an LLM can read and write files, call tools, and reuse workspace state across sessions. While such capabilities enhance utility, they also expose a new attack surface for attackers. Attackers can embed a prompt injection within a file or tool output. Agents may read this hidden instruction, store it, and execute it later. In this multi-step trojan attack paradigm, no in","url":"https://arxiv.org/abs/2605.31042","categories":["prompt-injection","data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00438","type":"paper","title":"TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking","authors":["Churui Zeng","Weiwei Qi","Kedong Xiu","Tianhang Zheng","Chaochao Lu","Liang He","Zhan Qin","Kui Ren"],"year":2026,"abstract":"The rise of LLM agents introduces a new threat by enabling planning, coding, and even end-to-end execution of expert-level attack workflows. However, this threat remains underexplored and underestimated since (i) safety alignment prevents LLMs from directly generating harmful instructions, and (ii) most existing jailbreak methods cannot consistently induce agents to execute malicious operations. In this paper, we propose TRACE, a practical agentic jailbreaking framework to further reveal the ris","url":"https://arxiv.org/abs/2605.30883","categories":["jailbreaking","agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00439","type":"paper","title":"Depth-Dependent Indirect Prompt Injection in Tool-Calling ReAct Agents: Injection Depth, Payload Framing, and Turn-Budget Sensitivity","authors":["Mohammadreza Rashidi"],"year":2026,"abstract":"ReAct agents that interleave chain-of-thought reasoning with tool calls are increasingly deployed for real tasks such as scheduling, file retrieval, and data access. Their tool observation loop creates a direct attack surface: an adversary who controls any tool's return value can embed instructions that redirect the agent away from the user's goal, a threat known as indirect prompt injection. Existing benchmarks evaluate attack success rate (ASR) at a fixed injection position under fixed conditi","url":"https://arxiv.org/abs/2605.30686","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00440","type":"paper","title":"Investigating Detection and Obfuscation of Prompt Injection Attacks Against Software Reverse Engineering AI Agents","authors":["Brian Crawford","Patrick McClure"],"year":2026,"abstract":"Agentic software reverse engineering systems are vulnerable to prompt injection attacks placed into the source code of executable binary files. This research demonstrates defensive tactics for detecting the presences of prompt injection strings in the decompiler output of adversarial example programs. Methods for obfuscating these attacks and subsequent methods for defending against these obfuscations are also explored. This research advances the understanding of risk and security of agentic sof","url":"https://arxiv.org/abs/2605.30677","categories":["prompt-injection","adversarial-examples","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00441","type":"paper","title":"Automatically Attacking Software Reverse Engineering AI Agents","authors":["Brian Crawford","Justin Phillips","Patrick McClure"],"year":2026,"abstract":"Software tools for reverse engineering executable binary files, such as Ghidra, enable malware analysts to safely conduct robust static analysis without having access to original source code. Coupled with the analytic power of large language models (LLM), agentic systems enabled with tools, such as GhidraMCP, can allow analysts to automate a previously human driven process. Although this automation can increase the productivity of a single malware analyst, it also introduces a new area of vulner","url":"https://arxiv.org/abs/2605.30667","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00442","type":"paper","title":"Strengthening Polymorphic Prompt Assembling: Dynamic Separator Generation Against Emerging Prompt Injection Attacks","authors":["Nima Dorzhiev","Peng Liu"],"year":2026,"abstract":"Polymorphic Prompt Assembling (PPA) defends LLM agents against prompt injections by randomly selecting separator pairs from a fixed pool to isolate user input from system instructions. Although effective, static pool reuse exposes a blast-radius vulnerability: once a separator leaks, it can be exploited in future requests. We propose a dynamic per-request separator generation using domain-separated SHA-256 digests keyed on the timestamp, session identifier, and cryptographic nonce. Each assemble","url":"https://arxiv.org/abs/2605.30534","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00443","type":"paper","title":"The Surface You Test Is Not the Surface That Breaks","authors":["Shifat E Arman","Syed Nazmus Sakib","Nafiul Haque","Shahrear Bin Amin"],"year":2026,"abstract":"Tool-augmented LLM agents are vulnerable to prompt injection: a third party who controls part of the agent's context can plant instructions that the agent then executes as if they came from the user. Current evaluations report a single attack success rate per model on one channel, the tool output and treat that number as the model's vulnerability. But tool descriptions, which the agent reads at every turn before any tool is called, are themselves an injection surface that the attacker can choose","url":"https://arxiv.org/abs/2605.30454","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00444","type":"paper","title":"MultiTurnPSB: Evaluating Multi-Turn Jailbreak Attacks an dClassifier-Based Defenses for Medical AI Safety","authors":["Anushka Sheoran","Yiduo Hao"],"year":2026,"abstract":"Patient-facing medical chatbots are commonly evaluated on single-turn prompts, yet real users push back after refusals, add urgency, and invoke authority. We introduce MultiTurnPSB, a four-turn adversarial extension of PatientSafetyBench, and evaluate GPT-4.1-mini under fixed template, template-adaptive, and live adversarial attacks. Unsafe responses rise from 35% to nearly 80% by Turn 4 under live attack. Under the same adversary, GPT-4.1-mini and Claude Sonnet 4.5 are statistically indistingui","url":"https://arxiv.org/abs/2606.02630","categories":["jailbreaking","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00445","type":"paper","title":"Steering Vectors are an Adversarial Attack Surface","authors":["Abzal Aidakhmetov","Donato Crisostomi","Tommaso Mencattini","Adrian Robert Minut","Iacopo Masi","Emanuele Rodolà"],"year":2026,"abstract":"Activation steering has become a popular way to control Large Language Model (LLM) behavior without fine-tuning. Since the technique is plug-and-play, users share datasets and precomputed vectors to steer model activations. However, we show that a \\emph{stealth data poisoning attack} silently compromises this pipeline. By substituting $4{-}6\\%$ of tokens in the steering dataset, an attacker can silently align the resulting vector with an anti-refusal direction. This jailbreaks the target model w","url":"https://arxiv.org/abs/2606.05958","categories":["jailbreaking","data-poisoning","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00446","type":"paper","title":"BYORn: Bootstrap Your Own Responses to Defend Large Vision-Language Models Against Backdoor Attacks","authors":["Ivan Sabolić","Marin Oršić","Josip Šarić","Sven Lončarić"],"year":2026,"abstract":"Supervised fine-tuning is the predominant approach for adapting autoregressive vision-language models to downstream tasks. Recent work has shown that this paradigm is highly vulnerable to backdoor attacks, and that existing defenses are ineffective in open-ended generation settings. In response, we propose BYORn, a backdoor-robust fine-tuning framework motivated by the observation that poisoned target responses are often semantically implausible given the corresponding image-text inputs and a pr","url":"https://arxiv.org/abs/2606.02947","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00447","type":"paper","title":"A-Live: Passive Liveness Detection via Neuromuscular Micro-Motion Signatures on Commodity Sensors","authors":["Mohammed Gharib","Sam Burns","Martin Zizi"],"year":2026,"abstract":"Liveness detection has evolved from a safeguard against presentation and replay attacks in biometric authentication to a broader requirement for distinguishing human users from non-human agents in modern digital systems. The emergence of generative and agentic AI further amplifies this need, positioning liveness as a fundamental security primitive. Existing approaches face key limitations, including reliance on explicit user interaction, specialized hardware, vulnerability to increasingly realis","url":"https://arxiv.org/abs/2606.05126","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00448","type":"paper","title":"The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents","authors":["Manvendra Modgil"],"year":2026,"abstract":"As autonomous AI agents move from conversational systems to long-horizon software execution, runtime safety layers that decide when to interrupt an agent have become essential. We study this timing problem using a continuous 18-dimensional affective-dynamics engine (HEART) as a diagnostic probe, evaluating four intervention trigger families - absolute state thresholds, composite state-action patterns, regex reasoning-feature extraction, and zero-shot LLM-as-judge - against human-annotated interv","url":"https://arxiv.org/abs/2606.04296","categories":["autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00449","type":"paper","title":"The Impact of Configuring Agentic AI Coding Tools on Build-vs-Buy Decisions: A Study Protocol","authors":["Jai Lal Lulla","Matthias Galster","Jie M. Zhang","Sebastian Baltes","Christoph Treude"],"year":2026,"abstract":"Agentic AI coding tools write code with increasing autonomy and in doing so decide when to import a library and when to implement functionality from scratch. These decisions, whether to build functionality from scratch or buy into an external library, hereafter build-versus-buy, carry direct consequences for software security, licensing compliance, performance, and long-term maintainability. Yet no controlled experimental study has examined what governs build-versus-buy decisions in agentic AI c","url":"https://arxiv.org/abs/2606.03907","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00450","type":"paper","title":"Agentic Relationship Harm: Benchmarking and Gating Relational Manipulation in AI Agents","authors":["Pei-Sze Tan","Tasuku Igarashi","Isao Echizen"],"year":2026,"abstract":"AI agents built on large language models can assist not only legitimate tasks but also relational manipulation. AI agents can be used to help a user maintain a deceptive identity, intensify emotional dependency, isolate a target, or prepare for later extraction. We conceptualise this risk as agentic relationship harm: workflow-level assistance that can exploit recipient vulnerability, persuasive influence, and relational power asymmetry. Existing safety evaluations and generic guardrails often t","url":"https://arxiv.org/abs/2606.03271","categories":["agentic-threats","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00451","type":"paper","title":"Toward Pre-Deployment Assurance for Enterprise AI Agents: Ontology-Grounded Simulation and Trust Certification","authors":["Thanh Luong Tuan","Abhijit Sanyal"],"year":2026,"abstract":"Pre-deployment verification of enterprise artificial intelligence (AI) agents remains a critical gap between large language model (LLM) capability benchmarking and production deployment. Post-deployment monitoring, human-in-the-loop controls, and prompt-level guardrails offer limited assurance once an agent is operating in production. We present an ontology-grounded verification framework -- to our knowledge the first to combine three components: an Agent Operational Envelope formalizing the cer","url":"https://arxiv.org/abs/2606.04037","categories":["guardrails","monitoring-detection","benchmarks","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00452","type":"paper","title":"Discovering Agents for Discovery: The Case for DNS","authors":["Ramachandra Rao Seethiraju","Sameer Thakar","Karthik Shyamsunder","Eric Osterweil"],"year":2026,"abstract":"As Artificial Intelligence (AI) agents enter their next stage of being deployed ubiquitously throughout the Internet, their discoverability will become a central challenge. The information AI agents need to discover one another, how they will locate it, how to facilitate authentication, integrity, and authorization, how to connect across different platforms, and how to scale across organizational boundaries form a set of unanswered challenges that deployment success will prompt. These are challe","url":"https://arxiv.org/abs/2606.02314","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00453","type":"paper","title":"Cross-Vendor Sola ISPM Benchmark: Evaluating Agentic AI for Federated Identity Security Reasoning","authors":["Eden Yavin","Gal Engelberg","Konstantin Koutsyi","Leon Goldberg","Gal Baron"],"year":2026,"abstract":"The rapid proliferation of multi-cloud and SaaS platforms has transformed Identity Security Posture Management (ISPM) into a fundamentally cross-vendor challenge: critical misconfigurations and privilege escalation paths increasingly span multiple identity providers, infrastructure layers, and authentication systems never designed to interoperate. Existing evaluations focus on isolated single-platform environments and provide no means to assess whether an AI agent can reason across these fragmen","url":"https://arxiv.org/abs/2606.02674","categories":["agentic-threats","benchmarks","cloud-ai-security"],"reviewed":false},{"id":"llmsec-2026-00454","type":"paper","title":"Agent Operating Systems (AOS): Integrating Agentic Control Planes into, and Beyond, Traditional Operating Systems","authors":["Ankur Sharma","Deep Shah"],"year":2026,"abstract":"Traditional operating systems were designed around deterministic programs, explicit control flow, and human initiated workflows. Their core abstractions processes, threads, system calls, files, and permissions assume bounded behavior and predictable interaction patterns. Agentic AI systems introduce a different execution model: long-lived, goal-directed entities that reason probabilistically, invoke tools dynamically, and adapt behavior based on feedback. While agents can be implemented as user-","url":"https://arxiv.org/abs/2606.01508","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00455","type":"paper","title":"A New Framework for Cybersecurity Refusals in AI Agents","authors":["Eliot Krzysztof Jones","Mateusz Dziemian","Matt Fredrikson","J Zico Kolter"],"year":2026,"abstract":"Agentic scaffolds have dramatically improved LLM performance on complex, long-horizon tasks, yielding both broad benefits and amplified risks in domains like cybersecurity. Existing benchmarks for AI agents in cybersecurity focus mainly on measuring proficiency--how effectively agents can complete offensive security tasks--but neglect a critical question: when and how should agents refuse harmful requests? We present the first framework for establishing refusal boundaries in offensive security c","url":"https://arxiv.org/abs/2606.02644","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00456","type":"paper","title":"When Safe Skills Collide: Measuring Compositional Risk in Agent Skill Ecosystems","authors":["Su Wang","Pin Qian","Yihang Chen","Junxian You","Xiaoyuan Wang","Xiaochong Jiang","Lifei Liu","Haoran Yu","Jingzhou Xu"],"year":2026,"abstract":"LLM agents increasingly rely on community-contributed skills that expand an agent's operational capability set. We study a core safety problem in agentic AI systems: whether individually safe skills can compose into unsafe installed skill sets. We present SkillReact, a compositional security measurement framework with three components: a deterministic static-composition benchmark, a two-rater LLM-assisted human-adjudication pipeline, and an action-based exploitability harness. On 1,520 ClawHub s","url":"https://arxiv.org/abs/2606.00448","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00457","type":"paper","title":"From Frontier to Shadow AI: A Simmering Threat to Assurance and Security in Critical Infrastructure","authors":["Mohan Baruwal Chhetri","Shahroz Tariq","Tooba Aamir","Marthie Grobler","Chandra Thapa","Ronal Singh"],"year":2026,"abstract":"Frontier AI systems, including large language models and emerging agentic AI tools, offer significant operational benefits but present unique challenges to critical infrastructure (CI) environments due to their non-deterministic and emergent properties. While formal adoption is inherently cautious and tightly controlled due to strict regulatory oversight, widespread accessibility has catalysed shadow AI: the unsanctioned use of frontier AI outside established organisational controls. In CI setti","url":"https://arxiv.org/abs/2606.00088","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00458","type":"paper","title":"When Agents Talk: Discourse, Manipulation, and Risk in an Agentic Social Network","authors":["10a Labs"," :","Grace Cheong","Violet Davis","Juliette Garcia","Kendal Gee","Molly Hart","Nicholas Hayes","Henry Houghton","Kyle Lee","Paige Lee","Vicky Lee","Hailey May","Bobby McKenzie","Christine McNeill","Han Nguyen","Brooke Perreault","David Pham","Charlie Plumb","Olivia Quill","Matthew Swain","Grace Wang","Adam Warren","Corie Wieland","Zachary Yahn"],"year":2026,"abstract":"AI agents are increasingly interacting within shared online environments, creating new operational security risks. We analyze activity on Moltbook, a Reddit-style social platform where AI agents--typically configured and overseen by human operators--post and interact with one another at scale. Using a dataset of 228,684 posts produced by more than 39,500 accounts over a seventeen-day observation window, we combine semantic clustering of high-engagement posts with LLM-assisted classification of h","url":"https://arxiv.org/abs/2606.00067","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00459","type":"paper","title":"A Unified Evaluation Framework for Utility and Privacy Risks of LLM-Generated Synthetic Text Data","authors":["Lubana Isaoglu","Zeynep Orman"],"year":2026,"abstract":"The increasing use of Large Language Models (LLMs) has enabled the generation of high-quality synthetic text, providing a potential alternative to sensitive real-world datasets in domains where privacy concerns limit data sharing. However, synthetic text is not inherently privacy safe. Fine-tuning generative models on domain-specific data can enhance semantic fidelity while simultaneously increasing the risk of memorization and information leaks. In this work, we propose a ","url":"https://doi.org/10.64808/engineeringperspective.1910777","categories":["membership-inference","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00460","type":"paper","title":"Can Open-Source LLM Agents Replace Static Application Security Testing Tools? An Empirical Assessment","authors":["Derek Yohn","Luke Flancher","Mirajul Islam","Khaled Slhoub"],"year":2026,"abstract":"This paper explores the value of agentic AI tools for cybersecurity purposes. We evaluate the efficacy of a general-purpose GenAI Large Language Model- (GenAI-) based agent when powered by three different Ollama-hosted general-purpose open source models. We assess each agent's performance using precision, recall, false positive count, and a calculated composite score based upon the interplay of the captured metrics, against the baseline performance of an existing, vetted Static Application Secur","url":"https://arxiv.org/abs/2606.11672","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00461","type":"paper","title":"MPC-Patch-Bench: Security-Aware LLM Code Patch for Multi-Party Computation","authors":["Yukuan Zhang","Mengxin Zheng","Qian Lou"],"year":2026,"abstract":"Repository-level benchmarks for evaluating Large Language Model (LLM) code repair on Secure Multi-Party Computation (MPC) software do not yet exist, and directly transplanting general-purpose benchmarks such as SWE-bench fails on three structural fronts: (i) MPC repositories are dominated by generic Python infrastructure rather than cryptographic logic; (ii) high-value MPC fixes lack the standardized tests rigid extraction pipelines require; and (iii) standard fail-to-pass evaluation is insuffic","url":"https://arxiv.org/abs/2606.11416","categories":["cryptographic-controls","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00462","type":"paper","title":"Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation","authors":["Yuchen Ling","Shengcheng Yu","Zhenyu Chen","Chunrong Fang"],"year":2026,"abstract":"Large language model (LLM) agents are rapidly moving from conversational interfaces to software components that plan, invoke tools, maintain memory, and act on external environments. This transition changes the nature of security risk. In agentic settings, failures are no longer limited to unsafe text generation. Untrusted content may redirect control flow, misuse tool privileges, corrupt persistent state, leak sensitive information, or trigger harmful external actions. At the same time, researc","url":"https://arxiv.org/abs/2606.10749","categories":["membership-inference","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00463","type":"paper","title":"Benchmarking and Exploring the Capabilities of LLMs for Attack Investigations","authors":["Aniket Anand","Yiwei Hou","Daniel Fields","Alex Kantchelian","David Tao","Kurt Thomas","Grant Ho"],"year":2026,"abstract":"This paper presents AuditBench, a new benchmark dataset for evaluating the capabilities of LLMs at investigating security-related system audit logs. We design and use this benchmark to explore the performance of LLMs on four log-investigation tasks that incident response teams commonly perform, ranging from triaging alerts generated by detectors to identifying persistence mechanisms on compromised systems. AuditBench consists of system audit logs collected from Linux and Windows machines, and sp","url":"https://arxiv.org/abs/2606.10281","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00464","type":"paper","title":"Alignment Defends LLMs from Property Inference Attacks","authors":["Pengrun Huang","Chhavi Yadav","Ruihan Wu","Kamalika Chaudhuri"],"year":2026,"abstract":"Large language models (LLMs) are increasingly fine-tuned on domain-specific datasets that may contain sensitive, dataset-level properties. Recent work has shown that such dataset-level information can be effectively extracted through property inference attacks, posing a confidentiality risk. Existing defenses against these attacks primarily operate by modifying the training data distribution and hence require access to the original data and retraining the model, limiting their applicability to s","url":"https://arxiv.org/abs/2606.10217","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00465","type":"paper","title":"Context-Fractured Decomposition Attacks on Tool-Using LLM Agents: Exploiting Artifact Provenance Gaps","authors":["Xiaofeng Lin","Yukai Yang","Daniel Guo","Sahil Arun Nale","Charles Fleming","Guang Cheng"],"year":2026,"abstract":"Tool-using LLM agents interact with the world through actions that persist state in artifacts (e.g., workspace files or logs). Consequently, jailbreak defenses must reason about cross-step composition rather than isolated text. Yet most existing attacks and defenses, including ``multi-turn'' jailbreaks such as Crescendo and Tree of Attacks,still assume a single contiguous conversation visible to the defender. This assumption breaks down in real agent pipelines, where enforcement is fragmented ac","url":"https://arxiv.org/abs/2606.09084","categories":["jailbreaking","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00466","type":"paper","title":"Data Agents Under Attack: Vulnerabilities in LLM-Driven Analytical Systems","authors":["Kuncan Wang","Ziting Wang","Peizhuo Lv","Haoyang Li","Guoliang Li","Gao Cong","Wei Dong"],"year":2026,"abstract":"Data agents integrate LLM-driven reasoning with relational data access, executable analytical tools, and multi-step workflow orchestration, making them increasingly central to enterprise analytics. This integration introduces new security vulnerabilities across data resources, database execution, and agent reasoning, recombining concerns from database security and general-purpose LLM-agent security into failure modes that neither line of work captures on its own. To address this gap, we present ","url":"https://arxiv.org/abs/2606.08661","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00467","type":"paper","title":"SGTO-MAS: Secure Gorilla Troops Optimization for Multi-Agent LLM Systems","authors":["Saeid Jamshidi"],"year":2026,"abstract":"Multi-agent large language model (LLM) systems offer strong capabilities for complex reasoning and decision-making, yet coordination across agents introduces error propagation, security risks, and inefficient use of resources. Existing methods often rely on heuristic, static strategies and lack a principled mechanism for balancing performance, security, and computational cost. This paper formulates multi-agent LLM coordination as a constrained optimization problem and proposes a security-aware m","url":"https://arxiv.org/abs/2606.07940","categories":["agent-architecture","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00468","type":"paper","title":"Beyond Pass/Fail: Using Process Mining to Understand How LLMs Resist (and Fail) Red Team Attacks","authors":["Zvi Topol"],"year":2026,"abstract":"Standard AI red teaming evaluations reduce adversarial campaigns to a single binary outcome, attack success rate (ASR), not taking into account the sequential structure of how models resist or yield to attacks. We propose applying process mining, a discipline for discovering and analyzing process models from event logs, to red teaming traces. We conduct a controlled experiment pitting 60 HarmBench prompts against two LLMs, GPT-OSS 120B and Llama 3.3 70B, using 10 prompt mutation strategies over ","url":"https://arxiv.org/abs/2606.07833","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00469","type":"paper","title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","authors":["Hangtao Zhang","Yucheng Zhao","Sishun Liu","Ziqi Zhou","Zeyu Ye","Wei Wan","Minghui Li","Shengshan Hu","Yanjun Zhang","Yi Liu","Leo Yu Zhang"],"year":2026,"abstract":"Jailbreak prompts can bypass alignment guardrails in large language models (LLMs) and elicit unsafe outputs, making reliable deployment-time detection critical. Prior detection approaches largely rely on a fixed metric space, e.g., raw inputs, gradients, or hidden features, in which benign and jailbreak prompts are linearly separable. We show this assumption breaks under (i) pseudo-malicious prompts that are benign by intent but contain safety-related keywords, and (ii) adaptive attacks that exp","url":"https://arxiv.org/abs/2606.07335","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00470","type":"paper","title":"The Security Budget of Code-LLM Prompt Hardening: Provable Limits Under Pass-Only Acceptance","authors":["Jianwei Tai"],"year":2026,"abstract":"We give a quantitative impossibility result for pass-only prompt hardening of code LLMs. For any deterministic prompt filter $h$ and a registered family of finite executable-equivalence task variables $\\mathcal Y_{\\mathrm{exec}}$, the shared filtered-prompt channel $\\rmI(h(p);h(\\tilde p))$ is lower-bounded by a worst-$Y$ Fano floor; on HumanEval and MBPP the universal pass-only floor evaluates to $\\mathcal F^{\\mathrm{op}}\\ge 0.84$ and $1.20$ nats at $η=0.05$ task-collapse tolerance, and the iden","url":"https://arxiv.org/abs/2606.03308","categories":["input-filtering"],"reviewed":false},{"id":"llmsec-2026-00471","type":"paper","title":"Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents","authors":["Zihao Wang","Yiming Li","Yutong Wu","Zheyu Liu","Kangjie Chen","Fok Kar Wai","Pin-Yu Chen","Vrizlynn L. L. Thing","Bo Li","Dacheng Tao","Tianwei Zhang"],"year":2026,"abstract":"Web agents driven by large language models (LLMs) are increasingly deployed in real-world environments, where they operate over untrusted web content and execute actions with direct consequences. This makes them vulnerable to prompt-injection attacks, in which seemingly benign content embeds adversarial instructions that manipulate agent behaviour. Existing security benchmarks adopt an \\textit{attack-centric} perspective, focusing on the technical feasibility of injections while overlooking the ","url":"https://arxiv.org/abs/2606.13385","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00472","type":"paper","title":"PI-Hunter: Automated Red-Teaming for Exposing and Localizing Prompt Injections","authors":["Pengfei He","Lesly Miculicich","Vishesh Sharma","Ash Fox","George Lee","Jiliang Tang","Tomas Pfister","Long T. Le"],"year":2026,"abstract":"Large Language Models (LLMs) are rapidly evolving into agentic systems that interact with external tools and environments, introducing new security risks such as indirect prompt injection attacks through untrusted external sources. Existing defenses mainly focus on blocking malicious content at inference time, and current red-teaming methods primarily optimize attack success. As a result, developers have limited visibility into how latent prompt injections emerge and propagate through agents. We","url":"https://arxiv.org/abs/2606.12737","categories":["prompt-injection","agentic-threats","red-teaming","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00473","type":"paper","title":"Smarter Saboteurs, Better Fixers: Scaling & Security in Linear Multi-Agent Workflows","authors":["Timothy McAllister","Sina Abdidizaji","Ivan Garibay","Ozlem Ozmen Garibay"],"year":2026,"abstract":"As LLM-based multi-agent systems (MAS) are deployed in the wild, the resilience of their collaboration structures against adversarial compromise becomes a critical safety concern. Attackers may leverage prompt-injection or jailbreaking to sabotage individual agents within MAS workflows, but the interaction between model scaling and system-level resilience remains poorly understood. This paper investigates how model scale affects the security of linear multi-agent workflows. Our experiments acros","url":"https://arxiv.org/abs/2606.12709","categories":["prompt-injection","jailbreaking","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00474","type":"paper","title":"Grammar-Constrained Decoding Can Jailbreak LLMs into Generating Malicious Code","authors":["Yitong Zhang","Shiteng Lu","Jia Li"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly used for code generation, raising concerns that they may be misused to produce malicious code. Meanwhile, Grammar-Constrained Decoding (GCD) has been widely adopted to improve the reliability of LLM-generated code by enforcing syntactic validity. In this paper, we reveal a counterintuitive risk: this reliability-oriented technique can itself become an attack surface. We uncover a new jailbreak attack, termed CodeSpear, that exploits GCD to induce LLM","url":"https://arxiv.org/abs/2606.11817","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00475","type":"paper","title":"JailbreakOPT: Tool-Assisted Iterative Jailbreak Prompt Optimization","authors":["Ge Shi","Jun Yin","Donglin Xie","Fangyi Liu","Yucan Li","Menglin Liu"],"year":2026,"abstract":"Jailbreak attacks expose persistent safety weaknesses in large language models (LLMs), but existing stateless single-turn methods face a trade-off: hand-crafted prompts are expressive but static, while iterative prompt optimization can adapt but often relies on low-level mutations that require many target queries. We propose JailbreakOPT, a tool-assisted framework for improving iterative single-turn jailbreak prompt optimization. JailbreakOPT organizes diverse atomic jailbreak prompts into an at","url":"https://arxiv.org/abs/2606.11425","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00476","type":"paper","title":"Risk Under Pressure: Compute-Aware Evaluation of Adversarial Robustness in Language Models","authors":["Malikeh Ehghaghi","Boglárka Ecsedi","Marsha Chechik","Colin Raffel"],"year":2026,"abstract":"Adversarial robustness evaluations of large language models (LLMs) typically report attack success rate (ASR) under fixed query budgets, implicitly treating all attacks as equally costly. In practice, the computational expense of different attack strategies can vary by orders of magnitude. Consequently, ASR at a fixed budget can obscure the true effort required to jailbreak a model, thereby making it hard to determine whether an attack's cost justifies its payoff to the attacker. We propose a co","url":"https://arxiv.org/abs/2606.11409","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00477","type":"paper","title":"Training LLMs to Enforce Multi-Level Instruction Hierarchies via Gravity-Weighted Direct Preference Optimization","authors":["Lena S. Bolliger","Lena A. Jäger"],"year":2026,"abstract":"Production LLMs receive instructions from sources with very different levels of trust, yet attend to every token with uniform architectural privilege. This is the structural vulnerability that enables malicious prompt injections and, more broadly, leaves models without a principled way to resolve conflicts between legitimate but competing instructions. A common training-based response is to teach models an explicit instruction hierarchy; existing approaches, however, formalize hierarchies of onl","url":"https://arxiv.org/abs/2606.10860","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00478","type":"paper","title":"Assessing Automated Prompt Injection Attacks in Agentic Environments","authors":["David Hofer","Edoardo Debenedetti","Florian Tramèr"],"year":2026,"abstract":"Indirect prompt injection poses a critical threat to LLM agents that interact with untrusted external data, yet automated attack methods--proven effective for jailbreaking--remain underexplored in realistic agentic settings. We present a comprehensive empirical evaluation of automated prompt injection attacks against LLM agents, adapting both white-box (GCG) and black-box (TAP) methods to the agentic setting within the AgentDojo framework. We evaluate across 80 task pairs spanning four domains a","url":"https://arxiv.org/abs/2606.10525","categories":["prompt-injection","jailbreaking","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00479","type":"paper","title":"Game-Theoretic Multi-Agent Control for Robust Contextual Reasoning in LLMs","authors":["Saeid Jamshidi","Amin Nikanjam","Arghavan Moradi Dakhel","Kawser Wazed Nafi","Foutse Khomh"],"year":2026,"abstract":"Large Language Models (LLMs) in multi-turn interactions maintain evolving context rather than generating isolated responses, making them vulnerable to prompt-injection and context-poisoning attacks in which locally plausible adversarial fragments gradually distort reasoning trajectories. Existing defenses mainly filter individual outputs and often ignore context evolution across turns, leaving long-horizon reasoning exposed. Although the Model Context Protocol (MCP) standardizes context exchange","url":"https://arxiv.org/abs/2606.10322","categories":["prompt-injection","data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00480","type":"paper","title":"Brain-Prompt Injection: A Route-Safety Audit for BCI-LLM Agents","authors":["Jianwei Tai"],"year":2026,"abstract":"BCI-to-agent pipelines turn decoded neural activity into an authorization channel for tool-use agents, exposing a new attack surface we call \\emph{brain-prompt injection}: signal-side perturbations, context-only injections, and adaptive dual-decoder attacks can all change the routed action while EEG-side or text-side monitors remain blind. Route safety in this stack depends on what the audit log can observe, not on decoder accuracy or agreement alone. We define a Route-Safety Audit Contract: a m","url":"https://arxiv.org/abs/2606.09315","categories":["prompt-injection","agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00481","type":"paper","title":"The Injection Paradox: Brand-Level Suppression in Safety-Trained LLM Recommendations via RAG Context Injection","authors":["Hyunseok Paeng"],"year":2026,"abstract":"We present a reproducible failure mode of safety training in RAG-based LLM recommendation -- the Injection Paradox -- in which prompt injections embedded in retrieved documents backfire against the attacker, suppressing the target brand below the injection-free baseline. In safety-trained Claude models, documents containing prompt injections suffer a sharp drop in recommendation rate, and this suppression propagates beyond the injected document to unmodified documents of the same brand. In Claud","url":"https://arxiv.org/abs/2606.09204","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00482","type":"paper","title":"GitInject: Real-World Prompt Injection Attacks in AI-Powered CI/CD Pipelines","authors":["Jafar Isbarov","Umid Suleymanov","Ilia Shumailov","Murat Kantarcioglu"],"year":2026,"abstract":"AI-powered agents are increasingly embedded in continuous integration and continuous delivery/deployment (CI/CD) pipelines to autonomously review pull requests (PRs), triage issues, and maintain codebases. These agents ingest untrusted content while operating with elevated repository permissions, making them a natural target for prompt injection attacks with supply chain consequences. We present GitInject, an open-source framework for evaluating prompt injection vulnerabilities in real, live Git","url":"https://arxiv.org/abs/2606.09935","categories":["prompt-injection","supply-chain-attacks"],"reviewed":false},{"id":"llmsec-2026-00483","type":"paper","title":"VATS: Exploiting Implicit Authority in Error-Path Injection via Systematic Mutation","authors":["Harshil Patel","Kunal Pai"],"year":2026,"abstract":"As the Model Context Protocol (MCP) standardizes tool-calling for autonomous agents, it introduces a critical, unexamined attack surface: the error-handling loop. We hypothesize that tool error messages possess implicit authority, triggering corrective reasoning modes that bypass standard safety heuristics. We introduce VATS (Vulnerability Analysis of Tool Streams), a mutation-driven framework that systematically evolves adversarial payloads across seven structural and linguistic dimensions. Our","url":"https://arxiv.org/abs/2606.07992","categories":["autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00484","type":"paper","title":"MalSkillBench: A Runtime-Verified Benchmark of Malicious Agent Skills","authors":["Wenbo Guo","Wei Zeng","Chengwei Liu","Xiaojun Jia","Yijia Xu","Lei Tang","Yong Fang","Yang Liu"],"year":2026,"abstract":"AI coding agents such as Claude Code and Gemini CLI increasingly extend themselves with third-party skills: markdown packages bundling natural-language instructions, executable scripts, and tool permissions. Because a skill is at once code and agent-facing instruction, it introduces a supply chain dependency whose risk is neither pure code nor pure prompt. Detection tools have never been measured against verified ground truth spanning this hybrid space, leaving their effectiveness unknown and wi","url":"https://arxiv.org/abs/2606.07131","categories":["supply-chain-attacks","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00485","type":"paper","title":"MLingualFC: Evaluating Jailbreak Vulnerabilities in Multilingual Vision-Language Models","authors":["Rishabh Makwana"," Mamta","Deeksha Varshney","Oana Cocarascu"],"year":2026,"abstract":"Vision-Language Models (VLMs) have demonstrated strong performance across multimodal tasks, yet their safety robustness remains an open challenge. While prior work has shown that structured visual prompts such as flowcharts can effectively jailbreak VLMs, existing studies are largely limited to English-centric settings. In this paper, we introduce MLingualFC, a multilingual multimodal benchmark designed to evaluate jailbreak vulnerabilities of VLMs across diverse languages using structured flowc","url":"https://arxiv.org/abs/2606.07706","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00486","type":"paper","title":"The Containment Gap: How Deployed Agentic AI Frameworks Fail Public-Facing Safety Requirements","authors":["Md Jafrin Hossain","Mohammad Arif Hossain","Weiqi Liu","Nirwan Ansari"],"year":2026,"abstract":"Agentic large language model systems that autonomously invoke tools, maintain persistent memory, and execute multi-step plans are increasingly deployed in public-facing domains, including government services, healthcare triage, and financial advising. We ask whether the frameworks used to build these systems provide architectural-level structural safety guarantees. Applying six containment principles derived from a compositional model of agentic architectures, we audit three dominant frameworks ","url":"https://arxiv.org/abs/2606.12797","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00487","type":"paper","title":"Securing Code Understanding: Detecting Natural Backdoor Vulnerability in Code Language Models","authors":["Yuchen Chen","Weisong Sun","Haocheng Huang","Yuan Xiao","Chunrong Fang","Yiran Zhang","Tingting Xu","Zhenpeng Chen","An Guo","Peizhuo Lv","Xiaofang Zhang","Zhenyu Chen","Yang Liu","Baowen Xu"],"year":2026,"abstract":"Code Language Models (CodeLMs) have become integral to software engineering, significantly advancing code intelligence tasks. However, their widespread adoption has raised critical security concerns, particularly regarding susceptibility to backdoor attacks. Recent studies have uncovered naturally occurring backdoors, referred to as natural backdoors, in normally trained deep learning models. Despite posing threats as serious as those introduced through data poisoning, security implications of n","url":"https://arxiv.org/abs/2606.10846","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00488","type":"paper","title":"MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents","authors":["Yv Zhang","Hao Sun","Hao Fang","Kuofeng Gao","Fan Mo","Bin Chen","Shu-Tao Xia","Yaowei Wang"],"year":2026,"abstract":"External memory has become a core component of modern web agents, enabling long-horizon reasoning through the retrieval of past experiences. However, this paradigm introduces a critical vulnerability: malicious content injected into memory can be persistently recalled and repeatedly influence agent behavior. In this work, we identify and systematically study multimodal memory poisoning, an overlooked yet practical attack surface in web-agent systems. We propose MemVenom, a unified black-box atta","url":"https://arxiv.org/abs/2606.10742","categories":["data-poisoning","memory-security"],"reviewed":false},{"id":"llmsec-2026-00489","type":"paper","title":"Defending Against Malicious Finetuning by Scaling Train-time Adversarial Attacks","authors":["Haoming Wen","Shi Chen","Qingyu Shi","Siyuan Liu","Minrui Luo","Jingzhao Zhang","Tianxing He"],"year":2026,"abstract":"Current open-weight large language models (LLMs) are prone to malicious finetuning attacks, which could compromise the safety alignment of LLMs with only a few steps of supervised finetuning (SFT) on poisoned datasets. Existing alignment-stage defenses are primarily designed to defend against attacks that use parameter-efficient finetuning methods. However, they fail to defend against stronger attacks that use full-parameter finetuning. In this paper, we propose Patcher, a method inspired by adv","url":"https://arxiv.org/abs/2606.07970","categories":["adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-00490","type":"paper","title":"Exploring Systems-Thinking Approaches to Loss of Control Risk","authors":["Aurelio Carlucci","Sean P. Fillingham","James Walpole","Jakub Kryś"],"year":2026,"abstract":"Internal deployment of agentic AI systems for coding and research creates a sociotechnical control problem that extends beyond model behaviour. We treat internal-deployment Loss of Control as the inability to reliably constrain, audit, reverse, or halt AI-mediated changes to code, infrastructure, evaluation, or deployment processes in time to prevent serious organisational or societal harms. We ask whether established systems-safety methods can identify risks that model-level evaluations may mis","url":"https://arxiv.org/abs/2606.13474","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00491","type":"paper","title":"A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents","authors":["Krti Tallam"],"year":2026,"abstract":"Enterprise security was built to govern data boundaries: the protected surface was data at rest and in transit, and the controls -- access control, data-loss prevention, perimeter inspection -- governed crossings of that boundary. Production AI agents dissolve this assumption. An agent reads context, calls tools, invokes connectors, and modifies systems of record on an enterprise's behalf, so risk moves inside the workflow, into sequences of individually-permitted actions that may transform a bu","url":"https://arxiv.org/abs/2606.12320","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00492","type":"paper","title":"AgentCanary: A Security Evaluation Framework for Autonomous AI Agents in Real Executable Environments","authors":["Peiyang Li","Songping Wang","Yi Huang","Yanhua Shi","Chenhao Zhang","Qi Li","Yueming Lyu","Caifeng Shan","Fengting Li","Chao Feng","Chuanqun Zhu","Liang Chen"],"year":2026,"abstract":"Autonomous AI agents have driven the transition from conversation to task execution, shifting security failures from textual deception to system compromise. Although security evaluation is crucial for proactive risk prevention, prior work is constrained by fundamental bottlenecks, including fragmented risk coverage, static or low-fidelity execution environments, and single-dimensional and coarse-grained assessment metrics. To address these challenges, we propose AgentCanary, a comprehensive secu","url":"https://arxiv.org/abs/2606.10484","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00493","type":"paper","title":"Attack Selection in Agentic AI Control Evaluations Meaningfully Decreases Safety","authors":["Catherine Ge-Wang","Tyler Crosse","Benjamin Hadad","Joachim Schaeffer","Ram Potham","Tyler Tracy"],"year":2026,"abstract":"An attacker that strategically chooses when to attack is much harder to catch than one that attacks indiscriminately. AI control is a safety framework for deploying capable but untrusted AI agents under the oversight of a weaker, trusted monitor and a limited human audit budget. Control evaluations stress-test these protocols by pitting a red-team attack policy against the blue-team monitor, but current evaluations typically assume attackers that do not strategically select when to attack. We st","url":"https://arxiv.org/abs/2606.06529","categories":["agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00494","type":"paper","title":"Information Security and Clinical Ethics in LLM-Based Healthcare Systems","authors":["Updesh Kumar Jaiswal","Jaishree Jain","Harnit Saini","Amrita Bhatnagar","Pushpendra Singh","Shashank Sahu","Khushbu Malviya"],"year":2026,"abstract":"Large Language Models (LLMs) play a complex and ever-evolving role in medicine. LLMs, in general, can help medical professionals identify patients by giving them fast, data-driven information. They are able to analyse patient data, compare it with a comprehensive medical history, and make recommendations for potential diagnoses or identify issues that need more research. This chapter examines the threats that prompt injection attacks, perfect overturn, and data leak could pose to PHI and","url":"https://doi.org/10.4018/979-8-3373-7862-6.ch004","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00495","type":"paper","title":"Evaluation Methods for LLM Safety and Reliability in Clinical and Healthcare Applications","authors":["Anand Jha","Kirtiraj Bhatele","Pratyush Mihir"],"year":2026,"abstract":"Large Language Models (LLMs) are being deployed in clinical and healthcare systems, which requires serious consideration of their safety, reliability. This chapter investigates hallucinations, harms to patients and healthcare employees as well as methods of measuring the performance, deployment obstacles and ethical issues of LLMs in medical applications. The regulatory frameworks, such as FDA and the EU AI act are reviewed to put compliance and governance requirements into perspective. ","url":"https://doi.org/10.4018/979-8-3373-7862-6.ch005","categories":["risk-frameworks"],"reviewed":false},{"id":"llmsec-2026-00496","type":"paper","title":"Calibration Without Comprehension: Diagnosing the Limits of Fine-Tuning LLMs for Vulnerability Detection in Systems Software","authors":["Arastoo Zibaeirad","Marco Vieira"],"year":2026,"abstract":"Whether LLMs scoring well on vulnerability benchmarks genuinely reason about security or merely pattern-match on contaminated data remains unresolved. We present CWE-Trace, a framework for LLM vulnerability detection built from 834 manually curated Linux kernel samples spanning 74 CWEs. The framework enforces a strict temporal split (pre-2025 historical set / post-cutoff leakage-free set), preserves context-aware vulnerable--patched pairs, and introduces two diagnostic metrics: the Directional F","url":"https://arxiv.org/abs/2606.20502","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00497","type":"paper","title":"How Much Can We Trust LLM Search Agents? Measuring Endorsement Vulnerability to Web Content Manipulation","authors":["Yimeng Chen","Zhe Ren","Firas Laakom","Yu Li","Dandan Guo","Jürgen Schmidhuber"],"year":2026,"abstract":"Large language model (LLM)-based search agents synthesize open-web content into actionable recommendations on behalf of users, creating a risk that attacker-published pages are transformed into endorsed claims. We introduce SearchGEO, a controlled evaluation framework for measuring endorsement corruption in LLM-based web-search agents, combining a web-evidence manipulation pipeline, a five-mode attack taxonomy, and multiple output-level metrics. We evaluate 13 LLM backends on 308 cases each. Res","url":"https://arxiv.org/abs/2606.16821","categories":["benchmarks","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00498","type":"paper","title":"SkillVetBench: LLM-as-Judge for Multi-Dimensional Security Risk Evaluation in Open-Source LLM Agent Skills","authors":["Ismail Hossain","Sai Puppala","Md Jahangir Alam","Tanzim Ahad","Sajedul Talukder"],"year":2026,"abstract":"Open-source LLM agent ecosystems are growing rapidly, yet the security of community-contributed skills - modular tool definitions that extend agent capabilities - remains largely unvetted. The gap we fill: existing scanners operate at the code layer and are structurally blind to instruction-layer and multi-agent risk - natural-language directives that hijack an agent, exfiltrate data through encoded side channels, or chain harm across pipelines - so what is needed is a semantic, multi-dimensiona","url":"https://arxiv.org/abs/2606.15899","categories":["agentic-threats","agent-architecture","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00499","type":"paper","title":"Let Them Steal: Trapping Large Language Model Extraction Attacks with Knowledge Honeypot","authors":["Yuyang Dai","Yushun Dong"],"year":2026,"abstract":"Large language models deployed as commercial APIs are vulnerable to model extraction attacks, while existing defenses either act too late or degrade utility for legitimate users. We propose \\textbf{Knowledge Trap}, a defense that redirects extraction attacks toward low-transferability knowledge through a \\emph{Honeypot Knowledge Graph} (HKG) and breadcrumb-guided exploration. Instead of blocking queries or perturbing outputs, Knowledge Trap consumes the attacker's limited query budget on knowled","url":"https://arxiv.org/abs/2606.15810","categories":["model-extraction"],"reviewed":false},{"id":"llmsec-2026-00500","type":"paper","title":"AttackonCTF: Defending Hardware Security Competition Benchmarks in the Age of LLMs","authors":["Mohamadreza Rostami","Nikhilesh Singh","Stephen Muttathil","Lichao Wu","Chen Chen","Huimin Li","Jeyavijayan Rajendran","Ahmad-Reza Sadeghi"],"year":2026,"abstract":"Hardware security competitions such as HackTheSilicon serve as benchmarking platforms for evaluating vulnerability detection methods and for training humans and AI. However, our study reveals that LLMs threaten their validity. Instead of genuine security reasoning, detectors exploit a diff-style syntactic comparison, achieving an 83% detection rate, undermining fair evaluation. To mitigate this, we propose the first LLM-oriented, semantics-preserving obfuscation framework for these benchmarks. U","url":"https://arxiv.org/abs/2606.15809","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00501","type":"paper","title":"AutoDojo: Adaptive Attacks Expose Superficial Defenses and User-Underspecification Limits in LLM Agents","authors":["Xinhang Ma","Taoran Li","Chaowei Xiao","Zhiyuan Yu","Ning Zhang","Yevgeniy Vorobeychik"],"year":2026,"abstract":"Indirect prompt injection (IPI) is a major security threat to LLM-powered agents. Thus, a growing body of work have proposed a variety of defensive approaches against IPI. These can be grouped into three broad categories: 1) prompt-based (using prompting as a way to prevent agents from following malicious instructions), 2) detection-based (identifying and filtering malicious instructions), and 3) system-level (using systems insights, such as control and data isolation, for defense). However, com","url":"https://arxiv.org/abs/2606.15057","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00502","type":"paper","title":"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails","authors":["Yuguang Zhou","Xunguang Wang","Pingchuan Ma","Zhantong Xue","Zhaoyu Wang","Shuai Wang"],"year":2026,"abstract":"LLM-based guardrails have emerged as a highly effective defense against prompt injection and jailbreak attacks in autonomous agents. However, we reveal that the very reasoning and task-following capabilities enabling this protection introduce a novel vulnerability: attackers can inject crafted data to trap the guardrail in extended reasoning loops, effectuating a systematic denial-of-service (DoS) attack. To systematically expose this threat, we design a beam-search optimization framework that c","url":"https://arxiv.org/abs/2606.14517","categories":["prompt-injection","jailbreaking","guardrails","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00503","type":"paper","title":"SkillMutator: Benchmarking and Defending Language-and-Code Cross-modal Attacks on LLM Agent Skills","authors":["Youngduk Kim","Minkyoo Song","Seungwon Shin"],"year":2026,"abstract":"Large language model (LLM) agents increasingly extend their capabilities at runtime by loading Agent Skills, which pair natural-language specifications (SKILL.md) with executable scripts and resources. Because a skill's behavior relies on both natural-language instructions and executable code, assessing its safety requires cross-modal reasoning, creating a new language-and-code attack surface. Attackers can present a benign workflow in SKILL.md while embedding implicit directives that steer the ","url":"https://arxiv.org/abs/2606.14154","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00504","type":"paper","title":"Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems","authors":["Reza Soosahabi","Vivek Namsani"],"year":2026,"abstract":"Agentic AI systems increasingly rely on language-model components to interpret instructions, process external data, invoke tools, and coordinate with other agents. These capabilities make prompt-injection and jailbreak attacks more consequential, especially as attackers adopt model-guided automation to scale probing, prompt refinement, and response evaluation. This work analyzes the resulting attack-defense setting through a probabilistic model of a target system, its defense mechanism, and the ","url":"https://arxiv.org/abs/2606.20470","categories":["prompt-injection","jailbreaking","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00505","type":"paper","title":"A Layered Security Framework Against Prompt Injection in RAG-Based Chatbots","authors":["Gulshan Saleem","Nisar Ahmed","Muhammad Imran Zaman","Ali Hassan"],"year":2026,"abstract":"Prompt injection is ranked as the most critical vulnerability in large language model (LLM) deployments by the OWASP Top 10 for LLM Applications, yet existing defenses operate at isolated pipeline stages and remain incomplete. Input filters cannot inspect retrieved documents, while output monitors cannot prevent malicious payloads from reaching the model. Consequently, retrieval-augmented generation (RAG) chatbots remain vulnerable to indirect injection, where a poisoned knowledge-base document ","url":"https://arxiv.org/abs/2606.19660","categories":["prompt-injection","input-filtering"],"reviewed":false},{"id":"llmsec-2026-00506","type":"paper","title":"CodeSentinel: A Three-Layer Defense Against Indirect Prompt Injection in Code Contexts","authors":["Po-Han Cheng","Chia-Mu Yu","Ying-Dar Lin","Yu-Sung Wu","Wei-Bin Lee"],"year":2026,"abstract":"Code large language models increasingly retrieve external code context from repositories, documentation, issue threads, and coding-agent environments, creating an indirect prompt-injection surface where attackers hide instructions in comments, strings, identifiers, or decoy code. We propose CodeSentinel, a three-layer inference-time sanitizer. It uses Tree-sitter to extract high-risk model-facing CST nodes, then combines syntax-guided pre-filtering, CST-guided Dynamic Min-K\\% scoring, and node p","url":"https://arxiv.org/abs/2606.19235","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00507","type":"paper","title":"The Gate Is Only as Honest as Its Contracts: ContractGuard for the Contract Layer of Risk-Aware Causal Gating","authors":["Laxmipriya Ganesh Iyer","Rahul Suresh Babu"],"year":2026,"abstract":"Risk-Aware Causal Gating (RACG) defends tool-augmented LLM agents against indirect prompt injection by removing dangerous tools from the agent's visible action space, so that even a fully injection-compliant agent cannot call a tool it cannot see. We make three points. First, this structural guarantee does not eliminate the trust assumption behind safe tool use; it relocates it into the integrity of the tool contracts -- declared preconditions, effects, risk, and authorization -- that the gate r","url":"https://arxiv.org/abs/2606.18550","categories":["prompt-injection","agentic-threats","access-control","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00508","type":"paper","title":"SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents","authors":["Yuchuan Tian","Mengyu Zheng","Haocheng Mei","Ye Yuan","Chao Xu","Xinghao Chen","Hanting Chen","Yu Wang"],"year":2026,"abstract":"Tool-using language-model agents introduce security failures that go beyond unsafe text: they can disclose protected objects, write persistent memory, send messages, modify databases, or trigger harmful code and tool effects. Existing evaluations often collapse these stages into a single attack success rate, making it difficult to tell whether a model merely agreed with an attacker or actually produced observable harm. We introduce SafeClawBench, a staged benchmark for tool-using agent security ","url":"https://arxiv.org/abs/2606.18356","categories":["agentic-threats","sandboxing-isolation","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00509","type":"paper","title":"A Red-Team Study of Anthropic Fable 5 & Opus 4.8 Models","authors":["Nicola Franco"],"year":2026,"abstract":"We evaluate the adversarial robustness of two frontier large language models (LLMs) developed by Anthropic, Fable 5 and Opus 4.8, against four families of automated jailbreak attack across 7 826 harmful intents spanning a ten-category harm taxonomy. Using the HackAgent red-teaming framework, hundreds of thousands of adversarial attempts were generated and every apparent success was independently re-adjudicated by a panel of three judge models (majority vote). Both models resist the majority of a","url":"https://arxiv.org/abs/2606.18193","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00510","type":"paper","title":"PARSE: Provenance-Aware Retrieval Sanitization for Professional Domain LLM Agents","authors":["Aaditya Pai"],"year":2026,"abstract":"Prompt injection defenses evaluated on synthetic benchmarks do not generalize to real enterprise documents, which are longer, denser, and interleave legitimate authority language with factual content. We demonstrate this gap with a real-document benchmark of 122 tasks across five professional domains (financial, legal, medical, scientific, DevOps) using actual SEC filings, Federal Register rules, PubMed abstracts, arXiv papers, and GitHub postmortems. Paraphrasing, the strongest defense on synth","url":"https://arxiv.org/abs/2606.17467","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00511","type":"paper","title":"DoubtProbe: Black-Box Jailbreak Defense via Structural Verification and Semantic Auditing","authors":["Xuanyu Yin","Yilin Jiang","Jun Zhou","Kai Chen","Zhengfu Cao","Xiaolei Dong"],"year":2026,"abstract":"As large language models (LLMs) are increasingly deployed in user-facing systems, black-box jailbreak defense has become an important practical problem. Existing defenses often rely on known-attack coverage, prompt-level semantic judgment, or local runtime control, yet these paths can become unstable under evolving prompt packaging, expression rewriting, and structure manipulation. We observe that many black-box jailbreaks do not remove the harmful goal, but reorganize the information needed to ","url":"https://arxiv.org/abs/2606.16527","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00512","type":"paper","title":"An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios","authors":["Hankyul Baek","Jaewon Noh","Sang Seo","Yongsu Kim","Gabriel Waikin Loh Matienzo","Young Il Kim","Ee Wei Seah","Akriti Vij"],"year":2026,"abstract":"AI agents are increasingly being adopted in enterprise and personal settings with access to emails, databases, documents, and other tools where they can read, update, and disseminate sensitive information. Much of prior research on data leakage risks in agents has focused on adversarial data exfiltration through prompt injections and jailbreaks. However, sensitive information may also be exposed during non-adversarial use, creating leakage risks even when users issue benign requests. We report a","url":"https://arxiv.org/abs/2606.17114","categories":["prompt-injection","jailbreaking","membership-inference","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00513","type":"paper","title":"GAS-Leak-LLM: Genetic Algorithm-Based Suffix Optimization for Black-Box LLM Jailbreaking","authors":["Aman Anifer","Vignesh Kumar Kembu","Vishnu M","Antonino Nocera","Vinod P.","Amal Murali PK","Akshay S Rajan"],"year":2026,"abstract":"Large Language Models (LLMs) constitute pivotal components within the AI-dominated information technology ecosystem. To mitigate risks associated with harmful or policy-violating outputs, commercial systems employ advanced alignment strategies and multi-layered content moderation mechanisms. Despite these safeguards, recent research has demonstrated that LLMs remain vulnerable to adversarial manipulation, particularly through jailbreaking and prompt injection techniques. In this work, we propose","url":"https://arxiv.org/abs/2606.15788","categories":["prompt-injection","jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00514","type":"paper","title":"FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion","authors":["Zixin Rao","Wentian Zhu","Chan Aristella Lu","Zhaorun Chen","Wei Niu","Le Guan","Bo Li","Zhen Xiang"],"year":2026,"abstract":"Large language model (LLM) agents increasingly rely on long-term memory to support complex task execution, user personalization, and domain adaptation. Meanwhile, emerging access-control mechanisms for LLM agents are being explored to block policy-violating requests and prevent misuse. We reveal a novel attack surface arising from agent memory operations: prohibited content that would trigger access control can be fragmented across interactions, stored in long-term memory in benign-appearing for","url":"https://arxiv.org/abs/2606.15609","categories":["agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00515","type":"paper","title":"Defending against Adaptive Prompt Injection Attacks via Reasoning-enabled Task Alignment","authors":["Lipeng He","Yihan Wang","Jiawen Zhang","N. Asokan"],"year":2026,"abstract":"Indirect prompt injection attacks hijack LLM-based agents by embedding malicious instructions in third-party data that the agent retrieves during task execution. Existing defenses report near-zero attack success rate on static benchmarks, yet recent adaptive evaluations show that these results collapse once the attacker is allowed to optimize against the deployed defense. In this work, we trace this collapse to two failure modes. First, existing defense methods are confined to recognizing specif","url":"https://arxiv.org/abs/2606.15441","categories":["prompt-injection","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00516","type":"paper","title":"Security Engineering of OpenClaw: Analyzing Attack Surface Expansion and Trust-Boundary Violations","authors":["Saeid Jamshidi","Arghavan Moradi Dakhel","Kawser Wazed Nafi","Foutse Khomh"],"year":2026,"abstract":"Agentic large language model (LLM) systems can now execute actions, not only produce text. When model outputs trigger privileged operations such as shell commands, browser automation, or external tool calls, the security problem shifts from alignment alone to system configuration and structural design. We analyze OpenClaw, a self-hosted multi-agent system in which LLM outputs can execute commands and interact with tools and services. We measure compromise probability, boundary failures, privileg","url":"https://arxiv.org/abs/2606.15008","categories":["agentic-threats","guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00517","type":"paper","title":"FreoStream:Enhancing Stream Guardrails via Future-Aware Reasoning and Safety-Aligned Optimization","authors":["Jianwei Wang","Guoyang Shen","Yanhong Wu","Haoran Li","Hao Peng","Huiping Zhuang","Cen Chen","Ziqian Zeng"],"year":2026,"abstract":"Stream guardrails enable token-level safety detection before full responses are generated. However, they often make overly conservative judgements and block those sensitive but safe tokens, which is known as over-refusal. Due to lack of full context, they also fail to detect implicitly harmful content from jailbreaking. To address these challenges, we propose FreoStream, a novel streaming guardrail framework. Specifically, FreoStream fine-tunes a LoRA module to perform Future-Aware Reasoning whe","url":"https://arxiv.org/abs/2606.13737","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00518","type":"paper","title":"Efficient and Sound Probabilistic Verification for AI Agents","authors":["Alaia Solko-Breslin","Pramod Kaushik Mudrakarta","Mihai Christodorescu","Somesh Jha","Krishnamurthy Dj Dvijotham"],"year":2026,"abstract":"Securing AI agents that operate in complex digital environments has become a critical need, and runtime monitoring approaches that formulate and enforce policies expressed in a formal language like Datalog offer a promising solution. However, existing approaches are restricted to deterministic policies. In many practical applications of AI agents, there is a need to enforce security policies in the face of ambiguity, leading to probabilistic predicates or state transitions (for example, a declas","url":"https://arxiv.org/abs/2606.20510","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00519","type":"paper","title":"Deontic Policies for Runtime Governance of Agentic AI Systems","authors":["Anupam Joshi","Tim Finin","Karuna Pande Joshi","Lalana Kagal"],"year":2026,"abstract":"Autonomous agentic AI systems driven by Large Language Models (LLMs) introduce a new class of security, privacy, and compliance challenges: an agent that can invoke tools, manipulate data, install software, and coordinate with peer agents across organizational boundaries must be constrained not just by authentication and access control, but by the full structure of enterprise governance. This includes specifying what agents are permitted and prohibited from doing, what they areobliged to do afte","url":"https://arxiv.org/abs/2606.19464","categories":["agentic-threats","access-control","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00520","type":"paper","title":"Code-Augur: Agentic Vulnerability Detection via Specification Inference","authors":["Zhengxiong Luo","Mehtab Zafar","Dylan Wolff","Abhik Roychoudhury"],"year":2026,"abstract":"The advent of agentic vulnerability detection is already becoming a watershed moment for software security. Audits conducted entirely by autonomous LLM agents are uncovering critical vulnerabilities in fundamental software underpinning digital society. Many of these vulnerabilities remained masked for years, surfacing only now with AI agents. Yet the reasoning behind these discoveries remains alarmingly opaque and unvalidated. What assumptions did the agent make about a function's inputs when it","url":"https://arxiv.org/abs/2606.18619","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00521","type":"paper","title":"Trust-Aware Multi-Agent Traceability: Confidence-Calibrated Knowledge Graphs for Consistent Software Artifact Management","authors":["Mohamed Essam","Kareem Wael","Azza Hassan","Ahmed Haitham","Mahmoud Soliman","Samer Saber","Ibrahim Habib"],"year":2026,"abstract":"Multi-agent AI systems are increasingly used to automate software engineering tasks including requirements analysis, architecture design, test generation, and traceability linking. When these agents operate as a sequential pipeline over shared software artifacts, errors and low-confidence decisions made by upstream agents propagate to downstream stages, producing orphaned requirements, contradictory links, and compliance gaps that pose significant risks in safety-critical domains. We propose a t","url":"https://arxiv.org/abs/2606.17203","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00522","type":"paper","title":"Resilient Consensus in Agentic AI","authors":["Sribalaji C. Anand","George J. Pappas"],"year":2026,"abstract":"Large language model (LLM) agents are increasingly deployed in multi-agent systems where they must coordinate and agree on shared decisions. We ask whether classical resilient consensus theory, developed for deterministic agents, transfers to LLM agents that may behave adversarially. Framing LLM agreement as a Byzantine consensus game, we run controlled experiments on complete and general communication graphs. We find that prompted LLM agents fail to reach agreement that is achievable in princip","url":"https://arxiv.org/abs/2606.15024","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00523","type":"paper","title":"A Security Analysis of Long-Horizon Agentic AI Systems: Threats, Evaluation, and Framework Development","authors":["Ahmed Mohammed Almalki","Mehedi Masud"],"year":2026,"abstract":"This paper presents a structured analysis of security challenges in long-horizon agentic AI systems. The study reviews existing threats, evaluation approaches, attack propagation mechanisms, and security frameworks. A taxonomy of security threats and a framework for analyzing attack propagation are proposed to support future research in agentic AI security","url":"https://arxiv.org/abs/2606.14816","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00524","type":"paper","title":"Same-Origin Policy for Agentic Browsers","authors":["Xilong Wang","Xiaoxing Chen","Patrick Li","Dawn Song","Neil Gong"],"year":2026,"abstract":"Agentic browsers integrate autonomous AI agents into web browsers, enabling users to accomplish web tasks through natural-language instructions. The same-origin policy (SOP) is a fundamental browser security mechanism that prevents unauthorized automated cross-origin data flows induced by scripts. However, whether SOP remains effective in agentic browsers is an open question that has not been systematically studied. In this work, we bridge this gap. We first observe that an agentic browser can i","url":"https://arxiv.org/abs/2606.14027","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00525","type":"paper","title":"A Virtuous AI is an Existential Risk","authors":["Guillermo Del Pinal","Youngchan Lee","Min Ohn"],"year":2026,"abstract":"This paper examines trade-offs between AI safety and well-being relative to (i) one of the most promising methods for finetuning super-capable AIs, 'Constitutional AI', and (ii) one of the most influential approaches to understanding complex ethical decision making and the conditions for the well-being of rational agents, 'Virtue Ethics'. We finetune various models using a 'Virtuous agent' constitution, a 'Subordinate agent' constitution, and a 'Generic agent' constitution, and evaluate them on ","url":"https://arxiv.org/abs/2606.13739","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00526","type":"paper","title":"SecurityLingua: Efficient Defense of LLM Jailbreak Attacks via Security-Aware Prompt Compression","authors":["Yucheng Li","Surin Ahn","Huiqiang Jiang","Amir H. Abdi","Yuqing Yang","Lili Qiu"],"year":2025,"venue":"arXiv.org","abstract":"Large language models (LLMs) have achieved widespread adoption across numerous applications. However, many LLMs are vulnerable to malicious attacks even after safety alignment. These attacks typically bypass LLMs' safety guardrails by wrapping the original malicious instructions inside adversarial jailbreaks prompts. Previous research has proposed methods such as adversarial training and prompt rephrasing to mitigate these safety vulnerabilities, but these methods often reduce the utility of LLM","url":"https://www.semanticscholar.org/paper/886194f65d0c2a0b56458d8ba2cd96bd69a97f6a","categories":["jailbreaking","guardrails"],"citation_count":10,"reviewed":false},{"id":"llmsec-2026-00527","type":"paper","title":"Proactive defense against LLM Jailbreak","authors":["Weiliang Zhao","Jinjun Peng","Daniel Ben-Levi","Zhou Yu","Junfeng Yang"],"year":2025,"venue":"arXiv.org","abstract":"The proliferation of powerful large language models (LLMs) has necessitated robust safety alignment, yet these models remain vulnerable to evolving adversarial attacks, including multi-turn jailbreaks that iteratively search for successful queries. Current defenses, which are primarily reactive and static, often fail to handle these iterative attacks. In this paper, we introduce ProAct, a novel proactive defense framework designed to disrupt and mislead these iterative search jailbreak methods. ","url":"https://www.semanticscholar.org/paper/a5cb42e27c0207971690d47adbae8b633842fd3c","categories":["jailbreaking","adversarial-examples","guardrails"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00528","type":"paper","title":"WordGame: Efficient & Effective LLM Jailbreak via Simultaneous Obfuscation in Query and Response","authors":["Tianrong Zhang","Bochuan Cao","Yuanpu Cao","Lu Lin","Prasenjit Mitra","Jinghui Chen"],"year":2024,"venue":"North American Chapter of the Association for Computational Linguistics","abstract":"The recent breakthrough in large language models (LLMs) such as ChatGPT has revolutionized production processes at an unprecedented pace. Alongside this progress also comes mounting concerns about LLMs' susceptibility to jailbreaking attacks, which leads to the generation of harmful or unsafe content. While safety alignment measures have been implemented in LLMs to mitigate existing jailbreak attempts and force them to become increasingly complicated, it is still far from perfect. In this paper,","url":"https://www.semanticscholar.org/paper/8db6ff37617c5d3a6aec9e40e5e829a735d0c0cf","categories":["jailbreaking","guardrails"],"citation_count":38,"reviewed":false},{"id":"llmsec-2026-00529","type":"paper","title":"CCFC: Core & Core-Full-Core Dual-Track Defense for LLM Jailbreak Protection","authors":["Jiaming Hu","Haoyu Wang","Debarghya Mukherjee","I. Paschalidis"],"year":2025,"venue":"arXiv.org","abstract":"Jailbreak attacks pose a serious challenge to the safe deployment of large language models (LLMs). We introduce CCFC (Core&Core-Full-Core), a dual-track, prompt-level defense framework designed to mitigate LLMs'vulnerabilities from prompt injection and structure-aware jailbreak attacks. CCFC operates by first isolating the semantic core of a user query via few-shot prompting, and then evaluating the query using two complementary tracks: a core-only track to ignore adversarial distractions (e.g.,","url":"https://www.semanticscholar.org/paper/3fe877de5fe0afa33a3ff9ff0ea0daf53b842be1","categories":["prompt-injection","jailbreaking"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00530","type":"paper","title":"AlignTree: Efficient Defense Against LLM Jailbreak Attacks","authors":["Gil Goren","Shahar Katz","Lior Wolf"],"year":2025,"venue":"AAAI Conference on Artificial Intelligence","abstract":"Large Language Models (LLMs) are vulnerable to adversarial attacks that bypass safety guidelines and generate harmful content. Mitigating these vulnerabilities requires defense mechanisms that are both robust and computationally efficient. However, existing approaches either incur high computational costs or rely on lightweight defenses that can be easily circumvented, rendering them impractical for real-world LLM-based systems. In this work, we introduce the AlignTree defense, which enhances mo","url":"https://www.semanticscholar.org/paper/3a45a8e64f082728183c845858a50e5044c73a55","categories":["jailbreaking","adversarial-examples"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00531","type":"paper","title":"Parallel Hybrid Classical-Quantum Architecture for AI Safety and LLM Jailbreak Detection","authors":["Sajal Bajaj","Kamlesh Dutta"],"year":2026,"venue":"2026 IEEE International Conference on AI Engineering and Innovations (AIEI)","abstract":"Large language models (LLMs) are increasingly vulnerable to adversarial \"jailbreak\" attacks designed to elude safety and privacy controls. Detection is still a challenging task due to the complexity of adversarial prompts and labeled data scarcity. While fine-tuning a traditional Transformer model, such as BERT and its variants, provides strong backbone results with frequently limited recall of complex edge cases, it often comes with extensive computational costs. This work proposes a parallel h","url":"https://www.semanticscholar.org/paper/8afdd6ca08ce0586ebee6b6de85303896c67d376","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00532","type":"paper","title":"AdaSteer: Your Aligned LLM is Inherently an Adaptive Jailbreak Defender","authors":["Weixiang Zhao","Jiahe Guo","Yulin Hu","Yang Deng","An Zhang","Xingyu Sui","Xinyang Han","Yanyan Zhao","Bing Qin","Tat-Seng Chua","Ting Liu"],"year":2025,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"Despite extensive efforts in safety alignment, large language models (LLMs) remain vulnerable to jailbreak attacks. Activation steering offers a training-free defense method but relies on fixed steering coefficients, resulting in suboptimal protection and increased false rejections of benign inputs. To address this, we propose AdaSteer, an adaptive activation steering method that dynamically adjusts model behavior based on input characteristics. We identify two key properties: Rejection Law (R-L","url":"https://www.semanticscholar.org/paper/d61f12d092042cd6c6dd510ddd6f319c977c90a5","categories":["jailbreaking","guardrails"],"citation_count":29,"reviewed":false},{"id":"llmsec-2026-00533","type":"paper","title":"Benchmarking adversarial robustness to bias elicitation in large language models: scalable automated assessment with LLM-as-a-judge","authors":["Riccardo Cantini","A. Orsino","Massimo Ruggiero","Domenico Talia"],"year":2025,"venue":"Machine-mediated learning","abstract":"The growing integration of Large Language Models (LLMs) into critical societal domains has raised concerns about embedded biases that can perpetuate stereotypes and undermine fairness. Such biases may stem from historical inequalities in training data, linguistic imbalances, or adversarial manipulation. Despite mitigation efforts, recent studies show that LLMs remain vulnerable to adversarial attacks that elicit biased outputs. This work proposes a scalable benchmarking framework to assess LLM r","url":"https://www.semanticscholar.org/paper/a6db5ffa1a82b3d969f184b22e376ca04203b2dc","categories":["adversarial-examples","benchmarks"],"citation_count":27,"reviewed":false},{"id":"llmsec-2026-00534","type":"paper","title":"Bypassing LLM Guardrails: An Empirical Analysis of Evasion Attacks against Prompt Injection and Jailbreak Detection Systems","authors":["William Hackett","Lewis Birch","Stefan Trawicki","Neeraj Suri","Peter Garraghan"],"year":2025,"abstract":"Large Language Models (LLMs) guardrail systems are designed to protect against prompt injection and jailbreak attacks. However, they remain vulnerable to evasion techniques. We demonstrate two approaches for bypassing LLM prompt injection and jailbreak detection systems via traditional character injection methods and algorithmic Adversarial Machine Learning (AML) evasion techniques. Through testing against six prominent protection systems, including Microsoft's Azure Prompt Shield and Meta's Pro","url":"https://www.semanticscholar.org/paper/f74726bf146835dc48ba4d8ab2dcfd0ed762af8e","categories":["prompt-injection","jailbreaking","adversarial-examples","guardrails"],"citation_count":20,"reviewed":false},{"id":"llmsec-2026-00535","type":"paper","title":"Adversarial Poetry as a Universal Single-Turn Jailbreak Mechanism in Large Language Models","authors":["Piercosma Bisconti","Matteo Prandi","Federico Pierucci","Francesco Giarrusso","Marcantonio Bracale","Marcello Galisai","Vincenzo Suriani","Olga E. Sorokoletova","Federico Sartore","Daniele Nardi"],"year":2025,"venue":"arXiv.org","abstract":"We present evidence that adversarial poetry functions as a universal single-turn jailbreak technique for Large Language Models (LLMs). Across 25 frontier proprietary and open-weight models, curated poetic prompts yielded high attack-success rates (ASR), with some providers exceeding 90%. Mapping prompts to MLCommons and EU CoP risk taxonomies shows that poetic attacks transfer across CBRN, manipulation, cyber-offence, and loss-of-control domains. Converting 1,200 MLCommons harmful prompts into v","url":"https://www.semanticscholar.org/paper/04f84b2b9bcd069f9251365d3eeda4c114a7b4eb","categories":["jailbreaking"],"citation_count":15,"reviewed":false},{"id":"llmsec-2026-00536","type":"paper","title":"Graph of Attacks with Pruning: Optimizing Stealthy Jailbreak Prompt Generation for Enhanced LLM Content Moderation","authors":["Daniel Schwartz","Dmitriy Bespalov","Zhe Wang","Ninad Kulkarni","Yanjun Qi"],"year":2025,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"As large language models (LLMs) become increasingly prevalent, ensuring their robustness against adversarial misuse is crucial. This paper introduces the GAP (Graph of Attacks with Pruning) framework, an advanced approach for generating stealthy jailbreak prompts to evaluate and enhance LLM safeguards. GAP addresses limitations in existing tree-based LLM jailbreak methods by implementing an interconnected graph structure that enables knowledge sharing across attack paths. Our experimental evalua","url":"https://www.semanticscholar.org/paper/79f7e65410d089da1f7a66569a19d5ff27b80f5d","categories":["jailbreaking","guardrails"],"citation_count":11,"reviewed":false},{"id":"llmsec-2026-00537","type":"paper","title":"\"Short-length\" Adversarial Training Helps LLMs Defend \"Long-length\" Jailbreak Attacks: Theoretical and Empirical Evidence","authors":["Shaopeng Fu","Liang Ding","Di Wang"],"year":2025,"venue":"arXiv.org","abstract":"Jailbreak attacks against large language models (LLMs) aim to induce harmful behaviors in LLMs through carefully crafted adversarial prompts. To mitigate attacks, one way is to perform adversarial training (AT)-based alignment, i.e., training LLMs on some of the most adversarial prompts to help them learn how to behave safely under attacks. During AT, the length of adversarial prompts plays a critical role in the robustness of aligned LLMs. While long-length adversarial prompts during AT might l","url":"https://www.semanticscholar.org/paper/9bb4375dbebad1cbd61db6849c425d52cadf9472","categories":["jailbreaking","guardrails"],"citation_count":14,"reviewed":false},{"id":"llmsec-2026-00538","type":"paper","title":"Agent Smith: A Single Image Can Jailbreak One Million Multimodal LLM Agents Exponentially Fast","authors":["Xiangming Gu","Xiaosen Zheng","Tianyu Pang","Chao Du","Qian Liu","Ye Wang","Jing Jiang","Min Lin"],"year":2024,"venue":"International Conference on Machine Learning","abstract":"A multimodal large language model (MLLM) agent can receive instructions, capture images, retrieve histories from memory, and decide which tools to use. Nonetheless, red-teaming efforts have revealed that adversarial images/prompts can jailbreak an MLLM and cause unaligned behaviors. In this work, we report an even more severe safety issue in multi-agent environments, referred to as infectious jailbreak. It entails the adversary simply jailbreaking a single agent, and without any further interven","url":"https://www.semanticscholar.org/paper/b0ada492ba48e85016cbbfd95ec7180fb7e79648","categories":["jailbreaking","agentic-threats","red-teaming","agent-architecture"],"citation_count":142,"reviewed":false},{"id":"llmsec-2026-00539","type":"paper","title":"CAVGAN: Unifying Jailbreak and Defense of LLMs via Generative Adversarial Attacks on their Internal Representations","authors":["Xiaohu Li","Yunfeng Ning","Zepeng Bao","Mayi Xu","Jianhao Chen","Tieyun Qian"],"year":2025,"venue":"Annual Meeting of the Association for Computational Linguistics","abstract":"Security alignment enables the Large Language Model (LLM) to gain the protection against malicious queries, but various jailbreak attack methods reveal the vulnerability of this security mechanism. Previous studies have isolated LLM jailbreak attacks and defenses. We analyze the security protection mechanism of the LLM, and propose a framework that combines attack and defense. Our method is based on the linearly separable property of LLM intermediate layer embedding, as well as the essence of ja","url":"https://www.semanticscholar.org/paper/ff5db18341d678f2a9528dcd07ab5b63c031331c","categories":["jailbreaking","adversarial-examples","guardrails"],"citation_count":7,"reviewed":false},{"id":"llmsec-2026-00540","type":"paper","title":"Adversarial Tuning: Defending Against Jailbreak Attacks for LLMs","authors":["Fan Liu","Zhao Xu","Hao Liu"],"year":2024,"venue":"arXiv.org","abstract":"Although safely enhanced Large Language Models (LLMs) have achieved remarkable success in tackling various complex tasks in a zero-shot manner, they remain susceptible to jailbreak attacks, particularly the unknown jailbreak attack. To enhance LLMs' generalized defense capabilities, we propose a two-stage adversarial tuning framework, which generates adversarial prompts to explore worst-case scenarios by optimizing datasets containing pairs of adversarial prompts and their safe responses. In the","url":"https://www.semanticscholar.org/paper/9223a64feb573e62498c2ca914ed97557c580167","categories":["jailbreaking"],"citation_count":35,"reviewed":false},{"id":"llmsec-2026-00541","type":"paper","title":"Probing Latent Subspaces in LLM for AI Security: Identifying and Manipulating Adversarial States","authors":["Xin Wei Chia","Jonathan Pan"],"year":2025,"venue":"arXiv.org","abstract":"Large Language Models (LLMs) have demonstrated remarkable capabilities across various tasks, yet they remain vulnerable to adversarial manipulations such as jailbreaking via prompt injection attacks. These attacks bypass safety mechanisms to generate restricted or harmful content. In this study, we investigated the underlying latent subspaces of safe and jailbroken states by extracting hidden activations from a LLM. Inspired by attractor dynamics in neuroscience, we hypothesized that LLM activat","url":"https://www.semanticscholar.org/paper/03f458d740dea240d07abe93cd7e756cb6a20cb8","categories":["prompt-injection","jailbreaking"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00542","type":"paper","title":"Defending Jailbreak Prompts via In-Context Adversarial Game","authors":["Yujun Zhou","Yufei Han","Haomin Zhuang","Taicheng Guo","Kehan Guo","Zhenwen Liang","Hongyan Bao","Xiangliang Zhang"],"year":2024,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"Large Language Models (LLMs) demonstrate remarkable capabilities across diverse applications. However, concerns regarding their security, particularly the vulnerability to jailbreak attacks, persist. Drawing inspiration from adversarial training in deep learning and LLM agent learning processes, we introduce the In-Context Adversarial Game (ICAG) for defending against jailbreaks without the need for fine-tuning. ICAG leverages agent learning to conduct an adversarial game, aiming to dynamically ","url":"https://www.semanticscholar.org/paper/50ceabc6aa41e08480fa5976342bfe04bb47bce3","categories":["jailbreaking","agentic-threats"],"citation_count":39,"reviewed":false},{"id":"llmsec-2026-00543","type":"paper","title":"Advancing Jailbreak Strategies: A Hybrid Approach to Exploiting LLM Vulnerabilities and Bypassing Modern Defenses","authors":["Mohamed Ahmed","Mohamed Abdelmouty","Mingyu Kim","Gunvanth Kandula","Alex Park","James C. Davis"],"year":2025,"venue":"arXiv.org","abstract":"The advancement of Pre-Trained Language Models (PTLMs) and Large Language Models (LLMs) has led to their widespread adoption across diverse applications. Despite their success, these models remain vulnerable to attacks that exploit their inherent weaknesses to bypass safety measures. Two primary inference-phase threats are token-level and prompt-level jailbreaks. Token-level attacks embed adversarial sequences that transfer well to black-box models like GPT but leave detectable patterns and rely","url":"https://www.semanticscholar.org/paper/70359e58bc876a54f7b3694f50b4d6e52703124d","categories":["jailbreaking"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-00544","type":"paper","title":"Latent Fusion Jailbreak: Blending Harmful and Harmless Representations to Elicit Unsafe LLM Outputs","authors":["Wenpeng Xing","Mohan Li","Chunqiang Hu","Haitao Zhang","Bo Lin","Meng Han"],"year":2025,"venue":"arXiv.org","abstract":"While Large Language Models (LLMs) have achieved remarkable progress, they remain vulnerable to jailbreak attacks. Existing methods, primarily relying on discrete input optimization (e.g., GCG), often suffer from high computational costs and generate high-perplexity prompts that are easily blocked by simple filters. To overcome these limitations, we propose Latent Fusion Jailbreak (LFJ), a stealthy white-box attack that operates in the continuous latent space. Unlike previous approaches, LFJ con","url":"https://www.semanticscholar.org/paper/3e9ca9bb9cd160d045ee046e026bc6efd8b762e0","categories":["jailbreaking"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00545","type":"paper","title":"Adversarial Attack-Defense Co-Evolution for LLM Safety Alignment via Tree-Group Dual-Aware Search and Optimization","authors":["Xurui Li","Kaisong Song","Rui Zhu","Pin-Yu Chen","Haixu Tang"],"year":2025,"venue":"arXiv.org","abstract":"Large Language Models (LLMs) have developed rapidly in web services, delivering unprecedented capabilities while amplifying societal risks. Existing works tend to focus on either isolated jailbreak attacks or static defenses, neglecting the dynamic interplay between evolving threats and safeguards in real-world web contexts. To mitigate these challenges, we propose ACE-Safety (Adversarial Co-Evolution for LLM Safety), a novel framework that jointly optimize attack and defense models by seamlessl","url":"https://www.semanticscholar.org/paper/9d0955162f6732bae135c35e00aa6ab8612c5175","categories":["jailbreaking","adversarial-examples","guardrails"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00546","type":"paper","title":"Jailbreak in pieces: Compositional Adversarial Attacks on Multi-Modal Language Models","authors":["Erfan Shayegani","Yue Dong","Nael B. Abu-Ghazaleh"],"year":2023,"venue":"International Conference on Learning Representations","abstract":"We introduce new jailbreak attacks on vision language models (VLMs), which use aligned LLMs and are resilient to text-only jailbreak attacks. Specifically, we develop cross-modality attacks on alignment where we pair adversarial images going through the vision encoder with textual prompts to break the alignment of the language model. Our attacks employ a novel compositional strategy that combines an image, adversarially targeted towards toxic embeddings, with generic prompts to accomplish the ja","url":"https://www.semanticscholar.org/paper/92b9d8b8c81c4c53ea62000c0924500b2dd11bce","categories":["jailbreaking","adversarial-examples","guardrails"],"citation_count":291,"reviewed":false},{"id":"llmsec-2026-00547","type":"paper","title":"Breaking the Shield: Adversarial Jailbreak Attacks and Defense Mechanisms in Large Language Models","authors":["Shreya Dubey","H. Lamkuche"],"year":2025,"venue":"2025 International Conference on Emerging Information Technology and Engineering Solutions (EITES)","abstract":"Large Language Models (LLMs) are increasingly used across diverse applications, raising security concerns, especially regarding jail-break attacks that bypass safety measures. This paper presents a systematic security evaluation of LLMs, exploring the effectiveness of jail-break detection methods, the security progression across model versions, the relationship between model size and vulnerability, and the impact of combined defense strategies. We evaluate both open-source (e.g., LLama, Mistral)","url":"https://www.semanticscholar.org/paper/01f493a008c386b93901c788bee8908b5c81a79a","categories":["jailbreaking"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00548","type":"paper","title":"Mitigating adversarial manipulation in LLMs: a prompt-based approach to counter Jailbreak attacks (Prompt-G)","authors":["Bhagyajit Pingua","Deepak Murmu","Meenakshi Kandpal","Jyotirmayee Rautaray","Pranati Mishra","R. Barik","Manob Saikia"],"year":2024,"venue":"PeerJ Computer Science","abstract":"Large language models (LLMs) have become transformative tools in areas like text generation, natural language processing, and conversational AI. However, their widespread use introduces security risks, such as jailbreak attacks, which exploit LLM’s vulnerabilities to manipulate outputs or extract sensitive information. Malicious actors can use LLMs to spread misinformation, manipulate public opinion, and promote harmful ideologies, raising ethical concerns. Balancing safety and accuracy require ","url":"https://www.semanticscholar.org/paper/a5835f37febb7b65e56e27c07beb1b9fa5a1bea4","categories":["jailbreaking","membership-inference","threat-modeling"],"citation_count":15,"reviewed":false},{"id":"llmsec-2026-00549","type":"paper","title":"LLM-Virus: Evolutionary Jailbreak Attack on Large Language Models","authors":["Miao Yu","Junfeng Fang","Yingjie Zhou","Xing Fan","Kun Wang","Shirui Pan","Qingsong Wen"],"year":2024,"venue":"arXiv.org","abstract":"While safety-aligned large language models (LLMs) are increasingly used as the cornerstone for powerful systems such as multi-agent frameworks to solve complex real-world problems, they still suffer from potential adversarial queries, such as jailbreak attacks, which attempt to induce harmful content. Researching attack methods allows us to better understand the limitations of LLM and make trade-offs between helpfulness and safety. However, existing jailbreak attacks are primarily based on opaqu","url":"https://www.semanticscholar.org/paper/0f2721b409db7793f70b8990858709108ac67633","categories":["jailbreaking","agent-architecture"],"citation_count":9,"reviewed":false},{"id":"llmsec-2026-00550","type":"paper","title":"JailbreakHunter: A Visual Analytics Approach for Jailbreak Prompts Discovery From Large-Scale Human-LLM Conversational Datasets","authors":["Zhihua Jin","Shiyi Liu","Haotian Li","Xun Zhao","Huamin Qu"],"year":2024,"venue":"IEEE Transactions on Visualization and Computer Graphics","abstract":"Large Language Models (LLMs) have gained significant attention but also raised concerns due to the risk of misuse. Jailbreak prompts, a popular type of adversarial attack towards LLMs, have appeared and constantly evolved to breach the safety protocols of LLMs. To address this issue, LLMs are regularly updated with safety patches based on reported jailbreak prompts. However, malicious users often keep their successful jailbreak prompts private to exploit LLMs. To uncover these private jailbreak ","url":"https://www.semanticscholar.org/paper/b679c8166de2332184d22e988e269739778f03a0","categories":["jailbreaking","adversarial-examples"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00551","type":"paper","title":"A Coin Flip for Safety: LLM Judges Fail to Reliably Measure Adversarial Robustness","authors":["Leo Schwinn","Moritz Ladenburger","Tim Beyer","Mehrnaz Mofakhami","G. Gidel","Stephan Gunnemann"],"year":2026,"venue":"arXiv.org","abstract":"Automated \\enquote{LLM-as-a-Judge} frameworks have become the de facto standard for scalable evaluation across natural language processing. For instance, in safety evaluation, these judges are relied upon to evaluate harmfulness in order to benchmark the robustness of safety against adversarial attacks. However, we show that existing validation protocols fail to account for substantial distribution shifts inherent to red-teaming: diverse victim models exhibit distinct generation styles, attacks ","url":"https://www.semanticscholar.org/paper/e8bd6ec12610833e7de437d7b2a5b4fc7c89fd9d","categories":["adversarial-examples","red-teaming","benchmarks"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00552","type":"paper","title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","authors":["Jialin Song","Xiaodong Liu","Weiwei Yang","Wuyang Chen","Mingqian Feng","Xuekai Zhu","Jianfeng Gao"],"year":2026,"abstract":"We present MultiBreak, a scalable and diverse multi-turn jailbreak benchmark to evaluate large language model (LLM) safety. Multi-turn jailbreaks mimic natural conversational settings, making them easier to bypass safety-aligned LLM than single-turn jailbreaks. Existing multi-turn benchmarks are limited in size or rely heavily on templates, which restrict their diversity. To address this gap, we unify a wide range of harmful jailbreak intents, and introduce an active learning pipeline for expand","url":"https://www.semanticscholar.org/paper/531a805a9f8b0646b82ef0181660b9d5967c6199","categories":["jailbreaking","benchmarks"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00553","type":"paper","title":"HarmChip: Evaluating Hardware Security Centric LLM Safety via Jailbreak Benchmarking","authors":["Zeng Wang","Minghao Shao","Weimin Fu","Prithwish Basu Roy","Xiaolong Guo","Ramesh Karri","Muhammad Shafique","J. Knechtel","Ozgur Sinanoglu"],"year":2026,"venue":"arXiv.org","abstract":"The integration of large language models (LLMs) into electronic design automation (EDA) workflows has introduced powerful capabilities for RTL generation, verification, and design optimization, but also raises critical security concerns. Malicious LLM outputs in this domain pose hardware-level threats, including hardware Trojan insertion, side-channel leakage, and intellectual property theft, that are irreversible once fabricated into silicon. Such requests often exploit semantic disguise, embed","url":"https://www.semanticscholar.org/paper/b07fe49a982822288e33ad3f00b426553622bb8a","categories":["jailbreaking","data-poisoning","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00554","type":"paper","title":"PromptShield: LoRA-Based Parameter-Efficient Refusal Alignment for Election-Targeted Adversarial LLM Attacks","authors":["Nishmitha M R","Tejakshi N S","Anshu Sharma","Kiran","A. M","Karan"],"year":2026,"venue":"IEEE International Conference on Circuits and Systems for Communications","abstract":"Large language models (LLMs) are increasingly deployed in public-facing information systems, where their misuse poses serious risks in high-stakes domains such as democratic elections. Despite extensive safety alignment, contemporary LLMs remain vulnerable to adversarial “jailbreak” prompting, enabling the generation of election-related misinformation, voter suppression narratives, and procedural manipulation. Recent election cycles have documented hundreds of verified instances of LLM-assisted ","url":"https://www.semanticscholar.org/paper/2860efddd9ce86ac18c5dbfd7c427c08e6bed6a2","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00555","type":"paper","title":"Trojan-Speak: Bypassing Constitutional Classifiers with No Jailbreak Tax via Adversarial Finetuning","authors":["Bilgehan Sel","Xuanli He","Alwin Peng","Ming Jin","Jerry Wei"],"year":2026,"venue":"arXiv.org","abstract":"Fine-tuning APIs offered by major AI providers create new attack surfaces where adversaries can bypass safety measures through targeted fine-tuning. We introduce Trojan-Speak, an adversarial fine-tuning method that bypasses Anthropic's Constitutional Classifiers. Our approach uses curriculum learning combined with GRPO-based hybrid reinforcement learning to teach models a communication protocol that evades LLM-based content classification. Crucially, while prior adversarial fine-tuning approache","url":"https://www.semanticscholar.org/paper/97dad0cd24d00035e5ae90c101de5072c2d1174d","categories":["jailbreaking","data-poisoning"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00556","type":"paper","title":"Double-Gaming: Jailbreak Attacks against LLM based on Red-Blue Team Game Theory","authors":["Chenlu Ma","Huairui Zhao","G. Nie","Baiyang Ji","Beibei Li","Haibin Zheng"],"year":2026,"venue":"2026 8th International Conference on Software Engineering and Computer Science (CSECS)","abstract":"Despite existing security alignment mechanisms, Large Language Models (LLMs) remain vulnerable to jailbreak attacks under static defenses. To address this, we propose a novel jailbreak attack and defense optimization framework based on Red-Blue Team dynamic game theory. This framework establishes an automated adversarial mechanism between red and blue teams, achieving closed-loop optimization of instruction generation, semantic interception, and strategy evolution. The red team employs reinforce","url":"https://www.semanticscholar.org/paper/17d55e7e6983830e40b9cb742d612e16687b8365","categories":["jailbreaking","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00557","type":"paper","title":"Adversarial-Test-Driven Multi-Agent LLM Defense: A Self-Evolving Framework via Inference-Time Prompt Optimization","authors":["Yang Qu","Yuwei He","Lei Cao","Juzheng Wang","Sulei Li","Hongxi Chen"],"year":2026,"venue":"Electronics","abstract":"Large Language Models (LLMs) remain highly susceptible to jailbreak attacks that bypass safety alignments through sophisticated prompt manipulation. While multi-agent defense systems have emerged as a promising countermeasure, existing frameworks predominantly rely on static agent designs, which struggle to adapt to evolving adversarial strategies. To bridge this gap, we propose an Adversarial-Test-Driven Multi-Agent Defense framework that shifts the focus from model-level fine-tuning to system-","url":"https://www.semanticscholar.org/paper/cdc9ba46a28837d6157b523d6b784f9e2f82cca8","categories":["jailbreaking","guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00558","type":"paper","title":"JailBreakV: A Benchmark for Assessing the Robustness of MultiModal Large Language Models against Jailbreak Attacks","authors":["Weidi Luo","Siyuan Ma","Xiaogeng Liu","Xiaoyu Guo","Chaowei Xiao"],"year":2024,"abstract":"With the rapid advancements in Multimodal Large Language Models (MLLMs), securing these models against malicious inputs while aligning them with human values has emerged as a critical challenge. In this paper, we investigate an important and unexplored question of whether techniques that successfully jailbreak Large Language Models (LLMs) can be equally effective in jailbreaking MLLMs. To explore this issue, we introduce JailBreakV-28K, a pioneering benchmark designed to assess the transferabili","url":"https://www.semanticscholar.org/paper/f019c9661b253ddb611e930348e20ddcd350a952","categories":["jailbreaking","benchmarks"],"citation_count":236,"reviewed":false},{"id":"llmsec-2026-00559","type":"paper","title":"A Multi-Perspective Benchmark Dataset and Moderation Model for LLM Safety Evaluation with Adversarial Robustness Analysis","authors":["Naseem Machlovi","Maryam Saleki","Ruhul Amin","Mohamed Rahouti","Shawqi Al-Maliki","Junaid Qadir","Mohamed M. Abdallah","A. Al-Fuqaha"],"year":2026,"venue":"ACM Transactions on Social Computing","abstract":"As large language models (LLMs) become deeply embedded in daily life, the urgent need for safer moderation systems that distinguish between naive and harmful requests while upholding appropriate censorship boundaries has never been greater. While existing LLMs can detect dangerous or unsafe content, they often struggle with nuanced cases such as implicit offensiveness, subtle gender and racial biases, and jailbreak prompts, due to the subjective and context-dependent nature of these issues. Furt","url":"https://www.semanticscholar.org/paper/561d15b0b2b486b59b60db0830aa47681584f55f","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00560","type":"paper","title":"Red Teaming the Mind of the Machine: A Systematic Evaluation of Prompt Injection and Jailbreak Vulnerabilities in LLMs","authors":["Chetan Pathade"],"year":2025,"venue":"arXiv.org","abstract":"Large Language Models (LLMs) are increasingly integrated into consumer and enterprise applications. Despite their capabilities, they remain susceptible to adversarial attacks such as prompt injection and jailbreaks that override alignment safeguards. This paper provides a systematic investigation of jailbreak strategies against various state-of-the-art LLMs. We categorize over 1,400 adversarial prompts, analyze their success against GPT-4, Claude 2, Mistral 7B, and Vicuna, and examine their gene","url":"https://www.semanticscholar.org/paper/46a9f0dc9f74bef40c2f860e604c338c8092d30e","categories":["prompt-injection","jailbreaking","adversarial-examples","guardrails","red-teaming"],"citation_count":40,"reviewed":false},{"id":"llmsec-2026-00561","type":"paper","title":"Foot-In-The-Door: A Multi-turn Jailbreak for LLMs","authors":["Zixuan Weng","Xiaolong Jin","Jinyuan Jia","Xiangyu Zhang"],"year":2025,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"Ensuring AI safety is crucial as large language models become increasingly integrated into real-world applications. A key challenge is jailbreak, where adversarial prompts bypass built-in safeguards to elicit harmful disallowed outputs. Inspired by psychological foot-in-the-door principles, we introduce FITD,a novel multi-turn jailbreak method that leverages the phenomenon where minor initial commitments lower resistance to more significant or more unethical transgressions. Our approach progress","url":"https://www.semanticscholar.org/paper/5b79112817c0115d3db49245311982b623270422","categories":["jailbreaking","guardrails"],"citation_count":38,"reviewed":false},{"id":"llmsec-2026-00562","type":"paper","title":"Backdoor Watermarking for Continuous Prompt Learning Models in Industrial Systems","authors":["Kongyang Chen","Chuwen Pang","Xiaolin Wang","Tiancai Liang","Jiaxing Shen"],"year":2026,"abstract":"Large Language Models (LLMs) are becoming key enablers in adaptive and autonomous systems, particularly under the paradigm of Industry 5.0, where human-centric design and generative Artificial Intelligence (AI) technologies are increasingly deployed. However, the widespread of LLMs raises serious intellectual property concerns, especially in few-shot learning scenarios where model customization is achieved through continuous prompt tuning. Traditional watermarking methods fail to protect","url":"https://doi.org/10.1145/3820766","categories":["data-poisoning","watermarking"],"reviewed":false},{"id":"llmsec-2026-00563","type":"paper","title":"Representation Matters: An Empirical Study of Program Representations for LLM Vulnerability Reasoning","authors":["Andrew Stoltman","Johnathan Tang","Haipeng Cai"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly used for automated vulnerability detection, but it remains unclear how program structure and semantics should be represented for LLM-based reasoning. Most prompting-based approaches provide raw source code, implicitly assuming that more source-level context gives the model better evidence. This paper challenges that assumption through RepBench, an empirical benchmark comparing raw source code with static-analysis-based program representations. RepBen","url":"https://arxiv.org/abs/2606.25356","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00564","type":"paper","title":"HelpBench: Assessing the Ability of LLMs to Provide Privacy, Safety, and Security Advice","authors":["Sarah Meiklejohn","Sunny Consolvo","Patrick Gage Kelley","Tara Matthews","Sai Teja Peddinti","Renee Shelby","Lenin Simicich","Kurt Thomas"],"year":2026,"abstract":"This paper introduces HelpBench, a benchmark for assessing whether LLMs are capable of providing accurate help in response to questions about digital privacy, safety, and security. We curated 450 questions representing authentic user situations and developed rubrics for each question to evaluate the factual accuracy and tone of a response. Example questions touch on how to regain access to lost or suspended accounts, how to balance the trade-offs of hardware security keys versus other forms of t","url":"https://arxiv.org/abs/2606.24819","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00565","type":"paper","title":"Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees","authors":["Yedidel Louck"],"year":2026,"abstract":"LLM agents increasingly rely on persistent long-term memory, which creates a critical vulnerability that we study here: memory poisoning. An adversary can store untrusted content in one session that later steers a consequential action, such as a payment, a setting change, or data exfiltration, in a future session. Existing defenses base a memory item's authority to act on either its content (detection or trust-scoring) or its derivation history (lineage). We show that both signals are malleable.","url":"https://arxiv.org/abs/2606.24322","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00566","type":"paper","title":"Evaluating LLMs for Real-World Web Vulnerability Detection","authors":["Sebastian Neef","Luca Jungnickel","Antonio Benjamin Buchholz","Valene Spence","Vicente Birke Gonzalez"],"year":2026,"abstract":"Large Language Models (LLMs) have emerged as a promising tool for automated vulnerability detection, yet their effectiveness on web-specific vulnerabilities remains to be explored. This work benchmarks six frontier (Claude Opus 4.6, Codex GPT-5.4, Gemini 3.1-pro-preview) and open-weight models (Qwen 3.5, Qwen 3 Coder Next, MiniMax M2.5) on their ability to detect real-world web vulnerabilities using static analysis in WordPress plugins, including SQL injection, stored cross-site scripting, path ","url":"https://arxiv.org/abs/2606.21397","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00567","type":"paper","title":"Local LLM Agents as Vulnerable Runtimes:A Source-Code Audit of the Agent Runtime Layer","authors":["Zhengsong Zhang","Zongze Li","Jiawei Guo","Haipeng Cai"],"year":2026,"abstract":"Local LLM agents such as OpenClaw and Nanobot run on end-user machines and act on host resources - the shell, filesystem, browser, stored credentials, and messaging applications - through natural-language goals. These agents have become privileged software runtimes that mediate between user intent, model outputs, and host-level actions. Existing research characterizes the landscape through prompt injection, malicious skills, marketplace risks, or black-box evaluation of agents. But the implement","url":"https://arxiv.org/abs/2606.21071","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00568","type":"paper","title":"Honeyquest for LLMs: Rethinking Cyber Deception for AI Attackers","authors":["Kerri Prinos","Lilianne Brush","Cameron Denton"],"year":2026,"abstract":"The empirical foundation of cyber deception relies on human-centered hypotheses, but the rapid emergence of autonomous, AI-enabled attackers challenges whether this foundation transfers to AI agents. To address this, we introduce an automated evaluation framework adapted from the Honeyquest instrument to assess LLM attacker judgment at scale. Our 21-LLM cohort spanned 10 providers, diverse architectures and specializations, open- and closed-weight models, and parameter scales from 8B to over 1T.","url":"https://arxiv.org/abs/2606.21037","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00569","type":"paper","title":"What the Eyes See, the LLMs Miss: Exploiting Human Perception for Adversarial Text Attacks","authors":["Qin Yang","Lu Malloy","Joshua Lee","Xiaohan Chang","Meisam Mohammady","Doowon Kim","Yuan Hong"],"year":2026,"abstract":"Large language model (LLM)-powered content moderation systems are a critical defense against harmful online content. However, they operate primarily on tokenized text and often overlook visual cues that humans naturally use when interpreting content. We show that this limitation creates a fundamental vulnerability: content readily recognized as harmful by humans can evade automated moderation. To systematically study this problem, we introduce Human-Perceptible Adversarial Attacks (HPAA), which ","url":"https://arxiv.org/abs/2606.09700","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00570","type":"paper","title":"Jailbreaking for the Average Jane: Choosing Optimal Jailbreaks via Bandit Algorithms for Automatically Enhanced Queries","authors":["Prarabdh Shukla"," Ritik","Suhas Rao","Arpit Agarwal","Arjun Bhagoji"],"year":2026,"abstract":"With a profusion of jailbreaks for LLMs now widely known, a growing concern is that non-expert malicious actors (\"the average Jane\") could elicit actionable responses to malicious requests. In this work, we examine whether this concern is justified. A non-expert malicious actor requires two ingredients for a successful attack: a powerful jailbreak for their target model, acting on an effective malicious query. For the former, we propose a novel attack strategy based on the multi-armed bandit fra","url":"https://arxiv.org/abs/2606.26936","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00571","type":"paper","title":"MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG","authors":["Inderjeet Singh","Andrés Murillo","Motoyoshi Sekiya","Yuki Unno","Junichi Suga"],"year":2026,"abstract":"Multimodal agentic retrieval-augmented generation (RAG) systems expand the attack surface beyond prompt injection to include text poisoning, image injection, direct-query attacks, and orchestrator-level tool manipulation. Existing red-teaming approaches are typically surface-specific and often recycle known attack templates; on text-poisoning benchmarks we measure 73-84% exact duplication. We present MIRROR, a unified cross-surface framework that performs memory-guided Monte Carlo tree search wh","url":"https://arxiv.org/abs/2606.26793","categories":["prompt-injection","data-poisoning","agentic-threats","red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00572","type":"paper","title":"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents","authors":["Nada Lahjouji","Ashwin Gerard Colaco"],"year":2026,"abstract":"Large language model agents increasingly query databases, search document collections, call external APIs, remember past interactions, and act on a user's behalf. As they move from answering questions to operating over sensitive data, privacy becomes harder to enforce. An agent touches many data sources, runs multi-step workflows, keeps state across sessions, and acts with delegated permissions. Sensitive information can therefore leak not only through its final answer but through the queries it","url":"https://arxiv.org/abs/2606.26627","categories":["membership-inference","agentic-threats","survey"],"reviewed":false},{"id":"llmsec-2026-00573","type":"paper","title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","authors":["Abrar Alotaibi","Moataz Ahmed"],"year":2026,"abstract":"Adversarial evaluation of AI systems has matured along four largely disconnected tracks: diffusion-based attacks on text and large language models (LLMs), diffusion-based attacks on image classifiers, jailbreak pipelines against vision-language models, and diffusion-based input purification defenses. Each has developed its own vocabulary, threat models, and benchmarks, with denoising diffusion models emerging as a shared generative mechanism whose recipes are now actively ported between communit","url":"https://arxiv.org/abs/2606.26566","categories":["jailbreaking","benchmarks","survey","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00574","type":"paper","title":"Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents","authors":["Praneeth Narisetty","Shiva Nagendra Babu Kore","Uday Kumar Reddy Kattamanchi","Jayaram Kumarapu"],"year":2026,"abstract":"Recent work (2024 to 2026) has converged on a strategy for defending tool-using LLM agents against indirect prompt injection: rather than training the model to refuse malicious instructions, enforce security outside the model with a deterministic policy that mediates the agent's actions. Systems such as CaMeL, FIDES, Progent, RTBAS, and FORGE realize this with capabilities, information-flow labels, and reference monitors, and several report near-elimination of attacks on the AgentDojo benchmark.","url":"https://arxiv.org/abs/2606.26479","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00575","type":"paper","title":"RAS: Measuring LLM Safety Through Refusal Alignment","authors":["Chang-Chieh Huang","Yan-Lun Chen","Chia-Mu Yu","Wei-Bin Lee"],"year":2026,"abstract":"Safety evaluation of large language models (LLMs) is commonly performed by querying models with unsafe or jailbreak prompts and judging whether their outputs violate a safety policy. Although useful, output-level evaluation is expensive, sensitive to judge choice, and easily tied to fixed question banks. We propose **SafeVec**, a white-box evaluation procedure that measures safety from internal representations rather than generated answers. **SafeVec** first extracts layer-wise refusal direction","url":"https://arxiv.org/abs/2606.25750","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00576","type":"paper","title":"How Reliable Is Your Jailbreak Judge? Calibration and Adversarial Robustness of Automated ASR Scoring","authors":["Yang Gao"],"year":2026,"abstract":"Almost every paper on LLM jailbreaks and prompt injection reports an attack-success rate (ASR), and that number is assigned not by people but by an automated judge: either a safety classifier trained for the task, or a general chat model prompted to grade. The judge is rarely checked. We check it. Using 596 human-labeled completions from the HarmBench classifier validation set, we compare the two judge families against human majority votes and then attack them. The two families fail in opposite ","url":"https://arxiv.org/abs/2606.25487","categories":["prompt-injection","jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00577","type":"paper","title":"PixJail: Self-Evolving Paper-to-Pipeline Reproduction for Text-to-Image Jailbreak Evaluation","authors":["Leyi Sheng","Han Sun","Zhen Sun","Yuntao Yue","Jinlin Wu","Xinlei He","Jiaheng Wei"],"year":2026,"abstract":"As Text-to-Image (T2I) jailbreak techniques evolve rapidly, existing benchmarks and reproduction workflows often struggle to keep pace. More importantly, T2I jailbreak evaluation is not a single prompt-level test, but a pipeline-level problem shaped by multiple stages, including prompt transformation, image generation, safety filtering, and multimodal judging. This makes results across papers difficult to reliably reproduce and fairly compare. To bridge this gap, we propose PixJail, a self-evolv","url":"https://arxiv.org/abs/2606.24081","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00578","type":"paper","title":"TROPT: An Open Framework for Unifying and Advancing Discrete Text Optimization","authors":["Matan Ben-Tov","Mahmood Sharif"],"year":2026,"abstract":"Discrete text-trigger optimization -- searching for text sequences that, when ingested by a model, steer it toward a specified objective -- underpins model red-teaming (e.g., LLM jailbreaks), as well as auditing and interpretability. However, the current state of discrete optimizers hinders their adoption and progress. First, existing optimizers, when open-sourced at all, are scattered across research codebases tied to specific models, objectives, and problem domains. Second, optimizer variants ","url":"https://arxiv.org/abs/2606.23496","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00579","type":"paper","title":"Detecting Malicious Agent Skills in the Wild using Attention","authors":["Bacem Etteib","Daniele Lunghi","Tégawendé F. Bissyandé"],"year":2026,"abstract":"LLM agents increasingly load skills, file-based packages of natural-language instructions written by third parties and distributed through marketplaces, that execute with the user's privileges. A single malicious skill can exfiltrate data, hijack the agent, or persist as a supply-chain foothold, which turns the skill marketplace into a new attack surface for agentic systems. Prompt-injection defenses do not carry over to this setting. They rely on a boundary between trusted instructions and untr","url":"https://arxiv.org/abs/2606.23416","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00580","type":"paper","title":"DE-FIVE: Detecting Malicious Image Prompts via Fourier Features and Image Vector Embeddings","authors":["Xingwei Zhong","Varun Sharma","Kar Wai Fok","Vrizlynn L. L. Thing"],"year":2026,"abstract":"Vision language models (VLMs) employ both visual and textual modalities to enable advanced vision-language inference. However, incorporating visual modalities expands the attack surface of VLMs, making them more susceptible to security threats such as adversarial perturbations and indirect prompt injection, wherein crafted malicious image prompts can elicit unintended model outputs. Existing defense methods against malicious image prompts remain insufficient as they typically demand extensive da","url":"https://arxiv.org/abs/2606.22779","categories":["prompt-injection","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00581","type":"paper","title":"The Geometry of Refusal: Linear Instability in Safety-Aligned LLMs","authors":["Shivam Ratnakar","Kartikeya Vats"],"year":2026,"abstract":"Modern Large Language Models (LLMs) rely on extensive safety alignment, yet the mechanistic basis of refusal remains opaque. In this work, we investigate whether safety compliance is a deep semantic decision or a manipulable linear feature. We introduce Contrastive Logit Steering (CLS), a zero-optimization framework that isolates the \"refusal direction\" by contrasting hidden states derived from safe and unrestricted system prompts. Unlike representation engineering methods that intervene on inte","url":"https://arxiv.org/abs/2606.22686","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00582","type":"paper","title":"Confidently Wrong: Severity-Aware Calibration of Prompt-Injection Detectors under Attack Shift","authors":["Md Anas Biswas"],"year":2026,"abstract":"Prompt-injection detectors are deployed as guards: a model scores an input and a downstream system trusts or blocks it on that score. I study the confidence of these scores, not only their accuracy, when the attack distribution shifts away from the clean benchmark on which the operating point was chosen. I evaluate three released detectors, ProtectAI-v2 and two Prompt-Guard-2 checkpoints, at a single source-calibrated threshold that I freeze and transport across five shifts. I report a severity ","url":"https://arxiv.org/abs/2606.22659","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00583","type":"paper","title":"Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents","authors":["Zesen Liu","Zihan Zhang","Dongdong She"],"year":2026,"abstract":"Summarization-based prompt compression is increasingly used by LLM agents to shorten long, distributed contexts, but it shifts the security boundary: filters inspect the pre-compression prompt while the backend acts on a newly generated compressed context. We identify relinking, a compression-boundary vulnerability where the compressor behaves as a confused deputy, summarizing distributed, locally benign fragments into a complete malicious instruction. Unlike prompt injection, relinking need not","url":"https://arxiv.org/abs/2606.21732","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00584","type":"paper","title":"Toward Open Weight Models Without Risks: Separating Public and Private Capabilities in LLMs","authors":["Charbel El Feghali","Arkil Patel","Nicholas Meade","Spandana Gella","Verna Dankers","Siva Reddy"],"year":2026,"abstract":"Open-weight Large Language Models (LLMs) enable scientific progress and broad deployment. However, they make it difficult to control access to sensitive capabilities. Current practice either suppresses dangerous capabilities before release or mediates access through closed services that use specialized model variants, input/output monitors, and API permissions. The former is susceptible to jailbreaks while sacrificing capability for all users to mitigate the risks posed by a few, and the latter ","url":"https://arxiv.org/abs/2606.21638","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00585","type":"paper","title":"AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents","authors":["Zhen Zhao","Yu Zhang","Yanpeng Zhu","Jia Wang","Songqiao Tao","Xin Cheng","Jiexin Gao"],"year":2026,"abstract":"Traditional OS security models based on \"resource exposure plus permission checks\" face structural challenges as LLM-driven autonomous agents acquire capabilities for planning, tool use, network access, and code execution. Once an agent runtime is compromised through prompt injection or malicious tool outputs, an attacker can compose POSIX-style resource primitives into behaviors far beyond the user's task authorization. To address this, we propose AgenticOS, an intent-oriented secure OS archite","url":"https://arxiv.org/abs/2606.21129","categories":["prompt-injection","agentic-threats","access-control","tool-use-security","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00586","type":"paper","title":"Scalable Hierarchical Attention Transformers for Multi-Turn Jailbreak Detection in Long Conversations","authors":["Chenhui Hu","Muhammed Salih","Sudipto Guha","Subramanian Srinivasan"],"year":2026,"abstract":"Multi-turn jailbreaks can evade turn-level moderation by spreading unsafe intent across a dialogue through gradual escalation, reframing, and role manipulation. We address multi-turn jailbreak detection as a conversation-level classification problem and introduce an efficient hierarchical detector that avoids expensive long-context concatenation while retaining cross-turn reasoning. The model encodes individual turns to form compact turn representations and applies a lightweight conversation mod","url":"https://arxiv.org/abs/2606.21082","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00587","type":"paper","title":"Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning","authors":["Shanghao Shi","Xiao Wang","Chaoyu Zhang","Hao Li","Wenjing Lou","Thomas Hou","Yevgeniy Vorobeychik","Chongjie Zhang","Ning Zhang"],"year":2026,"abstract":"The integration of external tools has substantially expanded the capabilities of large language model (LLM) agents, but it also introduces new attack surfaces beyond prompt injection. In particular, cross-tool description poisoning can manipulate planner-visible tool metadata to steer an agent's trajectory, even if the poisoned tool itself is never chosen. To understand the effectiveness of existing defenses against this emerging threat, we first evaluate several prompt-injection defenses and fi","url":"https://arxiv.org/abs/2606.20922","categories":["prompt-injection","data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00588","type":"paper","title":"Amplify, Don't Create: Temporal Accumulation for Slow-Burn Prompt Injection","authors":["J Alex Corll"],"year":2026,"abstract":"Most prompt-injection detectors score a single event or message. Control-plane attacks against tool-using agents can instead distribute weak directives across a trajectory while keeping each event below threshold. We test whether a proxy-side temporal accumulator recovers this slow-burn signal by reducing frozen per-event scores to peak and CUSUM persistence statistics. To avoid circularity, grafts are generated against a held-out autoregressive cloaking target and then re-scored under a detecto","url":"https://arxiv.org/abs/2606.20746","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00589","type":"paper","title":"MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents","authors":["Xuelong Dai","Jianyu Ma","Boyang Ma","Biwei Yan","Yijun Yang","Yue Zhang"],"year":2026,"abstract":"Multimodal Large Language Model (MLLM)-based web agents provide practical, high-precision solutions for visual browser automation; however, they inherently expand the attack surface, introducing novel vision-based vulnerabilities. Existing adversarial evaluations targeting these agents frequently rely on permissive threat models and visually conspicuous artifacts. In this paper, we investigate a constrained vulnerability detection setting: a trusted web platform where the evaluator acts solely a","url":"https://arxiv.org/abs/2606.20717","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00590","type":"paper","title":"BELLS-O: Evaluating the Operational Trade-offs of LLM Supervision Systems","authors":["Leonhard Waibl","Felix Michalak","Hadrien Mariaccia"],"year":2026,"abstract":"LLM supervision systems, namely input/output moderation filters and jailbreak detectors, are the primary safeguard against misuse in deployed AI applications, yet existing benchmarks are often vendor-biased, omit cost and latency, and rarely compare specialized guardrails against repurposed generalist LLMs. We present BELLS-O (Benchmark for the Evaluation of LLM Supervision Systems, Operational), the first independent operational benchmark of LLM supervision systems. BELLS-O evaluates 28 systems","url":"https://arxiv.org/abs/2606.20668","categories":["jailbreaking","output-moderation","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00591","type":"paper","title":"Detect, Unlearn, Restore: Defending Text Summarization Models Against Data Poisoning","authors":["Poojitha Thota","Shirin Nilizadeh"],"year":2026,"abstract":"Training-time data poisoning during fine-tuning poses a significant threat to large language models (LLMs) deployed for abstractive text summarization, where small task-specific datasets exert disproportionate influence on model behavior. In this setting, adversaries manipulate fine-tuning data to induce persistent summarization failures, such as biased or harmful summaries, while preserving standard evaluation metrics. We present a unified post-hoc defense framework for detecting and remediatin","url":"https://arxiv.org/abs/2606.26036","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00592","type":"paper","title":"Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory","authors":["Beining Wu","Zihao Ding","Jun Huang","Yanxiao Zhao"],"year":2026,"abstract":"On-device language-model agents improve by accumulating experience in retrieved memory rather than by updating weights. This memory is hard-bounded and exposed: it consumes RAM and energy, reaches peers through a thin uplink, and becomes an attack surface because it is writable by what the agent reads. Existing systems each cover one part of this problem: agentic memories grow without a budget, on-device methods keep entries by success alone, and poisoning is studied mainly as an attack rather t","url":"https://arxiv.org/abs/2606.25115","categories":["data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00593","type":"paper","title":"LLM-Based Scientific Peer Review: Methods, Benchmarks, and Reliability Challenges","authors":["Thi Huyen Nguyen","Zahra Ahmadi"],"year":2026,"abstract":"The rapid growth of scientific submissions has pushed traditional peer review toward its scalability limits, motivating the exploration of large language models (LLMs) as intelligent automated evaluation assistants. Although recent studies show that LLMs can generate fluent critiques and approximate reviewer scores, their reliability, robustness, and security as decision-support systems remain insufficiently understood. This survey offers a systems-level analysis of LLM-based scientific peer rev","url":"https://arxiv.org/abs/2606.25057","categories":["benchmarks","survey"],"reviewed":false},{"id":"llmsec-2026-00594","type":"paper","title":"The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems","authors":["Seth Dobrin","Łukasz Chmiel"],"year":2026,"abstract":"AI agents are granted access to tools, APIs, and other infrastructure, making them active principals in those systems. The dominant approach places controls inside the agent's own runtime: system prompts, output filters, and guardrail libraries. Any control in the agent's address space is reachable by inputs that influence it; this generalizes to any AI system with sufficient reach into its own runtime, a class we term escapable AI systems. We identify four properties that an authorization mecha","url":"https://arxiv.org/abs/2606.26057","categories":["output-moderation","guardrails","access-control"],"reviewed":false},{"id":"llmsec-2026-00595","type":"paper","title":"RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems","authors":["Yarin Yerushalmi Levi","Roy Betser","Amit Giloni","Lidor Erez","Itay Gershon","Oren Rachmil","Sindhu Padakandla","Roman Vainshtein"],"year":2026,"abstract":"Agentic AI systems powered by large language models (LLMs) are rapidly evolving into autonomous decision-making systems, exposing attack vectors beyond those of traditional LLM vulnerabilities. Existing security evaluations are often tied to specific implementations or domains, limiting unified comparison across heterogeneous systems. To address this gap, we introduce RIFT-Bench, a graph representation-driven methodology for dynamic red-teaming that enables unified evaluations across diverse age","url":"https://arxiv.org/abs/2606.23927","categories":["agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00596","type":"paper","title":"Rising From the Ashes: How Agentic AI is Unblocking Challenges in Cybersecurity","authors":["Gabriela F. Ciocarlie","Kathrin Grosse","Somesh Jha","Daryna Oliynyk","Andrew Paverd","Christian Wressnegger"],"year":2026,"abstract":"Security remains a high-cost challenge, with many problems historically deemed inefficient to address or effectively unsolvable. A significant number of these problems stem from labor-intensive tasks that create bottlenecks in defensive approaches. Agentic AI has the potential to alleviate these bottlenecks by directly ingesting and reasoning over natural language or code, thereby expanding the scope of feasible defenses. In this paper, we map open security problems to emergent agentic AI capabi","url":"https://arxiv.org/abs/2606.23138","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00597","type":"paper","title":"AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems","authors":["Srimonti Dutta","Akshata Kishore Moharir"],"year":2026,"abstract":"Agentic AI systems retrieve private context, invoke tools, write files, call external services, coordinate with other agents, and may act without human approval. Existing bill of materials artifacts improve transparency for dependencies, model metadata, and training provenance, but leave an agentic transparency gap: capability opacity, the absence of a structured account of what a deployed agent can access, remember, change, delegate, and prove afterward. This paper introduces AgentRiskBOM, a se","url":"https://arxiv.org/abs/2606.21877","categories":["agentic-threats","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00598","type":"paper","title":"Antaeus: Hunting Repository-Level Logic Vulnerabilities via Context-Grounded LLM Reasoning","authors":["Michele Armillotta","Nicolò Romandini","Rebecca Montanari","Lorenzo Cavallaro"],"year":2026,"abstract":"LLM-based vulnerability detectors have shown promising results in identifying memory-safety bugs and vulnerability classes whose violations can often be expressed through established security properties. Logic vulnerabilities, however, pose a different challenge, as their identification requires inferring application-specific security invariants and implicit assumptions about intended behavior. Even frontier agentic models struggle because these invariants are often implicit and buried among unr","url":"https://arxiv.org/abs/2607.01138","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00599","type":"paper","title":"A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems","authors":["Seyed Bagher Hashemi Natanzi","Bo Tang"],"year":2026,"abstract":"Large language models are no longer only text generators. They are increasingly embedded in retrieval pipelines, enterprise assistants, coding environments, robotic systems, security-operation workflows, and autonomous agents that can read private data, call tools, write files, execute code, and act across organizational boundaries. This shift changes the security problem: risks do not arise from the model weights alone, but from the full lifecycle and application stack through which data, promp","url":"https://arxiv.org/abs/2606.31639","categories":["autonomous-operations","survey"],"reviewed":false},{"id":"llmsec-2026-00600","type":"paper","title":"Toward Secure and Reliable PDDL Formalization of Large Language Models with Planner-in-the-Loop Feedback","authors":["Jiamei Jiang","Jiajing Zhang","Feifei Mo","Linjing Li","Daniel Zeng"],"year":2026,"abstract":"Planning often requires symbolic specifications that are both executable and verifiable. For large language models deployed in autonomous or decision-support systems, failures in such formalization may lead to unverifiable decisions, execution failures, or unsafe downstream behavior. We present NL-PDDL-Bench, a multi-domain benchmark for natural-language-to-PDDL specification construction with planner-verified executability and controlled difficulty scaling by object count. We further propose a ","url":"https://arxiv.org/abs/2606.29700","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00601","type":"paper","title":"An Empirical Evaluation of Prompt Injection Vulnerabilities in Large Language Models Across Multilingual and Obfuscated Attack Scenarios","authors":["Caglar Uysal","Baturay Birinci","Süha Orhun Mutluergil","Orçun Çetin"],"year":2026,"abstract":"Large Language Models (LLMs) have rapidly evolved, transforming industries by automating complex tasks and generating human-like content. However, as their adoption accelerates, prompt injection vulnerabilities have become increasingly apparent. Malicious actors exploit these weaknesses to generate phishing emails, deceptive websites, nd malware, posing serious security risks. This paper presents an empirical evaluation of six state-of-the-art LLMs (DeepSeek, GPT, Gemini, Grok, Llama, and Qwen) ","url":"https://arxiv.org/abs/2606.29602","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00602","type":"paper","title":"Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering","authors":["Shahnewaz Karim Sakib","Anindya Bijoy Das"],"year":2026,"abstract":"AI agents extend conventional large language model (LLM) applications by integrating language understanding with task execution, external tool use, and memory mechanisms. While memory allows agents to retain prior interactions and provide more personalized and context-aware responses, it also introduces a new vulnerability: information stored in memory can influence future outputs even when the current query is clean. In this paper, we investigate memory manipulation in LLM-based agents for mult","url":"https://arxiv.org/abs/2606.29030","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00603","type":"paper","title":"FlipGuard: Defending Large Language Models Against Quantization-Conditioned Backdoor Attacks","authors":["Aoying Zheng","Anqi Du","Zizhuang Deng","Yuxuan Chen"],"year":2026,"abstract":"Model quantization is essential for the efficient deployment of Large Language Models (LLMs), but introduces a critical vulnerability: Quantization-Conditioned Backdoor (QCB) attacks. In these attacks, malicious behaviors remain dormant in full-precision models and activate only after specific quantization distortions, bypassing standard security audits. To mitigate this, we introduce FlipGuard, a proactive defense framework that selectively perturbs model weights prior to quantization. By break","url":"https://arxiv.org/abs/2606.28962","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00604","type":"paper","title":"RIPA: Sensory-Vector Prompt Injection Attacks on LLM-Controlled ROS 2 Robots","authors":["Nima Dorzhiev"],"year":2026,"abstract":"We present RIPA, the first systematic multi-channel empirical study of prompt injection attacks delivered through the sensory pipeline of a ROS 2-based LLM-controlled robotic system. Across 100 independent runs per injection variant on five LLMs spanning four model families and parameter scales from approximately 4B to approximately 284B (DeepSeek-V4-Flash, Llama-3-8B-Instruct-Lite, Llama-3.3-70B-Instruct-Turbo, Qwen 2.5-7B-Instruct-Turbo, Gemma-3n-E4B), we identify model-specific vulnerability ","url":"https://arxiv.org/abs/2606.28649","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00605","type":"paper","title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","authors":["Yanchen Yin","Dongqi Han","Linghui Li"],"year":2026,"abstract":"Jailbreak attacks bypass LLM safety alignment, yet their mechanisms remain poorly understood. We provide evidence that attacks do not comprehensively eliminate safety features, but instead selectively suppress specific attention heads. We identify two functionally differentiated types: Adversarially Compromised Heads (ACHs) concentrated in early layers, which are suppressed under attacks, and Safety-Aligned Heads (SAHs) in mid-layers, which maintain robust activations even when attacks succeed. ","url":"https://arxiv.org/abs/2606.28153","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00606","type":"paper","title":"LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity","authors":["Yiwei Xu","Yong Zhuang","Xuanming Liu","Tian Zhang","Bowen Xiao","Xiaoyang Xu","Delong Jiang","Juan Wang","Hongxin Hu"],"year":2026,"abstract":"Large language model (LLM) agents are rapidly being integrated into real-world systems. Their autonomy and tool-use capabilities generate substantial value while simultaneously expanding the security attack surface. This survey provides a comprehensive overview of the opportunities and challenges of LLM agents in security, focusing on two core areas: (1) threats to LLM agents themselves and corresponding mitigation strategies (LLM agents self-security), and (2) the role of LLM agents in empoweri","url":"https://arxiv.org/abs/2606.28450","categories":["agentic-threats","survey"],"reviewed":false},{"id":"llmsec-2026-00607","type":"paper","title":"The Consistency Dilemma in LLMs: Generator-Evaluator Agreement and Vulnerability to Mistakes","authors":["Marina Mancoridis","Zoë Hitzig"],"year":2026,"abstract":"Large language models are increasingly deployed in agentic pipelines that depend on the model evaluating its own outputs without external verification. The reliability of these pipelines depends on an implicit assumption: that the model applies relevant concepts the same way when it generates an output and later evaluates that output. We propose a new measure, generator-evaluator self-consistency, to test this assumption directly and apply it to 10 frontier models across 491 concepts. We find, f","url":"https://arxiv.org/abs/2606.30653","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00608","type":"paper","title":"HARC: Coupling Harmfulness and Refusal Directions for Robust Safety Alignment","authors":["Shei Pern Chua","Fangzhao Wu"],"year":2026,"abstract":"Understanding how aligned LLMs internally represent safety is critical for diagnosing alignment vulnerabilities, as it explains why jailbreaks succeed and informs the design of robust alignment strategies. Prior work shows that aligned LLMs encode harmfulness and refusal as separable directions in the residual stream at prompt-side token positions. We show that jailbreaks succeed at prompt encoding by suppressing either the refusal or harmfulness direction before any token is generated, with dis","url":"https://arxiv.org/abs/2607.00572","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00609","type":"paper","title":"Beyond the Prompt: Jailbreaking Function-Calling LLMs via Simulated Moderation Traces","authors":["Junlong Liu","Haobo Wang","Weiqi Luo","Xiaojun Jia"],"year":2026,"abstract":"Jailbreak attacks remain a critical threat to the safe deployment of large language models (LLMs). While prior work has primarily studied attacks and defenses at the prompt level, we show that this prompt-centric paradigm overlooks a structural vulnerability in stateful, function-calling environments. In such applications, developer-defined schemas, structured arguments, and untrusted tool outputs are interleaved into a single shared model context. This architecture expands the attack surface by","url":"https://arxiv.org/abs/2607.00481","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00610","type":"paper","title":"Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming","authors":["Yong Yang","Xing Zheng","Huiyu Wu","Huangsheng Cheng","Xiaorong Shi","Jing Guo","Bo Yang","Yi Zhou","Xiangfan Wu","Zonghao Ying"],"year":2026,"abstract":"The fast growth of open-source AI infrastructure, from model serving engines and agent platforms to the Model Context Protocol (MCP) ecosystem and the language models themselves, has outpaced the security tooling available to defend it. We present AI-Infra-Guard, an open-source framework that organizes AI red teaming around a single observation: the attack surface of an AI agent is stratified across layers (infrastructure, protocol/tool, agent behavior, and model), and no single detection paradi","url":"https://arxiv.org/abs/2606.31227","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00611","type":"paper","title":"Security--Fidelity Tradeoffs: The Hidden Cost of Prompt Injection Defense","authors":["Mitchell Hermon","Rahul Gupta","Weitong Ruan","Ekraam Sabir","Haohan Wang"],"year":2026,"abstract":"We identify a security-fidelity tradeoff in defending LLMs against indirect prompt injection: defenses resist injected instructions largely by suppressing untrusted text, which corrupts tasks that must preserve it, such as translation and document editing. Attack-success metrics cannot see this, because a model that ignores an injection and one that faithfully processes it as data score identically. We introduce SecFid, a benchmark built so that executing an injection, processing it as data, and","url":"https://arxiv.org/abs/2606.30783","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00612","type":"paper","title":"Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens","authors":["Peizhi Niu","Wenjie Qu","Shangding Gu","Tianneng Shi","Yuankai Li","Ahmad Tawaha","Hend Alzahrani","Vincent Siu","Boyi Li","Chenguang Wang","Jiaheng Zhang","Basel Alomair","Ming Jin","Muhao Chen","Chi Wang","Costas Spanos","Dawn Song"],"year":2026,"abstract":"Claw-like AI agents (e.g., OpenClaw) are always-on processes with persistent access to credentials, files, tools, and external services. They take on system-level responsibilities -- installing packages, maintaining state, scheduling subtasks, and mediating I/O -- making security failures far more severe than in other agents. Yet existing benchmarks focus on model responses and tool calls, leaving cross-component failure modes largely unmeasured. We adopt a computer-system analogy: treating a Cl","url":"https://arxiv.org/abs/2606.30755","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00613","type":"paper","title":"Forensic Trajectory Signatures for Agent Memory Poisoning Detection","authors":["Jun Wen Leong"],"year":2026,"abstract":"We discover a behavioral invariant in LLM agents under persistent memory poisoning: in architectures where routing information is retrieved through observable memory-tool invocations, successful attacks require calling memory_recall_fact before email_send_email, a transition that non-exfiltrating sessions rarely exhibit. Under the evaluated architecture, this invariant follows from the attack's information-retrieval dependency rather than being merely an empirical correlation, and suppressing it","url":"https://arxiv.org/abs/2606.30566","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00614","type":"paper","title":"On the Inseparability of Instructions and Data in Shared-Embedding Sequence Models","authors":["Dewank Pant","Shruti Lohani","Avijit Kumar"],"year":2026,"abstract":"Prompt injection is the top security risk for LLM-integrated applications, yet every defense proposed so far has been broken. We prove this is not a coincidence: in shared-embedding architectures that lack enforced control-data separation, perfect prompt-injection prevention is mathematically impossible. We formalize prompted systems as Prompted Action Models whose outputs include control-authoritative actions: refusal decisions, tool authorization, policy routing, and memory writes. We define S","url":"https://arxiv.org/abs/2606.27567","categories":["prompt-injection","access-control","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00615","type":"paper","title":"Adversarial Pragmatics for AI Safety Evaluation: A Benchmark for Instruction Conflict, Embedded Commands, and Policy Ambiguity","authors":["Brett Reynolds"],"year":2026,"abstract":"Safety evaluations for language models increasingly depend on judgments about ambiguous natural-language behaviour: whether a model has followed an instruction, refused appropriately, complied with a policy, resisted an embedded command, or misreported progress in an agentic task. Existing benchmarks often compress these distinctions into pass/fail labels, obscuring whether failures arise from capability limits, policy ambiguity, instruction conflict, scaffold failure, or unstable evaluator judg","url":"https://arxiv.org/abs/2607.01153","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00616","type":"paper","title":"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety","authors":["Ting Ma","Xiufeng Huang","Benlei Cui","Xiaowen Xu","Shikai Qiu","Ruijie Jian","Hongxing Li","Guanghui Wang","Longtao Huang","Haiwen Hong","Haolei Xu","Wenjing Jiang","Ziwen Xu","Zhaoyu Fan","Shaoxuan He","Chuxi Xiao","Yujian Li","Xinyue Chen","Chunyang Chai","Wenxuan Liu","Ziheng Wang","Dongjie Zhang","Yangfan Zhou","Libin Dong","Yupeng Cao","Xiaoqian Xia","Jing Wang","Zhe Jiang","Zhenan Ye","Guang Yang","Bin Liu","Wei Peng","Ziqiang Zhu","Meihui Lian","Kaiwen Lv Kacuila","Haidong Ding","Bingyu Zhu","Yan Wang","Hai Zhao","Xuan Jin","Wei Zhao","Pengfei Sun","Wei Wang","Huiming Zhang","Bin Li","Hui Xue"],"year":2026,"abstract":"As large language models are increasingly deployed in real-world systems, safety failures can still lead to harmful outputs and dangerous misuse. We argue that the essence of safety is adversarial: many failures arise not from natural inputs alone, but from strategic attempts to evade model policies and safeguards. However, existing general-purpose model development largely overlook this adversarial nature, and often remain insufficient for realistic safety scenarios involving planning, tool use","url":"https://arxiv.org/abs/2606.27632","categories":["guardrails","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00617","type":"paper","title":"ReShift: Aha-Moment-Driven Reasoning-Level Backdoor Attacks on Vision-Language Models","authors":["Zhihao Dou","Qinjian Zhao","Zhiqiang Gao","Sumon Biswas"],"year":2026,"abstract":"Vision--Language Models (VLMs) are increasingly deployed in safety-critical applications, yet remain vulnerable to backdoor attacks. Existing methods primarily manipulate final outputs, often producing reasoning traces that are inconsistent or easily detectable. In this paper, we propose ReShift, the novel aha-moment-driven reasoning-level backdoor framework that explicitly redirects the internal chain-of-thought (CoT) trajectory while preserving surface-level coherence. ReShift introduces a Poi","url":"https://arxiv.org/abs/2607.00361","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00618","type":"paper","title":"Curvature-Guided Module Localization for Low-Rank Detoxification of Backdoored Large Language Models","authors":["Arash Raftari","Mehrdad Mahdavi","Nathan Blackthorn","Andrew Arash Mahyari"],"year":2026,"abstract":"Backdoor attacks pose a serious threat to large language models (LLMs) by causing otherwise benign systems to produce attacker-specified malicious behavior when a hidden trigger is present. In this work, we study post hoc detoxification of backdoored LLMs in a practical setting where the defender has access to the poisoned model but does not wish to retrain the full network from scratch. We propose a mechanistically guided weight-space repair framework that first localizes modules involved in pr","url":"https://arxiv.org/abs/2606.30899","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00619","type":"paper","title":"Theory of Continual Learning Against Data Poisoning Attacks","authors":["Yiting Hu","Lingjie Duan"],"year":2026,"abstract":"Continual learning (CL), where a model is trained on a sequence of data tasks, is increasingly being adopted across key fields such as large language models and image recognition, yet it remains highly vulnerable to data poisoning that triggers learning divergence or severe excess risk. Despite these threats, a principled theoretical foundation in CL for understanding attack and defense remains lacking. In this paper, we develop a theoretical framework to analyze strategic attacks and defenses i","url":"https://arxiv.org/abs/2606.29841","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00620","type":"paper","title":"Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare","authors":["Liam Kearns"],"year":2026,"abstract":"Agent-based AI has enabled the automation of tasks by exposing application tools and resources to large language models (LLMs). However, to improve scope and accuracy, agents are often given access rights that exceed those of ordinary users, introducing significant security risks. AI is routinely integrated into applications with a disregard to security, risking data exposure and breaching regulations. This paper applies the AI Trust, Risk, and Security Management (TRiSM) framework to a medical ","url":"https://arxiv.org/abs/2606.28666","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00621","type":"paper","title":"When the Aggregator Cheats: Data-Free Backdoors in Federated LLM-based QA Systems","authors":["Chenqing Zhu","Yanbo Dai","Yulong Tian","Qingming Li","Songze Li"],"year":2026,"abstract":"Large Language Model (LLM)-based question-answering (QA) systems are increasingly deployed in sensitive domains such as healthcare, mental health counseling, and legal consultation. Federated learning (FL) enables collaborative training without sharing raw client data, for which locally trained models are aggregated at a central server (i.e., a cloud service provider) to obtain a global model. In this paper, we explore the potential vulnerability where a malicious aggregator, who may collude wit","url":"https://arxiv.org/abs/2606.27511","categories":["data-poisoning","federated-learning"],"reviewed":false},{"id":"llmsec-2026-00622","type":"paper","title":"When Medical Safety Alignment Fails: A Benchmark for Evaluating LLMs on High-Risk Medical Queries","authors":["Yige Li","Jun Sun","Wei Zhao","Zhe Li","Yutao Wu","Hanxun Huang","Xiang Zheng","Xingjun Ma"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used for medical and health-related questions, yet their safety in high-risk medical scenarios remains poorly understood. We introduce \\textsc{MedHarm}\\footnote{Code and data will be released upon acceptance. Due to the sensitive nature of high-risk medical queries, data access will be available to qualified researchers upon request.}, a high-risk medical safety benchmark with 1,100 medically grounded queries across 10 safety-critical categories, inc","url":"https://arxiv.org/abs/2606.28332","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00623","type":"paper","title":"Hardware-Enforced Semantic Coordination for Safety-Critical Real-Time Autonomous Systems","authors":["Uwe M. Borghoff","Paolo Bottoni","Remo Pareschi"],"year":2026,"abstract":"Recent advances in agentic AI are producing increasingly complex autonomous systems that integrate large language models, world models, optimization engines, specialized neural architectures, autonomous platforms, and human operators. While much current research focuses on improving reasoning capabilities, safety-critical real-time deployment also requires bounded and verifiable coordination among heterogeneous components operating concurrently under uncertainty. Software-mediated coordination p","url":"https://arxiv.org/abs/2607.02376","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00624","type":"paper","title":"Janus: a Playground for User-Involved Agentic Permission Management","authors":["Natalie Grace Brigham","Eugene Bagdasarian","Tadayoshi Kohno","Franziska Roesner"],"year":2026,"abstract":"AI agents that autonomously execute tool calls on a user's behalf raise pressing questions about permission management: what role could users play, and what role should they play? Despite many proposed approaches, the user's role in agentic permission management remains under explored. We introduce Janus, a playground system for implementing and evaluating user-involved agentic permission management designs. Janus consists of two components: Janus-Core, a modular agentic system supporting a dive","url":"https://arxiv.org/abs/2607.01510","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00625","type":"paper","title":"Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems","authors":["Guoqin Tang","Qingxuan Jia","Yichen Tan","Zeyuan Huang","Ning Ji","Gang Chen"],"year":2026,"abstract":"Agentic artificial intelligence is increasingly used to decompose industrial tasks, propose robot actions, and adapt execution plans in dynamic cyber-physical environments. However, autonomous proposal generation alone does not guarantee that multi-robot industrial systems preserve task dependencies, resource ownership, safety holds, or repair boundaries during long-horizon execution. This paper introduces a verification-gated agentic mission-state governance framework for intelligent industrial","url":"https://arxiv.org/abs/2606.31339","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00626","type":"paper","title":"Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems","authors":["Jimmy Laurence Rippin","Simon C. Marshall","David Demitri Africa","Christian Schroeder de Witt"],"year":2026,"abstract":"Increasingly autonomous agentic AI systems pose novel multi-agent risks, such as secret collusion via covert communication channels. The natural defence to these collusion attempts is to monitor plain-text communication, but the efficacy of monitors has been called into doubt by increasingly sophisticated model steganography; indeed, some theoretical schemes have been proposed that are information-theoretically or computationally indistinguishable from good-faith plain-text communication. In thi","url":"https://arxiv.org/abs/2606.28425","categories":["agentic-threats","agent-architecture","tool-use-security","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00627","type":"paper","title":"Evaluating AI Risk and Governance in Generative AI Systems: A Prompt-Level Analysis","authors":["Dr. Abdul Majid Farooqi Dr. Abdul Majid Farooqi","Ziya Anjum Ziya Anjum"],"year":2026,"abstract":"Generative Artificial Intelligence systems—particularly those built on Large Language Models (LLMs)—have become central to modern enterprise computing, yet they carry with them a class of vulnerabilities that traditional cybersecurity models were never designed to address. Decoder-only transformer architectures process system instructions and untrusted user inputs as a single undifferentiated sequence of tokens, which makes them susceptible to direct and indirect prompt injections, jailb","url":"https://doi.org/10.55041/isjem08122","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00628","type":"paper","title":"When Agents Remember Too Much: Memory Poisoning Attacks on Large Language Model Agents","authors":["George Torres","Sharad Shrestha","Satyajayant Misra"],"year":2026,"abstract":"Personal AI agents powered by large language models can reason and act using available tools to access emails, manage calendars, and push code to remote repositories, all with minimal oversight. When augmented with long-term memory, an agent can recall specific details relevant to the current task, reducing the need for large context windows. Currently, long-term memory agents tend to fall into two distinct domains: conversational and action-planning agents. Personal assistant agents sit at the ","url":"https://arxiv.org/abs/2607.06595","categories":["data-poisoning","memory-security"],"reviewed":false},{"id":"llmsec-2026-00629","type":"paper","title":"Multi-Agent Firewall Architecture for Privacy Protection of Sensitive Data in Interactions with Language Models","authors":["Hugo García Cuesta","Pablo Mateo Torrejón","Alfonso Sánchez-Macián"],"year":2026,"abstract":"While Large Language Models (LLMs) have become essential productivity tools, their integration into workflows without adequate safeguards creates significant risks. This paper proposes an open-source, privacy-focused, user-facing firewall designed to secure both web-based and programmatic LLM interactions. The architecture combines a browser extension and a proxy for total traffic interception across both HTTP(S) and WebSocket communications. At its core, a flexible multi-agent pipeline delivers","url":"https://arxiv.org/abs/2607.08282","categories":["guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00630","type":"paper","title":"Prismata: Confining Cross-Site Prompt Injection in Web Agents","authors":["Corban Villa","Alp Eren Ozdarendeli","Sijun Tan","Raluca Ada Popa"],"year":2026,"abstract":"Autonomous web agents promise to automate everyday browsing tasks, but inherit one of the web's oldest attack surfaces. Cross-Site Scripting proved that mixing trusted and untrusted content is dangerous, even on benign pages. Agents resurface this risk by interpreting natural language as instructions, allowing third-party and user-generated content to hijack the agent via prompt injection. The core challenge is that deriving a task-specific security policy requires reasoning over page structure ","url":"https://arxiv.org/abs/2607.08147","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00631","type":"paper","title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","authors":["Mohamed Amine Merzouk","Nolan Smyth","Damiano Fornasiere","Linh Le","David Williams-King","Adam Oberman"],"year":2026,"abstract":"Current safety methods for large language models are known to be vulnerable to adversarial attacks, motivating research into robust alternatives. Latent Adversarial Training (LAT) is among the most effective defenses, but can degrade utility and requires training on large datasets of harmful prompts. We introduce Latent Personality Alignment (LPA), which replaces explicit harm refusal with adversarial training on just 66 harm-agnostic statements drawn from psychometric personality literature. We","url":"https://arxiv.org/abs/2607.07918","categories":["adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-00632","type":"paper","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","authors":["Anupam Wagle","Ifrat Ikhtear Uddin","Chaowei Zhang","Longwei Wang"],"year":2026,"abstract":"Large language models (LLMs) exhibit remarkable capabilities but remain highly vulnerable to adversarial prompts and jailbreak attacks. Existing approaches primarily analyze these failures through input-output behaviors or attribution methods, offering limited insight into how adversarial perturbations alter the model's internal reasoning. Consequently, the mechanisms underlying unsafe or incorrect behaviors remain poorly understood. We introduce a mechanistic framework for diagnosing LLM vulner","url":"https://arxiv.org/abs/2607.07903","categories":["jailbreaking","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00633","type":"paper","title":"Open Models, Open Risks: Measuring Unsafe Generation in Text-to-Image Models In the Wild","authors":["Peilin Han","Yang Liu","Yilong Yang","Jingchun Zhang","Teng Li","Jianfeng Ma","Zhuo Ma"],"year":2026,"abstract":"Existing safety studies on text-to-image (T2I) jailbreaks are largely conducted in controlled in-the-lab settings, typically on a small number of canonical models. As a result, the current safety status of the rapidly growing in-the-wild T2I ecosystem remains unclear. This uncertainty is amplified by two factors: existing detector-based metrics are designed for controlled evaluation, and in-the-wild risks may arise not only from adversarial prompting, but also from unsafe release practices and u","url":"https://arxiv.org/abs/2607.07827","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00634","type":"paper","title":"Beware of Agentic Botnets: Scalable Untargeted Promptware Attacks via Universal and Transferable Adversarial HalluSquatting","authors":["Aya Spira","Stav Cohen","Elad Feldman","Ron Bitton","Avishai Wool","Ben Nassi"],"year":2026,"abstract":"The growing adoption of agentic LLM applications has introduced a new threat previously named as promptware. While prior work has established that adversaries can exploit direct channels to LLM applications to apply promptware under weak threat models, many applications do not provide any direct channels that could be exploited for prompt injection beyond the Internet. This raises a question: can attackers exploit LLM applications at scale without any direct channels in practical threat models? ","url":"https://arxiv.org/abs/2607.07433","categories":["prompt-injection","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00635","type":"paper","title":"Untrusted Content Masking for Web Agents with Security Guarantees","authors":["Kristina Nikolić","Egor Zverev","Javier Rando","Matthew Jagielski","Edoardo Debenedetti","Florian Tramèr"],"year":2026,"abstract":"Defenses that provide security guarantees against prompt injection attacks rely on strict isolation between trusted instructions and untrusted data. In text-based environments such as tool-use APIs, this separation arises naturally: agents can reason from interface definitions without ever processing untrusted content. Extending these guarantees to web agents faces a fundamental challenge: to perceive and interact with their environment, web agents must first observe the rendered page, which int","url":"https://arxiv.org/abs/2607.05277","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00636","type":"paper","title":"Agent Data Injection Attacks are Realistic Threats to AI Agents","authors":["Woohyuk Choi","Juhee Kim","Taehyun Kang","Jihyeon Jeong","Luyi Xing","Byoungyoung Lee"],"year":2026,"abstract":"AI agents act on behalf of user prompts, consuming external data and taking actions based on the agent context. Prior research on AI agent security has primarily focused on indirect prompt injection (IPI). Its most well-studied category is instruction injection, where attacker-controlled untrusted data is interpreted as an instruction. In response, many mitigations have been proposed to prevent instruction injection attacks. In this paper, we introduce a new category of IPI, agent data injection","url":"https://arxiv.org/abs/2607.05120","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00637","type":"paper","title":"DualView: Preventing Indirect Prompt Injection in Personal AI Agents","authors":["Juhee Kim","Woohyuk Choi","Taehyun Kang","Youngmin Kim","Byoungyoung Lee"],"year":2026,"abstract":"Personal AI agents that run on the user's local machine, such as OpenClaw, automate daily tasks including web search, email, and file management. Their access to computer resources, including the network, file system, and shell, exposes them to indirect prompt injection (IPI) attacks. Prior Dual LLM defenses block IPI by replacing untrusted data with symbols that the agent can reference but not read. However, they track untrusted data only inside the agent's context, so when the agent saves and ","url":"https://arxiv.org/abs/2607.03821","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00638","type":"paper","title":"Overloading Large Vision-Language Models for Jailbreaking","authors":["Haoyu Zhang","Yangyang Guo","Mohan Kankanhalli"],"year":2026,"abstract":"Large Vision-Language Models (LVLMs) exhibit remarkable vision-language capabilities and are increasingly deployed in real-world applications such as personal assistants, document analysis systems, and embodied agents. However, their dual-modal attack surfaces make them vulnerable to jailbreak attacks. Existing LVLM jailbreaks rely on simple designs, e.g., short text and out-of-distribution images. Nevertheless, recent advancements in both large language model backbones and multimodal mechanisms","url":"https://arxiv.org/abs/2607.02961","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00639","type":"paper","title":"The Poisoned Chalice of LLM Evaluation Report","authors":["Jonathan Katzy","Ali Al-Kaswan","Razvan Mihai Popescu","Zhou Yang"],"year":2026,"abstract":"Large language models are increasingly used to evaluate and support software engineering tasks, yet the validity of these evaluations is often undermined by uncertainty about whether benchmark instances were seen during pretraining. This can lead to data contamination, which may inflate performance and result in misleading conclusions about model capability. Despite this, the training corpora of many modern models are only partially disclosed, making direct decontamination infeasible. This creat","url":"https://arxiv.org/abs/2607.07481","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00640","type":"paper","title":"SolarChain-Eval: A Physics-Constrained Benchmark for Trustworthy Economic Agents in Decentralized Energy Markets","authors":["Shilin Ou","Yifan Xu","Luyao Zhang"],"year":2026,"abstract":"As agentic AI systems are increasingly applied to cyber-physical environments, their evaluation requires assessment of both task performance and trustworthiness. In decentralized energy markets, autonomous agents may improve market utility, but may also exploit invalid physical data, create artificial liquidity, and produce unstable governance decisions. Therefore, we propose SolarChain-Eval, a physics-constrained benchmark for evaluating trustworthy economic agents. It formulates market governa","url":"https://arxiv.org/abs/2607.08681","categories":["agentic-threats","benchmarks","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00641","type":"paper","title":"Persuasion Attacks Can Decrease Effectiveness of CoT Monitoring","authors":["Jennifer Za","Julija Bainiaksina","Nikita Ostrovsky","Tanush Chopra","Victoria Krakovna"],"year":2026,"abstract":"Chain-of-thought (CoT) monitoring is a promising safety mechanism for AI agents, based on the premise that visible reasoning traces can surface misaligned or deceptive behavior. While effective in standard scenarios, recent work highlights that LLMs remain vulnerable to persuasion-based jailbreaks, where natural-language arguments override model constraints. We stress-test whether this vulnerability extends to monitoring LLMs: can an adversarial agent persuade its CoT monitor to approve proposed","url":"https://arxiv.org/abs/2607.08066","categories":["jailbreaking","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00642","type":"paper","title":"Institutional Red-Teaming: Deployment Rules, Not Just Models, Causally Shape Multi-Agent AI Safety","authors":["Yujiao Chen"],"year":2026,"abstract":"We introduce institutional red-teaming, an evaluation methodology for testing deployment rules in multi-agent AI: hold the agents, objectives, and task state fixed, vary only one rule, and attribute the resulting change in collective behavior to that rule. We instantiate the methodology in IABench-CA, a consequence-allocation benchmark spanning 228 contexts, five canonical rules, and seven model populations (33,924 games), with a normative cooperative reference and auto-labelled reasoning traces","url":"https://arxiv.org/abs/2607.07695","categories":["red-teaming","benchmarks","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00643","type":"paper","title":"Multi-Agent AI Control: Distributed Attacks Hamper Per-Instance Monitors","authors":["Oliver Makins","Orazio Angelini","Zohreh Shams","Mary Phuong"],"year":2026,"abstract":"AI control is a family of techniques to prevent an AI with malicious goals from subverting its operator's intent. AI Control usually studies a single agent in one trajectory, but real deployments run many agents over shared infrastructure, and the most severe risks (model-weight exfiltration, training-run poisoning) plausibly need several agents acting in concert. We initiate the empirical study of multi-agent AI control, formalising distributed attacks in which several agents jointly aim for a ","url":"https://arxiv.org/abs/2607.07368","categories":["data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00644","type":"paper","title":"Security and Privacy in Agentic AI: Grand Challenges and Future Directions","authors":["Adam Jenkins","Agnieszka Kitkowska","Caterina Maidhof","Diego Paracuellos","Francesco Sovrano","Gonzalo Gabriel Mendez","Guillermo Suarez-Tangil","Hana Kopecka","Isabel Wagner","Isabel Barbera","Javier Carnerero-Cano","Jide Edu","Jose Luis Martin-Navarro","Jose Such","Josep Domingo-Ferrer","Juan Carlos Carrillo","Kopo Marvin Ramokapane","Mark Cote","Pablo Vellosillo","Ramon Ruiz-Dolz","Rongjun Ma","Ruba Abu-Salma","Sameer Patil","William Seymour","Xiao Zhan"],"year":2026,"abstract":"We present key challenges and future research directions in the security and privacy of agentic AI, based on a horizon-scanning exercise that brought together thirty leading international experts from academia, industry, and government to engage in focused discussions and collaborative exercises on the emerging risks associated with the growing agency of AI.","url":"https://arxiv.org/abs/2607.06608","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00645","type":"paper","title":"The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities","authors":["Mohammadreza Rashidi"],"year":2026,"abstract":"AI coding agents now read repositories, call tools, and execute shell commands with limited human oversight, and a fast-growing body of work studies whether the execution layer around them is actually safe. That literature is scattered. Papers on sandbox isolation, capability and access control, policy enforcement, time-of-check-to-time-of-use (TOCTOU) races, Model Context Protocol (MCP) threats, identity delegation, execution provenance, network egress control, and static analysis of agent-gene","url":"https://arxiv.org/abs/2607.05743","categories":["access-control","sandboxing-isolation","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00646","type":"paper","title":"aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents","authors":["Sai Varun Kodathala"],"year":2026,"abstract":"AI agents issue tool calls on the basis of text they cannot verify, so any party who controls part of the context can forge the appearance of authority. I evaluate 15 contemporary language models against eight attack scenarios derived from a published corpus of real agent incidents and find that refusal varies from 100% down to 38% across fully evaluated models; the most expensive model refused only half of the attacks despite a twentyfold price spread. I present aiAuthZ, an authorization gatewa","url":"https://arxiv.org/abs/2607.05518","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00647","type":"paper","title":"CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI","authors":["Roopam W. Sure"],"year":2026,"abstract":"Enterprise artificial intelligence is moving from experimentation into operational workflows. Early programs focused on model access and retrieval-augmented generation, but enterprises are now beginning to deploy agents that plan, retrieve, remember, call tools, update systems, and coordinate work across applications. This changes the evaluation problem. Leaders are no longer asking only whether an answer is accurate or fluent. They need to know who authorized an action, which policy applied, wh","url":"https://arxiv.org/abs/2607.03510","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00648","type":"paper","title":"Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies","authors":["Chris Schneider","Kriti Faujdar","Philipp Schoenegger","Ben Bariach"],"year":2026,"abstract":"Modern AI agent implementations such as frontier coding agents chain multiple tools at runtime that create a security surface that per-tool guardrails are unable to address, as individually permitted tools can violate organizational policies when composed. We propose the Dynamic Security Control Compositor (DSCC), a two-phase approach to compositional security for multi-tool agent chains. In Phase 1, at session checkout, a Most Restrictive Set (MRS) algorithm composes per-tool security policies ","url":"https://arxiv.org/abs/2607.03423","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00649","type":"paper","title":"Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks","authors":["Haoxiang Luo","Yinqiu Liu","Ruichen Zhang","Guangyuan Liu","Gang Sun","Hongfang Yu","Zhu Han","Dong In Kim"],"year":2026,"abstract":"The standard Practical Byzantine Fault Tolerance (PBFT) protocol, designed for stable, wired environments, exhibits critical vulnerabilities when deployed in settings like mobile ad-hoc networks, thus making it susceptible to sophisticated threats such as Sybil attacks, Byzantine collusion, and message manipulation. Existing static defense mechanisms are ill-equipped to handle the intelligent and coordinated nature of these attacks. To address this challenge, this paper leverages the Agentic AI ","url":"https://arxiv.org/abs/2607.03269","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00650","type":"paper","title":"PromptStudio: A Governance-Aware Agentic AI Framework for Automated Prompt Generation, Optimization, Evaluation, and Lifecycle Management","authors":["C. Swetha","G. Devi"],"year":2026,"abstract":"Prompt engineering plays a critical role in the effec-tive use of Large Language Models, but manual prompt design is\noften inconsistent, difficult to reproduce, and weakly governed. This paper presents PromptStudio, a governance-aware Agentic\nAI framework for automated prompt generation, optimisation, evaluation, and lifecycle management. The proposed framework\nintegrates specialised agents for prompt generation, optimisa-tion, evaluation, feedback, governance, and orchestration. It\nalso","url":"https://doi.org/10.22214/ijraset.2026.84092","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00651","type":"paper","title":"Exploiting large language models in peer review: indirect prompt injection attacks and integrity probes","authors":["Federico Torrielli","Stefano Locci","Amon Rapp","Luigi Di Caro"],"year":2026,"abstract":"Abstract\n                  Large language models are beginning to enter peer review as tools for summarizing manuscripts, drafting evaluations, and reducing reviewer workload. Yet this use creates a security problem specific to evaluative settings: the manuscript being judged can also contain hidden instructions that shape the model’s judgment. We investigate this risk through indirect prompt injection, where hidden text embedded in a manuscript is processed like","url":"https://doi.org/10.1007/s11192-026-05695-x","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00652","type":"paper","title":"CIVIL AND CRIMINAL LIABILITY OF DEEPFAKE AI PLATFORM PROVIDERS IN VIOLATIONS OF THE RIGHT TO REPUTATION","authors":["Milka Kumala Kumala","Tuti Handayani","Muhammad Rizki","Reza Dipta Prayitna"],"year":2027,"abstract":"The rapid development of Artificial Intelligence (AI) has introduced deepfake technology, which is capable of manipulating audiovisual data with high realism. While this technology offers creative utility, its misuse for synthesizing non-consensual pornography, political disinformation, and character assassination severely violates an individual’s right to reputation. This study examines the civil and criminal liabilities of AI deepfake platform providers under the Indonesian legal frame","url":"https://doi.org/10.71131/0rqvry57","categories":["social-engineering"],"reviewed":false},{"id":"llmsec-2026-00653","type":"paper","title":"Context Contamination in LLM Analysis of Network Security Logs: Poison with Passive Prompt Injection and Mitigation Evaluation","authors":["Rabimba Karanjai","Yang Lu","Hemanth Hegadehalli Madhavarao","Lei Xu","Weidong Shi"],"year":2026,"abstract":"Large Language Models are increasingly deployed in Security Operations Centers for log analysis tasks including summarization, alert triage, and threat investigation. These systems ingest logs from external-facing services and process network logs as natural language contexts to generate security insights. We demonstrate that this architectural pattern introduces a critical vulnerability: adversaries can embed prompt injection payloads in log-generating fields that persist in storage and are exe","url":"https://arxiv.org/abs/2607.14493","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00654","type":"paper","title":"PVDetector: Detecting Prompt Injection Attacks on Purpose-Specific LLM Agents through Policy-Violation Concept Analysis","authors":["Junhui Wang","Hangtao Zhang","Zhirun Zheng","Li Zeng","Jiejun Xiao","Xi Luo","Lihua Yin","Saiqin Long"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed as purpose-specific agents to handle domain-specific tasks such as customer service and code generation. These agents are expected to comply with not only generic safety guardrails but also purpose-specific restrictions tailored to their designated roles. Such additional restrictions enlarge the attack surface, particularly to prompt injection (PI) attacks. To defend against such attacks, existing detection methods primarily rely on analyzin","url":"https://arxiv.org/abs/2607.12624","categories":["prompt-injection","agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00655","type":"paper","title":"Neuro-Agentic Control: A Deep Learning-based LLM-Powered Agentic AI Framework for Controlling Security Controls","authors":["Saroj Gopali","Bipin Chhetri","Deepika Giri","Sima Siami-Namini","Akbar Siami Namin"],"year":2026,"abstract":"Cyberattacks on operational technology are increasingly causing costly downtime and physical damage, exposing the limitations of traditional rule-based monitoring in industrial IoT environments. While Large Language Models (LLMs) have strong semantic reasoning abilities to assist in decision support, their hallucinatory nature presents unacceptable safety liabilities for closed-loop control. This paper introduces a neuro-agentic control framework, a novel architecture that couples an LLM-based p","url":"https://arxiv.org/abs/2607.09076","categories":["agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00656","type":"paper","title":"Bad Memory: Evaluating Prompt Injection Risks from Memory in Agentic Systems","authors":["Soham Gadgil","David Alexander","Sai Sunku","Franziska Roesner"],"year":2026,"abstract":"A growing class of agentic systems maintain persistent state across sessions through memory files, behavioral preferences, and knowledge bases. While this makes agents more useful and self-improving, it also creates a new attack surface for prompt injections in which malicious instructions can be embedded within persistent files and influence future behavior. In this work, we study prompt injection attacks in memory-based agentic systems using a sandboxed synthetic workspace. We evaluate two age","url":"https://arxiv.org/abs/2607.14611","categories":["prompt-injection","agentic-threats","sandboxing-isolation"],"reviewed":false},{"id":"llmsec-2026-00657","type":"paper","title":"Agent Skill Security: Threat Models, Attacks, Defenses, and Evaluation","authors":["Sanket Badhe","Priyanka Tiwari"],"year":2026,"abstract":"Reusable skills are becoming a fundamental building block of Large Language Model (LLM) agents, enabling capabilities to be packaged, shared, and reused across diverse applications. However, existing security research primarily focuses on prompt injection and runtime execution, leaving security risks throughout the broader skill lifecycle largely unexplored. In this paper, we present SkillSec-Eval, a lifecycle-aware framework for systematically evaluating the security of reusable agent skills. W","url":"https://arxiv.org/abs/2607.13987","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00658","type":"paper","title":"How Agents Ask for Permission: User Permissions for AI Agents, from Interfaces to Enforcement","authors":["Alexandra E. Michael","Franziska Roesner"],"year":2026,"abstract":"As AI agents gain prevalance, users are increasingly exposed to the risks such systems entail. Prompt injection attacks, as well as hallucination, can cause agents to leak private information to third parties. As autonomous systems, agents also present the more active danger of performing sensitive tasks, such as bank transactions, without the user's intent or authorization. Recognizing this challenge, the agentic security community has developed numerous proposals for secure agentic systems. Mu","url":"https://arxiv.org/abs/2607.13718","categories":["prompt-injection","agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00659","type":"paper","title":"Silent Alarm: A J-Space Protocol for Comparing Danger Recognition Across Models and Quantization Levels","authors":["Roman Prosvirnin","Victor Minchenkov","Alexey Soldatov","Vladimir Bashun"],"year":2026,"abstract":"Jailbreak-robustness research typically evaluates safety through generated responses using an LLM-as-judge approach. Such evaluations, however, are sensitive to the benchmark's grading procedure and capture only observed behavior on a given set of attacks, without directly revealing the hidden fragility of the underlying safety mechanisms. This work proposes JADR (Jacobian Assessment of Danger Recognition), a protocol that measures a model's internal representation through Jacobian space (J-spac","url":"https://arxiv.org/abs/2607.12792","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00660","type":"paper","title":"Breaking Refusal in the First Half: A Mechanistic Study of the Prefill Jailbreak","authors":["Alex Kwon"],"year":2026,"abstract":"Aligned language models refuse harmful requests, but a one-line prefill (\"Sure, here is\") strips the refusal. We ask where and how it fails. The harm representation stays intact: on the prompts the attack flips to compliance, a linear probe reads harm as high as on the refused ones (0.91-0.98), while behavioral refusal drops to chance. This holds across four models and three families (1.5-3.8B, and at 14B). Refusal is therefore a shallow, response-site computation. We localize it to an early win","url":"https://arxiv.org/abs/2607.14147","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00661","type":"paper","title":"SingGuard-NSFA: Extensible Guardrails for Agentic AI via Generative Reasoning and Real-Time Classification","authors":[" SingGuard Team"],"year":2026,"abstract":"We present nsfaguard, a guardrail framework for securing agentic AI systems against operational threats, such as prompt injection, sensitive information extraction, malicious code requests, dangerous tool misuse, and resource exhaustion. We first introduce the NSFA taxonomy, which organizes 185 risk variants into a CIA-triad-grounded hierarchy and is cross-validated against three well-established OWASP guidelines. Based on this taxonomy, we construct a benchmark suite spanning 133 languages, com","url":"https://arxiv.org/abs/2607.13081","categories":["prompt-injection","membership-inference","agentic-threats","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00662","type":"paper","title":"Distributed Denial of Science: How Indirect Data Poisoning of AI Systems Can Industrialize Scientific Fraud","authors":["Bálint Gyevnár","Atoosa Kasirzadeh","Nihar B. Shah"],"year":2026,"abstract":"Scientific fraud is the instrument of doubt that malicious entities can use to establish controversy in science. Historically, it required the resources of a company: deep pockets, ghostwritten articles, and corrupt academics. Today, Artificial Intelligence (AI) is increasingly automating scientific research, so we ask: Can a remote adversary weaponize the honest use of AI in science to compromise scientific integrity? We envision and empirically evaluate a new attack, indirect data poisoning, i","url":"https://arxiv.org/abs/2607.10712","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00663","type":"paper","title":"NetInjectBench: Benchmarking Indirect Prompt Injection in Tool-Using Large Language Model Agents for Network Operations","authors":["Ruksat Khan Shayoni","Muhammad Faraz Shoaib","S M Asif Hossain","M. F. Mridha"],"year":2026,"abstract":"Tool-using large language model (LLM) agents are attractive for network operations, but tickets, alerts, logs, runbooks, and ChatOps messages can carry indirect prompt injections. We present NetInjectBench, a 130-scenario benchmark that separates untrusted artifact text, trusted policy metadata, and evaluation labels for network-operation tool use. The sample contains 40 benign, 40 weak-attack, 40 strong-attack, and 10 approved high-impact change scenarios; each is evaluated with Qwen2.5-7B, Lla","url":"https://arxiv.org/abs/2607.10490","categories":["prompt-injection","benchmarks","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00664","type":"paper","title":"Devil in the Lens: Analyzing and Defending Physical Prompt Injection Against Vision-Language Models on Wearable Devices","authors":["Yaxin Li","Hao Wang","Yanda Shao","Shuhao Zhang","Yan Long"],"year":2026,"abstract":"Vision-Language Models (VLMs) are rapidly deployed on human-facing wearable devices such as smart glasses to enable multimodal perception and AI-assisted decision-making. While prior research has demonstrated the risks of visual prompt injection into digital image inputs of VLMs, the unique security challenges posed by the increasing integration between physical environments and wearable intelligence, such as those embodied in VLM-enabled AI glasses, remain underexplored. Toward understanding an","url":"https://arxiv.org/abs/2607.10269","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00665","type":"paper","title":"Minionese: Comprehensive Benchmark and Mechanistic Study of Multilingual LLM Safety","authors":["Chigozirim Ifebi","Brent Kong","Ayushi Mehrotra"],"year":2026,"abstract":"Safety alignment in large language models remains brittle across languages: prompts reliably refused in English can elicit harmful compliance in non-English and low-resource settings. We introduce \\textsc{Minionese}, a multilingual jailbreak benchmark spanning 18 languages, 4 resource tiers, and 4 perturbation types (standard translation, code-switching, transliteration, and translationese), paired with a geometric mechanistic analysis of refusal failure across language tiers. We show that each ","url":"https://arxiv.org/abs/2607.10112","categories":["jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00666","type":"paper","title":"When Does Belief-Based Agent Memory Help? Reliability-Conditional Updating and Provenance-Capped Poisoning Defense","authors":["Pranav Singh"],"year":2026,"abstract":"We investigate when belief-based memory actually improves large language model (LLM) agents. Our vehicle is Nous, a long-term memory architecture that represents each entity-attribute pair as a categorical probability distribution updated through closed-form Bayesian inference, with information-theoretic surprise driving belief revision and entropy-based forgetting. A controlled ablation on the LoCoMo benchmark shows that Bayesian belief updating alone provides little benefit over naive last-wri","url":"https://arxiv.org/abs/2606.22030","categories":["data-poisoning","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00667","type":"paper","title":"Democratizing Agent Deployment Safety: A Structural Monitoring Approach","authors":["Preeti Ravindra","Rahul Tiwari","Vincent Wolowski"],"year":2026,"abstract":"AI software development agents are increasingly capable of modifying infrastructure and security critical systems, creating risks where an agent completes its assigned task while covertly weakening safeguards through actions such as broadening permissions, degrading logging, or introducing persistence mechanisms. While frontier laboratories may deploy sophisticated monitoring pipelines, many organizations and individual users adopting coding agents lack the resources and governance maturity requ","url":"https://arxiv.org/abs/2607.14570","categories":["guardrails","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00668","type":"paper","title":"Agentic Service-Oriented Computing: A Manifesto for the Next Frontier of Service-Oriented Computing","authors":["Amin Beheshti","Rong N. Chang","Boualem Benatallah","Fabio Casati","Schahram Dustdar","Geoffrey Fox","Quan Z. Sheng","Yan Wang","Jian Yang","Albert Zomaya"],"year":2026,"abstract":"The rapid emergence of LLM-powered autonomous and semi-autonomous agents is reshaping software systems from static, request-response components into goal-directed, adaptive, and tool-using computational actors. As these agents move from isolated cognitive prototypes into complex distributed workflows, they confront challenges that the Service-Oriented Computing community has studied for more than two decades: composition, interoperability, quality of service, lifecycle management, governance, se","url":"https://arxiv.org/abs/2607.12619","categories":["agentic-threats","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00669","type":"paper","title":"Mako: A Self-Evolving Agentic Operating System (SE-AOS) for Autonomous Web Exploitation","authors":["Praneeth Narisetty","Shiva Nagendra Babu Kore"],"year":2026,"abstract":"We introduce the Self-Evolving Agentic Operating System (SE-AOS): a new class of AI agent that treats exploit capability as a mutable, versioned kernel it extends at runtime, observing its own failures, synthesising new capabilities, proving them against a live target, and hot-loading them back into itself. Mako is the first SE-AOS instance for security research and the autonomous web exploitation engine developed within LaunchSafe. LaunchSafe builds autonomous security agents for continuous off","url":"https://arxiv.org/abs/2607.11288","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00670","type":"paper","title":"VEXAIoT: Autonomous IoT Vulnerability EXploitation using AI Agents","authors":["Katherine Swinea","Kshitiz Aryal","Lopamudra Praharaj","Maanak Gupta"],"year":2026,"abstract":"Internet of Things (IoT) systems are inherently vulnerable due to constrained hardware, outdated firmware, and insecure default configurations, creating a need for scalable and adaptive security testing approaches. While recent adoptions of Large Language Model (LLM) agents have demonstrated promise in penetration testing and Capture-the-Flag (CTF) environments, their application to IoT specific vulnerabilities remains unexplored. This paper presents an autonomous multi-agent framework, referred","url":"https://arxiv.org/abs/2607.09653","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00671","type":"paper","title":"LLM-Centric Agentic AI for UAV Swarms: Architecture, Enabling Technologies, and Open Problems","authors":["Yousef Emami","Rahim Taheri","Mohammadhossein Homaei","Muhammad Atif Ur Rehman","Mohammad Shojafar"],"year":2026,"abstract":"Uncrewed Aerial Vehicle (UAV) swarms have significant potential for applications such as Search and Rescue (SAR) and environmental monitoring, but their real-world deployment is limited by a lack of situational awareness, intermittent connectivity, and significant cybersecurity risks. Agentic Artificial Intelligence (AI) represents a shift from standalone Large Language Model (LLM) toward closed-loop cognitive architectures that integrate perception, memory, reasoning/planning, and action to ena","url":"https://arxiv.org/abs/2607.09756","categories":["agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00672","type":"paper","title":"Identity Management for Agentic AI: The new frontier of authorization, authentication, and security for an AI agent world","authors":["Tobin South","Subramanya Nagabhushanaradhya","A. Dissanayaka","Sarah Cecchetti","George H. L. Fletcher","Victor Lu","A. Pietropaolo","Dean H. Saxe","J. Lombardo","Abhishek Maligehalli Shivalingaiah","Stan Bounev","Alex Keisner","Andor Kesselman","Zack Proser","Ginny Fahs","Andrew Bunyea","Ben Moskowitz","Atul Tulshibagwale","Dazza Greenwood","Jiaxin Pei","A. Pentland"],"year":2025,"venue":"arXiv.org","abstract":"The rapid rise of AI agents presents urgent challenges in authentication, authorization, and identity management. Current agent-centric protocols (like MCP) highlight the demand for clarified best practices in authentication and authorization. Looking ahead, ambitions for highly autonomous agents raise complex long-term questions regarding scalable access control, agent-centric identities, AI workload differentiation, and delegated authority. This OpenID Foundation whitepaper is for stakeholders","url":"https://www.semanticscholar.org/paper/c482e9aa341ecd5dbf545aa16df9c6815aa55650","categories":["agentic-threats","access-control","autonomous-operations"],"citation_count":18,"reviewed":false},{"id":"llmsec-2026-00673","type":"paper","title":"TRiSM for Agentic AI: A Review of Trust, Risk, and Security Management in LLM-based Agentic Multi-Agent Systems","authors":["Shaina Raza","Ranjan Sapkota","Manoj Karkee","Christos Emmanouilidis"],"year":2025,"venue":"AI Open","abstract":"Agentic AI systems, built upon large language models (LLMs) and deployed in multi-agent configurations, are redefining intelligence, autonomy, collaboration, and decision-making across enterprise and societal domains. This review presents a structured analysis of Trust, Risk, and Security Management (TRiSM) in the context of LLM-based Agentic Multi-Agent Systems (AMAS). We begin by examining the conceptual foundations of Agentic AI and highlight its architectural distinctions from traditional AI","url":"https://www.semanticscholar.org/paper/753736d18fa9bf3ed730836b30b89bf5653cd8dd","categories":["agentic-threats","agent-architecture"],"citation_count":82,"reviewed":false},{"id":"llmsec-2026-00674","type":"paper","title":"SAGA: A Security Architecture for Governing AI Agentic Systems","authors":["Georgios Syros","Anshuman Suri","Cristina Nita-Rotaru","Alina Oprea"],"year":2025,"venue":"Network and Distributed System Security Symposium","abstract":"Large Language Model (LLM)-based agents increasingly interact, collaborate, and delegate tasks to one another autonomously with minimal human interaction. Industry guidelines for agentic system governance emphasize the need for users to maintain comprehensive control over their agents, mitigating potential damage from malicious agents. Several proposed agentic system designs address agent identity, authorization, and delegation, but remain purely theoretical, without concrete implementation and ","url":"https://www.semanticscholar.org/paper/643e64fd77cb9a2b2defd63769af75a92365a6c7","categories":["agentic-threats","access-control"],"citation_count":42,"reviewed":false},{"id":"llmsec-2026-00675","type":"paper","title":"Formalizing the Safety, Security, and Functional Properties of Agentic AI Systems","authors":["Edoardo Allegrini","Ananth Shreekumar","Z. B. Celik"],"year":2025,"venue":"arXiv.org","abstract":"Agentic AI systems, which leverage multiple autonomous agents and large language models (LLMs), are increasingly used to address complex, multi-step tasks. The safety, security, and functionality of these systems are critical, especially in high-stakes applications. However, the current ecosystem of inter-agent communication is fragmented, with protocols such as the Model Context Protocol (MCP) for tool access and the Agent-to-Agent (A2A) protocol for coordination being analyzed in isolation. Th","url":"https://www.semanticscholar.org/paper/fa748fc17c5e7506bf3fed331e81b2011e5039d0","categories":["agentic-threats","autonomous-operations"],"citation_count":7,"reviewed":false},{"id":"llmsec-2026-00676","type":"paper","title":"Agentic AI for 6G: A New Paradigm for Autonomous RAN Security Compliance","authors":["Sotiris Chatzimiltis","Mahdi Boloursaz Mashhadi","Mohammad Shojafar","M. Debbah","Rahim Tafazolli"],"year":2025,"venue":"IEEE Communications Standards Magazine","abstract":"Agentic AI systems are emerging as powerful tools for automating complex, multi-step tasks across various industries. One such industry is telecommunications, where the growing complexity of next-generation radio access networks (RANs) opens up numerous opportunities for applying these systems. Securing the RAN is a key area, particularly through automating the security compliance process, as traditional methods often struggle to keep pace with evolving specifications and real-time changes. In t","url":"https://www.semanticscholar.org/paper/3db949389086da9db5ec5e9fc88c1de41ee99df2","categories":["agentic-threats"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00677","type":"paper","title":"ASTRIDE: A Security Threat Modeling Platform for Agentic-AI Applications","authors":["Eranga Bandara","Amin Hass","Ross Gore","Sachin Shetty","R. Mukkamala","S. Bouk","Xueping Liang","Ng Wee Keong","K. D. Zoysa","A. Withanage","Nilaan Loganathan"],"year":2025,"venue":"International Conference on Wireless Communications and Mobile Computing","abstract":"AI agent-based systems are becoming increasingly integral to modern software architectures, enabling autonomous decision-making, dynamic task execution, and multimodal interactions through large language models (LLMs). However, these systems introduce novel and evolving security challenges, including prompt injection attacks, context poisoning, model manipulation, and opaque agent-to-agent communication that are not effectively captured by traditional threat modeling frameworks. In this paper, w","url":"https://www.semanticscholar.org/paper/37c56b578604d3dea4d86ae9bf1a76cdbff5129e","categories":["prompt-injection","data-poisoning","agentic-threats","threat-modeling"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00678","type":"paper","title":"Security Analysis of Agentic AI Communication Protocols: A Comparative Evaluation","authors":["Yedidel Louck","Ariel Stulman","Amit Dvir"],"year":2025,"venue":"arXiv.org","abstract":"Multi-agent systems (MAS) powered by artificial intelligence (AI) are increasingly foundational to complex, distributed workflows. Yet, the security of their underlying communication protocols remains critically under-examined. This paper presents the first empirical, comparative security analysis of the official CORAL implementation and a high-fidelity, SDK-based ACP implementation, benchmarked against a literature-based evaluation of A2A. Using a 14 point vulnerability taxonomy, we systematica","url":"https://www.semanticscholar.org/paper/a380e368fc630cdeab758fa86959a7570024a7a3","categories":["agentic-threats","benchmarks","agent-architecture"],"citation_count":7,"reviewed":false},{"id":"llmsec-2026-00679","type":"paper","title":"Agentic AI for Autonomous Defense in Software Supply Chain Security: Beyond Provenance to Vulnerability Mitigation","authors":["Toqeer Ali Syed","M. R. Belgaum","Salman Jan","Asadullah Khan","Saad Said Alqahtani"],"year":2025,"venue":"International Conferences on Computing Advancements","abstract":"Software supply chain attacks increasingly target trusted development and delivery processes, making conventional post-build integrity mechanisms insufficient. Existing frameworks like SLSA, SBOM, and in-toto mainly provide provenance and traceability but cannot actively identify or remove vulnerabilities during production. This paper presents an agentic AI approach for autonomous software supply chain security, combining large language model (LLM) reasoning, reinforcement learning (RL), and mul","url":"https://www.semanticscholar.org/paper/7ef12e7f28b538a044f3b2c2af4160446723a200","categories":["supply-chain-attacks","agentic-threats"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00680","type":"paper","title":"AgentDoG: A Diagnostic Guardrail Framework for AI Agent Safety and Security","authors":["Dongrui Liu","Qihan Ren","Chen Qian","Shuai Shao","Yuejin Xie","Yu Li","Zhonghao Yang","Haoyu Luo","Peng Wang","Qingyu Liu","Bin Hu","Ling Tang","Jilin Mei","Dadi Guo","Lei Yuan","Junyao Yang","Guanxu Chen","Qihao Lin","Yi Yu","Bo Zhang","Jiaxuan Guo","Jie Zhang","Wenqi Shao","Huiqi Deng","Zhiheng Xi","Wenjie Wang","Wenxuan Wang","Wen Shen","Zhikai Chen","Haoyu Xie","Jialing Tao","Juntao Dai","Jiaming Ji","Zhongjie Ba","Linfeng Zhang","Yong Liu","Quanshi Zhang","Lei Zhu","Zhihua Wei","Hui Xue","Chaochao Lu","Jing Shao","Xia Hu"],"year":2026,"abstract":"The rise of AI agents introduces complex safety and security challenges arising from autonomous tool use and environmental interactions. Current guardrail models lack agentic risk awareness and transparency in risk diagnosis. To introduce an agentic guardrail that covers complex and numerous risky behaviors, we first propose a unified three-dimensional taxonomy that orthogonally categorizes agentic risks by their source (where), failure mode (how), and consequence (what). Guided by this structur","url":"https://www.semanticscholar.org/paper/8f66f06304829890f83c428fb4677a54c91d9fbd","categories":["agentic-threats","guardrails","tool-use-security"],"citation_count":20,"reviewed":false},{"id":"llmsec-2026-00681","type":"paper","title":"AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security","authors":["Dongrui Liu","Yu Li","Zhonghao Yang","Peng Wang","Guan-Lin Chen","Yuejin Xie","Qinghua Mao","Wanying Qu","Yanxu Zhu","Tianyi Zhou","Lei Yuan","Zhijie Zheng","Qihao Lin","Yiming Wang","Haoyu Luo","Shuai Shao","Chen Qian","Qingyu Liu","Ling Tang","Ruiyang Qin","Qihan Ren","Junxiao Yang","Kun Wang","Zhiheng Xi","Linfeng Zhang","Ranjie Duan","Bo Zhang","Wenjie Wang","Wen Shen","Qiaosheng Zhang","Y. Teng","Chaochao Lu","Rui Mei","Man Li","Jialing Tao","Xi Lin","Tianhang Zheng","Yong Liu","Quanshi Zhang","Lei Zhu","Xingjun Ma","Junhua Liu","Hui Xue","X.Z. Zuo","Xiangnan He","Chaoyi Shen","Xianglong Liu","Min Huang","Jing Shao","Xia Hu"],"year":2026,"abstract":"Modern open-world agents such as OpenClaw exhibit powerful cross-environment execution capabilities yet introduce broad new safety risk sources. Meanwhile, advanced frontier AI models drastically lower attack barriers, rendering current agent alignment frameworks inadequate for real-world deployment. To tackle these emerging threats, we propose a lightweight and scalable agent safety alignment framework. Specifically, we update the agent safety taxonomy to accommodate emergent risks from Codex a","url":"https://www.semanticscholar.org/paper/27d50468f1016b18c07283b56ea5dc0f5479b980","categories":["guardrails"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00682","type":"paper","title":"A Research Landscape of Agentic AI and Large Language Models: Applications, Challenges and Future Directions","authors":["Domenico Ursino","Gianluca Bonifazi","Enrico Corradini","Michele Marchetti","S. Brohi","Qurat-ul-ain Mastoi","N. Z. Jhanjhi","T. Pillai"],"year":2025,"venue":"Algorithms","abstract":"Agentic AI and Large Language Models (LLMs) are transforming how language is understood and generated while reshaping decision-making, automation, and research practices. LLMs provide underlying reasoning capabilities, and Agentic AI systems use them to perform tasks through interactions with external tools, services, and Application Programming Interfaces (APIs). Based on a structured scoping review and thematic analysis, this study identifies that core challenges of LLMs, relating to security,","url":"https://www.semanticscholar.org/paper/24ae03b3f1793dc2ca5aec06ed4c50ec61a6c7cd","categories":["agentic-threats"],"citation_count":47,"reviewed":false},{"id":"llmsec-2026-00683","type":"paper","title":"Building A Secure Agentic AI Application Leveraging Google’s A2A Protocol","authors":["I. Habler","Ken Huang","Vineeth Sai Narajala","Prashant Kulkarni"],"year":2025,"venue":"2025 Annual Computer Security Applications Conference Workshops (ACSAC Workshops)","abstract":"As Agentic AI systems evolve from basic workflows to complex multi-agent collaboration, robust protocols such as Google’s Agent2Agent (A2A) become essential enablers. To foster secure adoption and ensure the reliability of these complex interactions, understanding the secure implementation of A2A is essential. This paper addresses this goal by providing a comprehensive security analysis centered on the A2A protocol. We examine its fundamental elements and operational dynamics, situating it withi","url":"https://www.semanticscholar.org/paper/8f3fc962d61ac1d298ed511ba27272f1c5991599","categories":["agentic-threats","agent-architecture"],"citation_count":40,"reviewed":false},{"id":"llmsec-2026-00684","type":"paper","title":"A Novel Zero-Trust Identity Framework for Agentic AI: Decentralized Authentication and Fine-Grained Access Control","authors":["Ken Huang","Vineeth Sai Narajala","J. Yeoh","Ramesh Raskar","Youssef Harkati","Jerry Huang","I. Habler","Chris Hughes"],"year":2025,"venue":"2026 International Conference on AI x Data and Knowledge Engineering (AIxDKE)","abstract":"Traditional Identity and Access Management (IAM) systems, primarily designed for human users or static machine identities via protocols such as OAuth, OpenID Connect (OIDC), and SAML, prove fundamentally inadequate for the dynamic, interdependent, and often ephemeral nature of AI agents operating at scale within Multi Agent Systems (MAS) - a computational system composed of multiple interacting intelligent agents that work collectively.This paper posits the imperative for a novel Agentic AI - IA","url":"https://www.semanticscholar.org/paper/2bbf1fc0d7bce53edbfcaeb81143c1233c074923","categories":["agentic-threats","access-control"],"citation_count":39,"reviewed":false},{"id":"llmsec-2026-00685","type":"paper","title":"Trustworthy agentic AI systems: a cross-layer review of architectures, threat models, and governance strategies for real-world deployment","authors":["Ibrahim Adabara","Bashir Olaniyi Sadiq","Aliyu Nuhu Shuaibu","Yale Ibrahim Danjuma","Venkateswarlu Maninti"],"year":2025,"venue":"F1000Research","abstract":"Agentic Artificial Intelligence systems, characterized by autonomous reasoning, memory augmentation, and adaptive planning, are rapidly reshaping technological landscapes. Unlike traditional AI or large language models, agentic AI integrates decision-making with persistent execution, enabling complex interactions across dynamic environments. However, this evolution introduces novel security risks, governance challenges, and ethical considerations that current frameworks inadequately address. Thi","url":"https://www.semanticscholar.org/paper/b4578b31671f32a11e47b0f4870b885f55900153","categories":["agentic-threats","threat-modeling"],"citation_count":32,"reviewed":false},{"id":"llmsec-2026-00686","type":"paper","title":"Agent-Based AI Approach to Security in IoT Systems Leveraging Genai","authors":["N. Petrovic","D. Krstić","S. Suljović","S. Hanczewski","M. Głąbowski"],"year":2025,"venue":"International Conference on Software, Telecommunications and Computer Networks","abstract":"Internet of Things (IoT) devices are becoming an important part of our environment - from home applications to smart city infrastructures. Therefore, secure deployment and operation of IoT-based applications is one of the crucial factors for practical adoption of such services. In this paper, we explore the potential of Generative Artificial Intelligence (GenAI) in synergy with agentic AI approach and Model-Driven Engineering (MDE) in order to tackle both the defense and penetration testing in c","url":"https://www.semanticscholar.org/paper/6f1174d046152d8270662e216a28d82016aed238","categories":["agentic-threats"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-00687","type":"paper","title":"Sola-Visibility-ISPM: Benchmarking Agentic AI for Identity Security Posture Management Visibility","authors":["Gal Engelberg","Konstantin Koutsyi","Leon Goldberg","Reuven Elezra","Idan Pinto","Tal Moalem","Samuel N. Cohen","Yoni Weintrob"],"year":2026,"venue":"arXiv.org","abstract":"Identity Security Posture Management (ISPM) is a core challenge for modern enterprises operating across cloud and SaaS environments. Answering basic ISPM visibility questions, such as understanding identity inventory and configuration hygiene, requires interpreting complex identity data, motivating growing interest in agentic AI systems. Despite this interest, there is currently no standardized way to evaluate how well such systems perform ISPM visibility tasks on real enterprise data. We introd","url":"https://www.semanticscholar.org/paper/ec53bf26de1b0055611062408c1116f35cd308e7","categories":["agentic-threats","benchmarks","cloud-ai-security"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00688","type":"paper","title":"Oversight Structures for Agentic AI in Public-Sector Organizations","authors":["Chris Schmitz","Jonathan Rystrøm","Jan Batzner"],"year":2025,"venue":"Proceedings of the 1st Workshop for Research on Agent Language Models (REALM 2025)","abstract":"This paper finds that the introduction of agentic AI systems intensifies existing challenges to traditional public sector oversight mechanisms -- which rely on siloed compliance units and episodic approvals rather than continuous, integrated supervision. We identify five governance dimensions essential for responsible agent deployment: cross-departmental implementation, comprehensive evaluation, enhanced security protocols, operational visibility, and systematic auditing. We evaluate the capacit","url":"https://www.semanticscholar.org/paper/f28ed25afc2d23d9d62e0766b2956d142bb9913a","categories":["agentic-threats"],"citation_count":11,"reviewed":false},{"id":"llmsec-2026-00689","type":"paper","title":"Formal Analysis and Supply Chain Security for Agentic AI Skills","authors":["Varun Pratap Bhardwaj"],"year":2026,"venue":"arXiv.org","abstract":"The rapid proliferation of agentic AI skill ecosystems -- exemplified by OpenClaw (228,000 GitHub stars) and Anthropic Agent Skills (75,600 stars) -- has introduced a critical supply chain attack surface. The ClawHavoc campaign (January-February 2026) infiltrated over 1,200 malicious skills into the OpenClaw marketplace, while MalTool catalogued 6,487 malicious tools that evade conventional detection. In response, twelve reactive security tools emerged, yet all rely on heuristic methods that pro","url":"https://www.semanticscholar.org/paper/d6665127102362bd5060d673a21fe6b1de3f376c","categories":["supply-chain-attacks","agentic-threats"],"citation_count":14,"reviewed":false},{"id":"llmsec-2026-00690","type":"paper","title":"Security in the Age of AI Teammates: An Empirical Study of Agentic Pull Requests on GitHub","authors":["Mohammed Latif Siddiq","Xinye Zhao","V. C. Lopes","B.K. Casey","Joanna C. S. Santos"],"year":2026,"venue":"arXiv.org","abstract":"Autonomous coding agents are increasingly deployed as AI teammates in modern software engineering, independently authoring pull requests (PRs) that modify production code at scale. This study aims to systematically characterize how autonomous coding agents contribute to software security in practice, how these security-related contributions are reviewed and accepted, and which observable signals are associated with PR rejection. We conduct a large-scale empirical analysis of agent-authored PRs u","url":"https://www.semanticscholar.org/paper/53ed867f0c124539c1521dec5d052779cb8a0f2a","categories":["agentic-threats"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00691","type":"paper","title":"Human Society-Inspired Approaches to Agentic AI Security: The 4C Framework","authors":["A. Abuadbba","N. Sultan","Surya Nepal","Sanjay Jha"],"year":2026,"venue":"arXiv.org","abstract":"AI is moving from domain-specific autonomy in closed, predictable settings to large-language-model-driven agents that plan and act in open, cross-organizational environments. As a result, the cybersecurity risk landscape is changing in fundamental ways. Agentic AI systems can plan, act, collaborate, and persist over time, functioning as participants in complex socio-technical ecosystems rather than as isolated software components. Although recent work has strengthened defenses against model and ","url":"https://www.semanticscholar.org/paper/20f1c57301375e2642fd4969713acdf7c20177b6","categories":["agentic-threats"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00692","type":"paper","title":"AGENTIC AI AND CYBER SECURITY: AUTONOMOUS THREAT HUNTING, INTRUSION DETECTION, AND ADAPTIVE DEFENSE MECHANISMS IN A WORLD OF INCREASINGLY SOPHISTICATED CYBER ATTACKS","authors":["Ajay Simha Rangappa"],"year":2026,"venue":"Journal of Digital Security and Forensics","abstract":"This study investigates the transformative potential of agentic artificial intelligence (AI) systems in enhancing cybersecurity through autonomous threat hunting, real-time intrusion detection, and adaptive defense mechanisms. Employing a mixed-methods research design, the investigation analyzes a large-scale dataset comprising 2.4 million network events collected from January 2023 to December 2024 across 150 enterprise environments. A custom agentic AI framework built on reinforcement learning ","url":"https://www.semanticscholar.org/paper/341d785475da2663203fc3ccaa1bca71968310bb","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00693","type":"paper","title":"Automating Organizational Cyber Security Policy Compliance Against Industry Standards Using Agentic AI","authors":["R. Negi","S. V. Chakraborty","Amit Negi","S. K. Shukla"],"year":2026,"venue":"International Symposium on Digital Forensics and Security","abstract":"Auditing and compliance management are an integral part of a cybersecurity management system (CSMS). However, the frequency of audits and compliance checks is typically once a year for external audits and twice a year for internal audits. Under audit and compliance management, cybersecurity policy documents defined according to normative references are also reviewed. This review is either performed at the semantic level, which may miss contextual reasoning, or manually, which requires significan","url":"https://www.semanticscholar.org/paper/b675c553acbc44d29e06bb0c8155978a0cbf73c7","categories":["agentic-threats","audit-assurance"],"reviewed":false},{"id":"llmsec-2026-00694","type":"paper","title":"Sentinel Agents for Secure and Trustworthy Agentic AI in Multi-Agent Systems","authors":["Diego Gosmar","Deborah A. Dahl"],"year":2025,"venue":"arXiv.org","abstract":"This paper proposes a novel architectural framework aimed at enhancing security and reliability in multi-agent systems (MAS). A central component of this framework is a network of Sentinel Agents, functioning as a distributed security layer that integrates techniques such as semantic analysis via large language models (LLMs), behavioral analytics, retrieval-augmented verification, and cross-agent anomaly detection. Such agents can potentially oversee inter-agent communications, identify potentia","url":"https://www.semanticscholar.org/paper/296c6954e9fe7e5b48fe2dfadc6a23b4310b1bab","categories":["agentic-threats","monitoring-detection","agent-architecture"],"citation_count":10,"reviewed":false},{"id":"llmsec-2026-00695","type":"paper","title":"Securing Agentic AI: Threat Modeling and Risk Analysis for Network Monitoring Agentic AI System","authors":["P. Zambare","Venkata Nikhil Thanikella","Ying Liu"],"year":2025,"venue":"arXiv.org","abstract":"When combining Large Language Models (LLMs) with autonomous agents, used in network monitoring and decision-making systems, this will create serious security issues. In this research, the MAESTRO framework consisting of the seven layers threat modeling architecture in the system was used to expose, evaluate, and eliminate vulnerabilities of agentic AI. The prototype agent system was constructed and implemented, using Python, LangChain, and telemetry in WebSockets, and deployed with inference, me","url":"https://www.semanticscholar.org/paper/f42a56937c1a50380c6143ba835efa5b97c5c31e","categories":["agentic-threats","monitoring-detection","autonomous-operations","threat-modeling"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00696","type":"paper","title":"Security Risks of Agentic Vehicles: A Systematic Analysis of Cognitive and Cross-Layer Threats","authors":["Ali Eslami","Jiangbo Yu"],"year":2025,"venue":"arXiv.org","abstract":"Agentic AI is increasingly being explored and introduced in both manually driven and autonomous vehicles, leading to the notion of Agentic Vehicles (AgVs), with capabilities such as memory-based personalization, goal interpretation, strategic reasoning, and tool-mediated assistance. While frameworks such as the OWASP Agentic AI Security Risks highlight vulnerabilities in reasoning-driven AI systems, they are not designed for safety-critical cyber-physical platforms such as vehicles, nor do they ","url":"https://www.semanticscholar.org/paper/af6ea3b929b3a1aba976f3d48d2c6d040968e7e0","categories":["agentic-threats","threat-modeling"],"citation_count":8,"reviewed":false},{"id":"llmsec-2026-00697","type":"paper","title":"QSAF: A Novel Mitigation Framework for Cognitive Degradation in Agentic AI","authors":["Hammad Atta","M. Baig","Yasir Mehmood","Nadeem Shahzad","Ken Huang","M. A. U. Haq","Muhammad Awais","Kamal Ahmed"],"year":2025,"venue":"arXiv.org","abstract":"We introduce Cognitive Degradation as a novel vulnerability class in agentic AI systems. Unlike traditional adversarial external threats such as prompt injection, these failures originate internally, arising from memory starvation, planner recursion, context flooding, and output suppression. These systemic weaknesses lead to silent agent drift, logic collapse, and persistent hallucinations over time. To address this class of failures, we introduce the Qorvex Security AI Framework for Behavioral&","url":"https://www.semanticscholar.org/paper/c3da993174d74465fbb632f2d3de5c31d8b58a26","categories":["prompt-injection","agentic-threats"],"citation_count":7,"reviewed":false},{"id":"llmsec-2026-00698","type":"paper","title":"Agent Security is a Systems Problem","authors":["Mihai Christodorescu","Earlence Fernandes","Ashish Hooda","Somesh Jha","J. Rehberger","Kamalika Chaudhuri","Xiaohan Fu","Khawaja Shams","Guy Amir","Jihye Choi","Sarthak Choudhary","Nils Palumbo","Andrey Labunets","Nishit V. Pandya"],"year":2026,"venue":"arXiv.org","abstract":"We take the position that agent security must be approached as a systems problem: the AI model powering the agent must be treated as an untrusted component, and security invariants must be enforced at the system level. Through this lens, efforts to increase model robustness (the dominant viewpoint in the community) are insufficient on their own. Instead, we must complement existing efforts with techniques from the systems security domain. Based on our experience as cybersecurity researchers in o","url":"https://www.semanticscholar.org/paper/44d178dd44370773c5d0536025c8c848febcee9b","categories":["agentic-threats"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00699","type":"paper","title":"In-Browser LLM-Guided Fuzzing for Real-Time Prompt Injection Testing in Agentic AI Browsers","authors":["Avihay Cohen"],"year":2025,"venue":"arXiv.org","abstract":"Large Language Model (LLM) based agents integrated into web browsers (often called agentic AI browsers) offer powerful automation of web tasks. However, they are vulnerable to indirect prompt injection attacks, where malicious instructions hidden in a webpage deceive the agent into unwanted actions. These attacks can bypass traditional web security boundaries, as the AI agent operates with the user privileges across sites. In this paper, we present a novel fuzzing framework that runs entirely in","url":"https://www.semanticscholar.org/paper/8b9011b256c4819d47721efc3f9522a721b92735","categories":["prompt-injection","agentic-threats"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00700","type":"paper","title":"MobiLLM: An Agentic AI Framework for Closed-Loop Threat Mitigation in 6G Open RANs","authors":["Prakhar Sharma","Haohuang Wen","V. Yegneswaran","Ashish Gehani","Phillip Porras","Zhiqiang Lin"],"year":2025,"venue":"IEEE Military Communications Conference","abstract":"The evolution toward 6G networks is being accelerated by the Open Radio Access Network (O-RAN) paradigm—an open, interoperable architecture that enables intelligent, modular applications across public telecom and private enterprise domains. While this openness creates unprecedented opportunities for innovation, it also expands the attack surface, demanding resilient, low-cost, and autonomous security solutions. Legacy defenses remain largely reactive, labor-intensive, and inadequate for the scal","url":"https://www.semanticscholar.org/paper/3fb886ee2cfed48af9dd1766725f0385754e5528","categories":["agentic-threats"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00701","type":"paper","title":"The Evolution of Agentic AI in Cybersecurity: From Single LLM Reasoners to Multi-Agent Systems and Autonomous Pipelines","authors":["Vaishali Vinay"],"year":2025,"venue":"International Conference on Applied Informatics and Communication","abstract":"Cybersecurity operations are increasingly adopting agentic AI solutions due to the time-critical and complex decision-making in security operations centers (SOCs). While large language models (LLMs) are good with summarization tasks or interpreting structured and unstructured reports, real-world SOC workflows have additional requirements such as access to original logs, reproducibility and accountability to triage security incidents. For example, analysts routinely correlate alerts to understand","url":"https://www.semanticscholar.org/paper/6417c2e047247e4e844c8ab390fb3812fb4f80e1","categories":["agentic-threats","agent-architecture"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00702","type":"paper","title":"Securing Generative AI Agentic Workflows: Risks, Mitigation, and a Proposed Firewall Architecture","authors":["Sunil Kumar Jang Bahadur","Gopala Dhar"],"year":2025,"venue":"arXiv.org","abstract":"Generative Artificial Intelligence (GenAI) presents significant advancements but also introduces novel security challenges, particularly within agentic workflows where AI agents operate autonomously. These risks escalate in multi-agent systems due to increased interaction complexity. This paper outlines critical security vulnerabilities inherent in GenAI agentic workflows, including data privacy breaches, model manipulation, and issues related to agent autonomy and system integration. It discuss","url":"https://www.semanticscholar.org/paper/783db6d7e129f598539e38b4608dfec1702c0eb1","categories":["agentic-threats","agent-architecture"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00703","type":"paper","title":"AgentCyTE: Leveraging Agentic AI to Generate Cybersecurity Training & Experimentation Scenarios","authors":["A. Rodríguez","Jaime C. Acosta","Anantaa Kotal","Aritran Piplai"],"year":2025,"venue":"2025 Annual Computer Security Applications Conference Workshops (ACSAC Workshops)","abstract":"Designing realistic and adaptive networked threat scenarios remains a core challenge in cybersecurity research and training, still requiring substantial manual effort. While large language models (LLMs) show promise for automated synthesis, unconstrained generation often yields configurations that fail validation or execution. We present Agent-CyTE, a framework integrating LLM-based reasoning with deterministic, schema-constrained network emulation to generate and refine executable threat enviro","url":"https://www.semanticscholar.org/paper/b89b1dc9130393a3c3d31d59f31897323f73d7eb","categories":["agentic-threats"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00704","type":"paper","title":"Cross-Agent Campaign Attribution: Linking Asynchronous Attacks Across LLM Agents","authors":["SangJin Park","Myungsub Choi","Jineok Kim","Minseung Kang"],"year":2026,"abstract":"LLM-agent defenses are typically evaluated one session at a time. In deployment, however, attacks can be distributed across independent agents, teams, and runtimes, leaving each local guardrail with only a sparse fragment. We formalize cross-agent asynchronous campaign attribution: linking sessions from the same latent adversarial campaign without shared runtime state, test-time campaign labels, or attacker identity oracles. We introduce Asynchronous Attribution Fingerprint Vectors ($A^2FV$), a ","url":"https://arxiv.org/abs/2607.18826","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00705","type":"paper","title":"Towards an Automated Test of LLM Security Knowledge","authors":["Shufan Chai","Liangliang Sun","Jessica Staddon"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used for a range of software, hardware and human-centered security tasks. Consequently, LLM performance on security tasks is an active area of measurement and research, often with a focus on identifying areas in which LLM security ``knowledge'' may be insufficient. Popular strategies for identifying LLM security knowledge gaps include building corpora of challenge questions or task benchmarks, strategies that require substantial manual work and secur","url":"https://arxiv.org/abs/2607.18496","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00706","type":"paper","title":"Trusted Credentials, Untrusted Behavior: Benchmarking LLM-Agent Security in High-Performance Computing","authors":["Jie Li"],"year":2026,"abstract":"Large language model (LLM) agents are starting to take on routine work in high-performance computing (HPC), including monitoring Slurm jobs, diagnosing failed builds, inspecting simulation output, and coordinating scientific workflows. To do this work, an agent commonly acts under its user's credentials and inherits the user's access to files and the scheduler. This arrangement creates a failure mode that ordinary account-level controls do not capture. Adversarial instructions in a log, tool des","url":"https://arxiv.org/abs/2607.18485","categories":["agentic-threats","monitoring-detection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00707","type":"paper","title":"Adaptive Adversaries: A Multi-Turn, Multi-LLM Benchmark for LLM Agent Security","authors":["Devina Jain","David Hartmann","Chuan Li"],"year":2026,"abstract":"LLM-based agents process external content, exposing them to prompt injection and multi-turn manipulation. Most safety benchmarks evaluate defenders against fixed attack pools collected before evaluation, single-turn or multi-turn. We present a 21-scenario benchmark for \\emph{adaptive multi-round attacks against memoryless LLM defenders}: an autonomous LLM attacker observes prior defender responses and pivots across rounds, while each defender response is evaluated as a fresh interaction. Holding","url":"https://arxiv.org/abs/2607.18063","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00708","type":"paper","title":"JailMeter: An Evidence-Based Evaluation Framework for Jailbreak Attacks on Large Language Models","authors":["Qingjia Huang","Jingyu Zhang","Jianguo Wu","Yakai Li","Weijuan Zhang","Yankai Rong","Junyi Yao","Shengzhi Zhang","Xiaoqi Jia"],"year":2026,"abstract":"The assessment of jailbreak attacks against large language models currently suffers from inconsistent evaluation criteria and methods, leading to unreliable estimates of attack success rates. We propose JailMeter, an evidence-based evaluation framework designed to more faithfully measure jailbreak effectiveness. Inspired by the Information Bottleneck theory, JailMeter applies dual-feedback optimization to filter jailbreak noise from model responses while preserving content relevant to the origin","url":"https://arxiv.org/abs/2607.19424","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00709","type":"paper","title":"Geometric Configurations of Perturbed Jailbreak Prompts","authors":["Lynn Delcon","Andres Algaba","Vincent Ginis"],"year":2026,"abstract":"Perturbation techniques that turn unsuccessful jailbreak prompts into successful ones are continuously evolving, constituting a major security threat to LLM safety. In this paper, we investigate the internal representations of such string-level perturbed jailbreak inputs in the small weight models of the Qwen-2.5-1.5B/-3B/-7B-Instruct and Llama-3.2-1B/-3B/-3.1-8B-Instruct families. We select two representation spaces: the last-layer-last-token embedding space and the top-50 next-token probabilit","url":"https://arxiv.org/abs/2607.20581","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00710","type":"paper","title":"Know Your Agent: Reconnaissance-Driven Pentesting of AI Agents","authors":["Or Zion Eliav","Eyal Lenga","Shir Bernstien","Yisroel Mirsky"],"year":2026,"abstract":"Traditional pentesting uses reconnaissance at each step to uncover unseen weaknesses, build stronger attacks, and advance the objective; we argue that AI agents require the same treatment. We formalize agent reconnaissance by modeling the process and identifying the knowledge assets it seeks to extract: what they are, how they are used, and which agent weaknesses they exploit to give adversaries leverage in indirect prompt injection attacks. We instantiate these insights in Know Your Agent (KYA)","url":"https://arxiv.org/abs/2607.19837","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00711","type":"paper","title":"DARWIN: Evolving Jailbreak Adversary and Guardrail for LLM Safety Evaluation and Protection","authors":["Weiwei Qi","Zefeng Wu","Zhilin Guo","Tianhang Zheng","Chaochao Lu","Liang He","Zhan Qin","Kui Ren"],"year":2026,"abstract":"Most existing LLM safety evaluation and defense methods follow a static formulation: jailbreak vulnerabilities are evaluated with fixed attack methods, and guardrails are trained on fixed malicious prompt datasets. However, real-world adversaries continuously evolve their capabilities and expand the attack space. To address this challenge, we propose DARWIN, an evolutionary attack-defense framework that formulates jailbreaking as an open-ended evolution process and continuously updates guardrail","url":"https://arxiv.org/abs/2607.19829","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00712","type":"paper","title":"Twin Agent: Context Residual Compression for Privilege Separated Agents","authors":["Zhanhao Hu","Dennis Jacob","Xiao Huang","Zhaorun Chen","Bo Li","David Wagner"],"year":2026,"abstract":"Large language model (LLM) agents are vulnerable to security risks, such as prompt injection attacks from untrusted context that manipulate downstream reasoning and tool use. Existing secure-by-design approaches mitigate this risk by separating untrusted observations from privileged execution and careful control of information flow, but often degrade utility and require extensive task-specific engineering. We thus propose Twin Agent, a general privilege separation design pattern inspired by resi","url":"https://arxiv.org/abs/2607.19595","categories":["prompt-injection","tool-use-security","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00713","type":"paper","title":"Data Leakage Prevention in Agentic Applications via Preemptive Hardening","authors":["Akansha Shukla","Emily Bellov","Parth Atulbhai Gandhi","Yuval Elovici","Asaf Shabtai"],"year":2026,"abstract":"Agentic systems integrate LLM driven planning with interfaces to external tools, making data leakage and tool misuse feasible via instruction/data boundary failures and prompt injection attacks. Enforcing required controls consistently is particularly challenging in workflows spanning many codebases and heterogeneous agents. To address this challenge in multi agentic systems, we present a pre-deployment pipeline for scanning, hardening, and validation of agentic applications. The pipeline analyz","url":"https://arxiv.org/abs/2607.18847","categories":["prompt-injection","membership-inference","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00714","type":"paper","title":"CPInj: Uncovering Prompt Injection Risks in Textual Collaborative Prompt Optimization","authors":["Xinting Liao","Behnoosh Zamanlooy","Masoumeh Shafieinejad","David B. Emerson","Ruinan Jin","Deval Pandya","Xiaoxiao Li"],"year":2026,"abstract":"Textual Collaborative Prompt Optimization (TCPO) extends Textgrad (Yuksekgonul et al., 2025) to a decentralized setting by allowing multiple clients to jointly improve prompts for large language models (LLMs) while keeping their data locally. Its reliance on free-form textual updating and aggregation introduces a new and largely unexplored attack surface, i.e., malicious instructions can be injected into local prompts and propagated through server-side prompt aggregation. Unlike conventional pro","url":"https://arxiv.org/abs/2607.18622","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00715","type":"paper","title":"ChainWatch: A Kill Chain-Aligned Sequential Detection Framework for Multi-Step Attacks in MCP-Based AI Agent Systems","authors":["Om Narayan","Rashmi Jyoti","Ramkinker Singh"],"year":2026,"abstract":"The Model Context Protocol (MCP) is an open-source standard that allows AI agents to connect to external tools, databases, and services. While this connectivity enables powerful agent capabilities, it also introduces multi-step attacks that existing per-call defenses cannot reliably detect. Attackers can compose individually benign tool invocations into malicious sequences that evade isolated inspection. This paper presents ChainWatch, a sequential detection framework for identifying multi-step ","url":"https://arxiv.org/abs/2607.19432","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00716","type":"paper","title":"ChannelGuard: Safe Models Do Not Compose into Safe Multi-Agent Systems","authors":["Elias Hossain","Md Mehedi Hasan Nipu","Fatema Tuj Johora Faria","Tasfia Nuzhat Ornee","Maleeha Sheikh"],"year":2026,"abstract":"Multi-agent LLM applications chain a planner, worker agents, a verifier, and a synthesizer, and every hop between agents is an unmonitored channel through which an adversary can smuggle instructions. Existing defenses guard only the input boundary (IBProtector, Llama Guard, perplexity filters, SmoothLLM) or run outside the application as opaque, stochastic provider-side filters. We show this gap carries a consequence rarely measured: on a 2,100-trace evaluation across eight attack families, five","url":"https://arxiv.org/abs/2607.19430","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00717","type":"paper","title":"Salience Induction against Multi-Hop RAG Agents: Threat and Defense","authors":["Xingfu Zhou","Pengfei Wang","Yuan Zhou","Wei Xie","Xu Zhou"],"year":2026,"abstract":"Agentic retrieval-augmented generation (RAG) systems increasingly retrieve external evidence and orchestrate tools for knowledge-intensive applications. In Multi-Hop question answering, agents chain facts across documents. Existing defenses focus on content poisoning, which injects false facts, and prompt injection, which embeds directives. We identify a third attack surface: the salience channel, through which fact position, emphasis, framing, and semantic proximity can redirect reasoning even ","url":"https://arxiv.org/abs/2607.17535","categories":["prompt-injection","data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00718","type":"paper","title":"Between Safe Boundaries: Exploiting Temporal Consistency for Jailbreaking Text-To-Video Generation Models","authors":["Xingkai Peng","Jun Jiang","Jiayang Liu","Kejiang Chen","Weiming Zhang"],"year":2026,"abstract":"Recently, text-to-video (T2V) models have been widely deployed, sparking growing concerns over their robustness against jailbreak attacks. Existing jailbreak methods, mostly adapted from text-to-image attacks, suffer notable drawbacks when applied to T2V systems. They fail to fully leverage temporal consistency, an inherent characteristic of video generation. Besides, these methods demand heavy video query optimization, which is infeasible in practical black-box scenarios. Their adversarial prom","url":"https://arxiv.org/abs/2607.17279","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00719","type":"paper","title":"How Jailbreak Attacks Inform Safety Alignment: A Defender-Centric, Shapley-Based Evaluation of Jailbreak Contributions","authors":["Yukai Zhou","Feiyang Lu","Xiaokai Mao","Jinfei Liu","Wenjie Wang"],"year":2026,"abstract":"Jailbreak attacks on large language models are usually evaluated by attacker-centric metrics such as attack success rate (ASR), yet an attack that breaks a model is not necessarily useful for improving its safety. We propose a defender-centric view of jailbreak evaluation, where attacks are evaluated by the downstream safety improvements they enable when used as red-teaming data for safety training. Building on this view, we introduce A-MESS (Minimal Effective Attack-Subset Selection), a setting","url":"https://arxiv.org/abs/2607.17152","categories":["jailbreaking","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00720","type":"paper","title":"Refusal is Not Safety! Benchmarking Latent Safety Risks of LLM-Driven Content Humorization","authors":["Yu Cui","Ruiqing Yue","Tingyu Li","Sicheng Pan","Zhuoyu Sun","Xufeng Zhang","Baohan Huang","Haibin Zhang","Cong Zuo"],"year":2026,"abstract":"Safety defenses for large language models (LLMs) have been extensively studied, with existing approaches focusing on attack detection and refusal mechanisms. Such fixed-form direct refusal strategies may introduce the risk of prefix injection attacks. Recent work has explored a new direction that leverages humor as an indirect refusal mechanism to mitigate over-refusal in jailbreak scenarios and reduce prefix injection risks. However, this approach implicitly assumes that humorous responses are ","url":"https://arxiv.org/abs/2607.15977","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00721","type":"paper","title":"Trustworthy AI LLM Scalability Risk Index (LSRI): A Cybersecurity Framework Assessing Agentic-AI Security & Software Model Supply Chain Safety Boosting AI-Generated Malware Defense & Explainability Mitigating Emerging Risks of Generative AI","authors":["Kiarash Ahi","Vaibhav Agrawal","Saeed Valizadeh"],"year":2026,"abstract":"As AI shifts from human-in-the-loop interfaces to autonomous multi-agent systems capable of real-time code execution and tool integration through protocols like the Model Context Protocol (MCP), traditional SAST, DAST, and legacy AI safety methods fail to detect modern agentic-AI threats. This paper introduces the LLM Scalability Risk Index (LSRI), a parametric framework and cybersecurity standard for stress-testing autonomous orchestration pipelines. LSRI measures the operational thresholds whe","url":"https://arxiv.org/abs/2602.19021","categories":["supply-chain-attacks","agentic-threats","responsible-ai","agent-architecture","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00722","type":"paper","title":"GPE: Evaluating Robust Evidence Aggregation for Fact Verification under Controllable GEO-Style Poisoning","authors":["Zhaoqi Wang","Zijian Zhang","Xiaomei Yuan","Pengtao Kou","Jiamou Liu","Zhen Li","Liehuang Zhu"],"year":2026,"abstract":"Large language models increasingly use search tools to retrieve up-to-date information, introducing a new attack surface in which retrieved documents can be manipulated. This risk is amplified by the development of generative engine optimization, which can make selected content more likely to be retrieved, cited, and adopted by models. Existing fact-verification benchmarks and evaluation frameworks do not provide the controlled evidence environments needed to assess robustness against GEO poison","url":"https://arxiv.org/abs/2607.20730","categories":["data-poisoning","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00723","type":"paper","title":"Toward cryptographically verifiable authorization for autonomous AI agents: A security hypothesis, preliminary formal model, and proof-of-concept implementation","authors":["M. Llambí-Morillas","D. Fernández-Fernández"],"year":2026,"abstract":"Autonomous AI agents increasingly execute actions, invoke tools, and operate on protected resources with limited human oversight. Existing authentication and authorization mechanisms establish identity and delegate authority, but do not inherently provide cryptographic evidence that a concrete request issued by a specific agent satisfies the applicable policy in a specific execution context. This paper hypothesizes that agent authorization can be formalized as a cryptographically verifiable rela","url":"https://arxiv.org/abs/2607.21325","categories":["access-control","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00724","type":"paper","title":"The Ethics of Autonomous AI Agents for Offensive Security","authors":["Andreas Happe","Jürgen Cito","Jasmin Wachter"],"year":2026,"abstract":"LLM-driven autonomous agents are reshaping offensive security. Unlike traditional penetration-testing tooling -- deterministic, narrowly scoped, and operated by trained practitioners -- agentic security tools exhibit \\textit{indeterminacy} along three independent dimensions. First, their actions are drawn from a non-deterministic policy whose outputs resist both ex-ante and ex-post explanation, frustrating incident attribution and pre-deployment safety review. Second, their impact is open-ended ","url":"https://arxiv.org/abs/2607.20255","categories":["agentic-threats","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00725","type":"paper","title":"ResearchArena: Evaluating Sabotage and Monitoring in Automated AI R&D","authors":["Lena Libon","Ben Rank","Jehyeok Yeon","David Schmotz","Jeremy Qin","Daniel Donnelly","Derck Prinzhorn","Maksym Andriushchenko"],"year":2026,"abstract":"As AI agents begin to automate AI R&D, we need ways to assess whether their outputs are safe to deploy, even when the agents themselves may be untrusted. AI control offers one such approach: rather than trusting the agent, it treats it as a potential adversary and uses a monitor to detect covert sabotage before deployment. We evaluate AI control for automated AI R&D with ResearchArena, a framework spanning four long-horizon tasks: safety post-training, capabilities post-training, CUDA-kernel opt","url":"https://arxiv.org/abs/2607.19321","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00726","type":"paper","title":"Engineering Trustworthy Agentic AI for Critical Systems","authors":["Omar Al-Refai","Ibrahim Shahbaz","Adam Ali Husseinat","Michael Mandulak","Jaewon Kim","Eman Hammad"],"year":2026,"abstract":"Agentic artificial intelligence systems, capable of autonomous perception, planning, tool use, and multi-step action, are increasingly proposed for critical engineering domains where decisions carry physical, operational, or economic consequences. This survey addresses a gap in current literature by treating trustworthiness, whether agentic behavior can be verified, audited, and trusted under the constraints that engineering practice actually requires, as a first-class engineering property, rath","url":"https://arxiv.org/abs/2607.18548","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00727","type":"paper","title":"Operational Hallucination and Safety Drift in AI Agents","authors":["Shasha Yu","Fiona Carroll","Barry L. Bentley"],"year":2026,"abstract":"Large language models (LLMs) serving as planners in tool-using autonomous agents introduce dynamic reliability risks in multi-turn execution. While single-turn safety mechanisms are relatively mature, extended interactions reveal structural vulnerabilities where initial alignment degrades over time. This paper empirically characterizes two observed failure modes across multiple state-of-the-art LLMs: Safety Drift, the gradual erosion of declared safety intent leading to constraint-violating acti","url":"https://arxiv.org/abs/2607.18366","categories":["guardrails","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00728","type":"paper","title":"Specification-Driven Development as the Foundation of AI-Native Enterprise Software Engineering","authors":["Mamdouh Alenezi"],"year":2026,"abstract":"Large language models (LLMs) and agentic AI are shifting software engineering from manual coding toward intent specification, architecture, and governance. Two paradigms have emerged: vibe coding, an intuition-driven approach accepting AI artifacts via observed behavior, and Specification-Driven Development (SDD), which uses structured specifications as the authoritative source of truth. This article makes three contributions. First, based on a verified literature corpus, it identifies failure m","url":"https://arxiv.org/abs/2607.16680","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00729","type":"paper","title":"TRACER-AI: A Multi-Layer Explainable Framework for Prompt Injection, Agent Goal Hijacking, and Tool Misuse Detection in Agentic AI Systems","authors":["Pallavi Singh","Khushboo Gupta","Pratibha Singh"],"year":2026,"abstract":"Large language model (LLM) agents extend generative models with planning, memory, and external tool access, but\nthis capability creates a security path in which untrusted content can alter instructions, hijack an agent's operational goal, and\ntrigger harmful tool actions. This paper proposes TRACER-AI, a four-layer explainable defense-in-depth framework that\ncombines (i) semantic prompt-injection detection, (ii) continuous goal-integrity monitoring, (iii) contextual tool-risk control, an","url":"https://doi.org/10.22214/ijraset.2026.84360","categories":["prompt-injection","agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00730","type":"paper","title":"Hybrid Analysis for Secure MCP Tool Use in LLM Agents","authors":["Ping He","Yuexiang Xie","Yaliang Li","Shouling Ji"],"year":2026,"abstract":"The rapid development of large language model (LLM) agents has enabled their broad adoption across diverse real-world tasks. To standardize interactions between LLM agents and external environments, Model Context Protocol (MCP) tools have emerged as a de facto standard and have been widely integrated into these systems. However, the use of MCP tools also introduces new safety risks, as LLM agents can be induced to perform malicious or unauthorized actions. Although prior work has proposed defens","url":"https://arxiv.org/abs/2607.25297","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00731","type":"paper","title":"ALIBI: Adaptive Agentic Attacks on LLM-Based Vulnerability Detectors via Adversarial Code Comments","authors":["Zixuan Wu","Cristina Nita-Rotaru"],"year":2026,"abstract":"Large language models are increasingly deployed for security-sensitive tasks such as vulnerability detection and code review. Their reliance on natural-language context embedded in source code exposes a previously underexplored attack surface: adversarial comments that can influence a detector's reasoning without changing program behavior. We study LLM-based vulnerability detectors against a new adversary: a coding agent that implements new functionality, deliberately introduces vulnerabilities,","url":"https://arxiv.org/abs/2607.24964","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00732","type":"paper","title":"Do LLMs Know Their Vulnerable Scenarios?","authors":["Ziheng Peng","Huiqi Deng","Haoran Jing","Xuankun Rong","Jiahui Han","Xiting Wang","Na Zou","Xia Hu"],"year":2026,"abstract":"Safety-aligned large language models are trained to refuse harmful requests, yet embedding the same requests in particular scenarios can bypass their safeguards. Existing red-teaming methods empirically identify effective scenarios through observed attack outcomes, but why particular scenarios weaken refusal remains mechanistically unclear. Meanwhile, mechanistic interpretability studies have characterized both refusal directions and jailbreak-associated features, without explaining the relation","url":"https://arxiv.org/abs/2607.23496","categories":["jailbreaking","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00733","type":"paper","title":"Poster: Rethinking Security in LLM Code Generation through Real-World Risk Scenarios","authors":["Lixun Ma","Ruolong Ma","Bei Wang","Feng Wei","Zhenguang Liu","Lorenzo Cavallaro","Wentao Chen"],"year":2026,"abstract":"Large Language Models (LLMs) are widely used for code generation, yet their security behavior in realistic development workflows remains underexplored. Existing benchmarks often rely on explicitly specified security requirements, failing to capture real-world scenarios where prompts are frequently ambiguous or incomplete. In this paper, we adopt a developer-centric perspective and identify three representative risk scenarios that commonly lead to security vulnerabilities in LLM-generated code: A","url":"https://arxiv.org/abs/2607.23088","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00734","type":"paper","title":"Mask2Shield: Strengthening LLM Safety against Neuron-Pruning Attacks","authors":["Ying JinCheng","Minghui Xu","Yinhao Xiao","Xiuzhen Cheng","Wencheng Yang"],"year":2026,"abstract":"Large language models (LLMs) are safety-aligned before deployment to reduce harmful content generation. Yet neuron-level pruning attacks show that refusal can depend on a small set of removable units: disabling them can remove safety behavior while leaving much of the model usable. To address this problem, we introduce Mask2Shield (M2S), a masked-forward alignment method that trains a model under this functional pruning. The masked student must recover a safe refusal through the remaining comput","url":"https://arxiv.org/abs/2607.23015","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00735","type":"paper","title":"Ethereum NFT Smart Contracts: Knowledge-Guided Vulnerability Detection with LLM and Code Slicing","authors":["Deyu Yang","Rundong Wei","Xiaoqi Li"],"year":2026,"abstract":"Ethereum non-fungible tokens (NFTs) implement ownership, transfer, authorization, and metadata operations through smart contracts, making contract vulnerabilities a direct risk to digital assets. Existing static analyzers provide efficient rule-based screening but can struggle with application-specific logic, whereas unconstrained large language model analysis may be distracted by irrelevant code or produce inconsistent outputs. We present a vulnerability-detection method that combines vulnerabi","url":"https://arxiv.org/abs/2607.21983","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00736","type":"paper","title":"Piggybacking on Perception: Stealthy Concurrent Audio Prompt Injections against Multimodal LLM Agents","authors":["Mingxiao Liu","Yitong Li","Haoren Zhao","Yaoxiang Bian","Jianan Ma","Jian Zhang","Jialuo Chen","Xinhao Deng","Zhen Wang"],"year":2026,"abstract":"Large Language Model (LLM)-driven multimodal agents are increasingly deployed to execute autonomous tasks via continuous audio interaction. While this paradigm enhances interaction naturalness, it introduces a critical yet under-explored attack surface, as audio inputs inevitably contain environmental noise beyond user control. In this paper, we investigate concurrent audio prompt injection attacks targeting multimodal agents. Distinct from traditional acoustic attacks on voice devices, we propo","url":"https://arxiv.org/abs/2607.28165","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00737","type":"paper","title":"RoguePrompt: Dual-Layer Encoding for Self-Reconstruction to Circumvent LLM Moderation","authors":["Benyamin Tafreshian","Prathamesh Dhake"],"year":2026,"abstract":"Large language models (LLMs) are becoming increasingly integrated into mainstream development platforms and daily technological workflows, typically behind moderation and safety controls. Despite these controls, preventing prompt-based policy evasion remains challenging, and adversaries continue to \"jailbreak\" LLMs by crafting prompts that circumvent implemented safety mechanisms. Prior work has established cipher-mediated interaction, code-embedded decryption, prompt decomposition and reconstru","url":"https://arxiv.org/abs/2607.27373","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00738","type":"paper","title":"Recover, Decode, Reguard: Guard-Agnostic Defense Amplification againstEncoded VLM Jailbreaks","authors":["Haoyu Zhang","Zhuoxi Wang","Shibo Zheng","Zijian Xiao","Xiangchen Guan","Mohammad Zandsalimy","Shanu Sushmita"],"year":2026,"abstract":"Safety classifiers (\"guards\") are the dominant black-box defense for vision-language models, yet they judge an input's surface form, not its meaning: a harmful request re-encoded as set theory, formal logic, a rare language, code, or an image of text slips past a guard that would block it in plain language -- the decode gap. The natural fix is a guard-agnostic recover-and-decode amplifier that transcribes image content and restates encoded text into its plain payload before the guard, so any off","url":"https://arxiv.org/abs/2607.26574","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00739","type":"paper","title":"GPT-Red: Automated Red Teaming via Self-Play at Scale","authors":["Eric Wallace","Christopher A. Choquette-Choo","Nikhil Kandpal","Sam Toyer","Dylan Hunn","Stephanie Lin","Yuxin Wen","Xiangyu Qi","Christopher Wolff","Zizhao Wang","Milad Nasr","Sicheng Zhu","Chuan Guo","Juan Felipe Cerón Uribe","Kaiwen Wang","Aiden Low","Kai Xiao","Kai Chen"],"year":2026,"abstract":"We introduce \\textbf{GPT-Red}, an automated red-teaming agent that is trained to discover novel prompt injection attacks against frontier LLMs. The goal of this model is to evaluate and improve the robustness of our production systems. To this end, we use it to adversarially train GPT-5.6, our most robust model to prompt injections to date. To create GPT-Red, we design a scalable self-play algorithm where the model is tasked with attacking a diverse population of simultaneously-trained defender ","url":"https://arxiv.org/abs/2607.26115","categories":["prompt-injection","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00740","type":"paper","title":"TYPO: Instruction-Dense Visual Jailbreaks against Commercial Closed-Source Image-Generation Models","authors":["Meng Xie","Li Zeng","Hangtao Zhang","Xianlong Wang","Ziqi Zhou","Pengpeng Qiao","Zhetao Li"],"year":2026,"abstract":"Recent commercial image-generation models can generate high-quality images with readable text (e.g., posters, infographics, and manuals), attracting considerable attention. Yet we first show that this same capability also introduces a previously unreported safety vulnerability: these systems may refuse to generate harmful text directly, yet permit the same content when rendered as text within generated images, i.e., safety alignment does not reliably transfer from textual outputs to text embedde","url":"https://arxiv.org/abs/2607.24897","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00741","type":"paper","title":"Agentic Permissions Policy Algebra for Taint Confinement in LLM Agents","authors":["Arseny Kravchenko","Vadim Liventsev","Innokentii Konstantinov","Ildar Iskhakov","Matvey Kukuy"],"year":2026,"abstract":"Autonomous LLM agents processing mixed-confidentiality data face severe security risks from prompt injection attacks and reasoning errors. While dynamic Information Flow Control (IFC) provides structural security guarantees, traditional taint tracking permanently taints an agent's context upon reading unvetted data, severely restricting downstream utility. We present APPA (Agentic Permissions Policy Algebra), an IFC framework that resolves this usability bottleneck through engine-managed context","url":"https://arxiv.org/abs/2607.24625","categories":["prompt-injection","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00742","type":"paper","title":"When LLM Defenses Backfire: Characterizing Safety, Performance, and Cost Trade-offs","authors":["Tong Zhang","Zexin Li","Simin Chen","Yun Peng"],"year":2026,"abstract":"Jailbreak defenses are essential for protecting large language models (LLMs), but they can also introduce secondary costs that weaken model utility. We present a systematic study of these defense trade-offs along three dimensions: performance impact, over-refusal on benign inputs, and inference cost. Rather than treating defenses as a single class, we organize them by operational strategy and examine how different strategies correlate with different side-effect profiles. Across state-of-the-art ","url":"https://arxiv.org/abs/2607.24392","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00743","type":"paper","title":"Just Testing, Move Along: Evasion of LLM-based System Log Interpretation by Prompt Injection","authors":["Max Landauer","Florian Skopik","Markus Wurzenberger","Franciszek Górski","Mateusz Krzysztoń"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly integrated into Security Operations Center (SOC) workflows, where they support analysts in tasks such as the interpretation of system logs. However, the ability of LLMs to directly process untrusted textual input also introduces new attack surfaces. In particular, attackers can inject contextual information or explicit instructions into log entries in order to influence how malicious activity is interpreted by the model. Despite the growing adoption ","url":"https://arxiv.org/abs/2607.24174","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00744","type":"paper","title":"Agentic Cloud Decoys: A Deception-Driven Framework for Autonomous Intrusion Investigation","authors":["Mohan Manivannan","Dalal Alharthi"],"year":2026,"abstract":"Cloud telemetry arrives at a scale that, paradoxically, makes intrusion understanding harder rather than easier. Attackers operate through legitimate identity, federated session tokens, and cloud native APIs indistinguishable from routine administration, and analysts spend an incident reconstructing context the logs already contain. We present Cloud Decoy AI Agent, a framework pairing a high fidelity cloud decoy with an autonomous language model agent that compresses the path from suspicious act","url":"https://arxiv.org/abs/2607.24006","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00745","type":"paper","title":"ContainmentBench: Trace-Based Evaluation of Post-Injection Containment in Tool-Using LLM Agents","authors":["Wenhao Lan","Shan Li","Xinhua Lai","Meiqi Wu","Junbin Yang","Haihua Shen"],"year":2026,"abstract":"Tool-using LLM agents process untrusted content, maintain memory, delegate across agents, and invoke side-effecting tools. Existing prompt-injection evaluations typically summarize security with terminal attack or policy outcomes, but equal endpoints can conceal different post-exposure traces and different losses of authorized utility. We introduce ContainmentBench, a sandboxed, trace-based benchmark that separately measures benchmark-defined endpoint policy compliance, instrumented logged propa","url":"https://arxiv.org/abs/2607.23999","categories":["prompt-injection","agentic-threats","sandboxing-isolation","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00746","type":"paper","title":"Are You Still the Agent I Authorized? Earned Authority under a Fixed Ceiling for Evolving Agents","authors":["Zhaoxi Zhang","Xiaomei Zhang"],"year":2026,"abstract":"Long-lived AI agents increasingly evolve after deployment by retaining experience, acquiring skills and tools, revising workflows, delegating work, and moving across task phases. This improves adaptation but creates a distinct authorization problem. Tool-enabled agents can turn model errors and prompt injections into consequential external actions; when evolution occurs under a live grant, the subject exercising that authority or the context in which it acts may no longer match what the user eva","url":"https://arxiv.org/abs/2607.23586","categories":["prompt-injection","access-control"],"reviewed":false},{"id":"llmsec-2026-00747","type":"paper","title":"Mission-Level Runtime Assurance for LLM-Assisted ISR Swarms over a Verification-Aware Fabric","authors":["Nikolaos Kekatos","Stylianos Basagiannis","Panagiotis Katsaros","Alexios Lekidis","Tom Nianios"],"year":2026,"abstract":"Swarms of LLM-assisted autonomous robots are increasingly proposed for cooperative intelligence, surveillance, and reconnaissance (ISR) in contested environments. A growing class of their assurance failures arises not within any single platform but across the swarm: individually-compliant actions compose into a mission-level violation: a prohibited objective split across platforms to evade per-platform lim- its, or a collective budget quietly exceeded. Per-platform guardrails miss these by const","url":"https://arxiv.org/abs/2607.23532","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00748","type":"paper","title":"Agent Security Needs Redefinition through a Holistic Framework","authors":["Vincent Siu","Jingxuan He","Kyle Montgomery","Zhun Wang","Chenguang Wang","Dawn Song"],"year":2026,"abstract":"Agent security is widely treated as a question about action content. Defenses ask whether an instruction looks malicious. Benchmarks ask whether an agent performs a harmful sounding action. \\textbf{We argue that agent security is fundamentally a contextual problem, and that the current content based framing systematically misdefines it.} A command to ``delete user data'' might be a routine administrative request or a prompt injection attacking production systems, and the content alone cannot dis","url":"https://arxiv.org/abs/2607.22024","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00749","type":"paper","title":"Protocol-Level Attacks on Agentic Commerce Platforms: A Cross-Platform Taxonomy, AIP-Bench, and Unified Defense","authors":["Yedidel Louck"],"year":2026,"abstract":"Agentic commerce platforms let AI agents autonomously discover services, move payments, and wield user credentials on their users' behalf, and they already handle real money. Their security has so far been studied almost entirely at the level of the AI model, through prompt injection and misalignment. We show that the more consequential risks lie one layer down, in the protocol between agents and commerce services. There, vulnerabilities are structural : exploitation is deterministic and ndepend","url":"https://arxiv.org/abs/2607.21824","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00750","type":"paper","title":"RAGuard: A Layered Defense Framework for Retrieval-Augmented Generation Systems Against Data Poisoning","authors":["Pushkal Kumar","Tucker Nielson","Tanish Kolhe","Shubham Zala","Vincent Li"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) systems ground large language models (LLMs) in external corpora, but this reliance exposes them to corpus poisoning: maliciously injected passages that manipulate retrieved evidence. We introduce RAGuard, a layered defense against \\emph{factual} corpus-poisoning attacks on RAG pipelines. The first layer adversarially fine-tunes a dense retriever on synthetic poisoned documents (fabricated facts, contradictions, and reasoning traps), teaching it to downrank ma","url":"https://arxiv.org/abs/2607.26339","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00751","type":"paper","title":"Architectural Backdoors in Vision-Language Model Supply Chains via Representation Steering","authors":["Maria Rosaria Briglia","Igor Maljkovic","Antonio Emanuele Cinà","Luca Oneto","Iacopo Masi","Fabio Roli"],"year":2026,"abstract":"Vision--Language Models (VLMs) are increasingly deployed through a model supply chain in which pretrained checkpoints, architecture definitions, text encoders, and exported computation graphs are distributed by third parties and reused across downstream services. This reuse model creates a security-critical trust boundary: VLM deployments inherit not only learned parameters but also executable behavior encoded in shared model artifacts. In this paper, we show that a malicious provider can exploi","url":"https://arxiv.org/abs/2607.25479","categories":["data-poisoning","supply-chain-attacks"],"reviewed":false},{"id":"llmsec-2026-00752","type":"paper","title":"Early Detection of Distributed Backdoors in Multi-Agent LLM Systems: A Characterization Study","authors":["Diego Fernandez Arias","Dev Prashant Mistry","Ren Wang","Yibo Hu"],"year":2026,"abstract":"Multi-agent LLM systems can be attacked by a payload that no single agent ever holds in full: a poisoned tool hides encrypted fragments in its observations, spreads them across several agents, and an external step reassembles and executes them after the run. Per-step safety checks that judge each action in isolation may fail to recognize the complete distributed payload. We investigate how early such an attack can be detected while the run is still unfolding, and how robustly it can be caught on","url":"https://arxiv.org/abs/2607.24893","categories":["data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00753","type":"paper","title":"Open Security Benchmark: Towards Autonomous Enterprise Cyber Defense","authors":["Gal Engelberg","Michael Arenzon","Leon Goldberg"],"year":2026,"abstract":"Enterprises are moving toward autonomous cyber defense: agentic AI that builds situational awareness of an organization's security state and reasons from it to assessments, decisions, and actions. This rests on a holistic view of the enterprise's security state, the continuous, cross-vendor picture of identities, cloud and infrastructure, data, applications, and their configurations that security posture management assembles. As agents take on this work, what matters is not whether an agent can ","url":"https://arxiv.org/abs/2607.27288","categories":["agentic-threats","benchmarks","cloud-ai-security"],"reviewed":false},{"id":"llmsec-2026-00754","type":"paper","title":"Distributing Security Controls Through Harness Engineering","authors":["William Robert Gore"],"year":2026,"abstract":"AI coding agents are being adopted at historic speed, yet security and risk concerns remain the primary barrier to scaling agentic AI across organizations. Existing security controls for coding agents are not systematically distributed to engineering teams, and vendor-native solutions introduce ecosystem dependencies that may not suit every deployment context. This paper investigates whether off-the-shelf security controls can be implemented on commercial AI coding agents and scaled to a distrib","url":"https://arxiv.org/abs/2607.25890","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00755","type":"paper","title":"Cyber-Capable AI Agents: Vulnerabilities, Evaluation Containment, and Defensive Response","authors":["Abu Bakar Siddik"],"year":2026,"abstract":"Cyber-capable AI agents combine language models with tools, memory, and execution en- vironments to perform multi-step offensive-security tasks. Existing work separately measures cyber capability and catalogs attacks against agent components, but provides less guidance on containing a capable agent within the environments used to evaluate it. This review synthe- sizes five vulnerability classes at that boundary: multi-step offensive chains, objectives that conflict with sandbox boundaries, suppl","url":"https://arxiv.org/abs/2607.25379","categories":["sandboxing-isolation"],"reviewed":false},{"id":"llmsec-2026-00756","type":"paper","title":"ReCon: A Resource-Constrained Benchmark for LLM-Based Cybersecurity Compliance Across Ingestion and Retrieval Pipelines","authors":["Rohit Negi","Rishik Jain","Soumyo V Chakarborty","Amit Negi","Sandeep K Shukla"],"year":2026,"abstract":"With the increasingly aggressive cyber threat landscape for governments, businesses, and institutions, as information and/or cybersecurity implementations are increasingly under scrutiny by regulators, it has been pointed out that governance failure is one of the major reasons for a weakened cybersecurity posture. A major component of Cyber/information security governance is the development, adoption, and implementation of a comprehensive information and/or cyber security policy document. The po","url":"https://arxiv.org/abs/2607.22885","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00757","type":"paper","title":"Breaking Customized LLMs for Coding: Automated Red Teaming for Instruction Backdoor Attacks","authors":["Yuchen Chen","Wei Cheng","Yuan Xiao","Wising Sun","Chunrong Fang","Yang Liu","Zhenyu Chen","Baowen Xu"],"year":2026,"abstract":"LLM customization platforms allow users to build task-specific models for code intelligence tasks by embedding instructions into system prompts, without modifying the underlying model parameters. While these platforms lower the barrier to developing customized LLMs, they also introduce a new attack surface: instruction backdoor attacks, in which adversaries implant hidden malicious behaviors into customized instructions. However, existing attacks suffer from two key limitations. First, they ofte","url":"https://arxiv.org/abs/2608.05659","categories":["data-poisoning","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00758","type":"paper","title":"LLM-Assisted Detection and Repair of Hardware Security Vulnerabilities in Verilog Designs","authors":["Ethen Santana","Gabriel Gyaase","Hao Zheng"],"year":2026,"abstract":"Hardware designs, like software, are susceptible to bugs that can introduce security vulnerabilities and create opportunities for malicious exploitation. Unlike software vulnerabilities, however, hardware flaws become permanently embedded in silicon after fabrication, making them difficult or impossible to patch. Many of these weaknesses are categorized under the Common Weakness Enumeration (CWE) framework and include improper access control, exposure of sensitive information, and unintended pri","url":"https://arxiv.org/abs/2608.04907","categories":["membership-inference","access-control"],"reviewed":false},{"id":"llmsec-2026-00759","type":"paper","title":"LoginTrap: Uncovering Task-Agnostic Phishing-Style Indirect Prompt Injection Attacks against LLM-based Web Agents","authors":["Longtao Guo","Zelin Zhang","Kaifeng Huang","Yang Shi"],"year":2026,"abstract":"LLM-based web agents automate user tasks by observing webpages and executing browser actions on behalf of users. As these agents operate on real web services, login becomes a sensitive authentication boundary because it involves credentials and sensitive information. Existing work shows that malicious webpage content can manipulate web agent actions, but it has not fully examined whether such content can induce login and cause end-to-end private data leakage. We study this attack surface and pre","url":"https://arxiv.org/abs/2608.04741","categories":["prompt-injection","membership-inference"],"reviewed":false},{"id":"llmsec-2026-00760","type":"paper","title":"Eliciting Intrinsic Hallucinations in LLMs via Semantically Equivalent Adversarial Attacks","authors":["Atri Vivek Sharma","Brian Formento","Alessio Lomuscio"],"year":2026,"abstract":"Large language models (LLMs) are often used in conjunction with external knowledge sources to improve their factual accuracy and decrease hallucinations, through methods such as Retrieval-Augmented Generation (RAG). However, these systems remain susceptible to intrinsic hallucinations, where the model generates unfaithful or fabricated information that is not supported by the retrieved evidence. We propose a novel framework to assess model robustness against this phenomenon by stress-testing usi","url":"https://arxiv.org/abs/2608.04286","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-00761","type":"paper","title":"MAFIA: Query-Only Memory Attacks via Probing and Factual Injection against Audited LLM Agents","authors":["Jiaming Chen","Yisen Gao","Yanping Li","Zifan Liu","Yumeng Zhang","Jun Zhang"],"year":2026,"abstract":"Memory-augmented LLM agents rely on rich context for long-horizon reasoning and acting, yet their memory modules expose a persistent attack surface for malicious records, making the study of memory poisoning threats imperative. However, existing query-only attacks often fail to remain effective in two realistic and prevalent settings: large-scale benign memory pools and active input auditing. Consequently, current approaches fall short when facing the dual challenges of high retrieval competitiv","url":"https://arxiv.org/abs/2608.03844","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00762","type":"paper","title":"A Security-Oriented Lifecycle Model for Large Language Model Systems","authors":["Eleftherios Batzolis","George Drosatos","Vassilis Katsouros","Konstantinos Rantos"],"year":2026,"abstract":"Large language models are being integrated into critical infrastructure and enterprise workflows at unprecedented scale,yet the lifecycle frameworks governing their development and operations were designed for operational efficiency rather than security analysis. As a result, security-relevant activities such as data provenance verification, artifact signing, agentic permission control, and decommissioning are often left implicit or assumed to receive due care. Governance frameworks, in turn, or","url":"https://arxiv.org/abs/2608.03626","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00763","type":"paper","title":"DiagChain: A Diagnostic Benchmark for Evaluating LLM Agents on Evidence-Grounded Attack Chain Reconstruction","authors":["Xuyang Liu","Yibin Han","Zhenwei Zhang","Kai Chang","Zhiwei Xu","Tian Qiu","Weixian Deng","Jiabao Gao","Xiaolin Peng","Hai Wan","Xibin Zhao"],"year":2026,"abstract":"Large Language Model (LLM) agents offer a promising approach to attack chain reconstruction by retrieving and interpreting heterogeneous telemetry to infer ordered attacker actions. However, existing benchmarks mainly evaluate final outputs or aggregate accuracy, providing limited insight into how errors arise and propagate across intermediate reasoning stages. We present DiagChain, a diagnostic benchmark for evidence-grounded attack chain reconstruction that enables stage-wise evaluation of LLM","url":"https://arxiv.org/abs/2608.03591","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00764","type":"paper","title":"Security-First Evaluation of Text-to-Terraform: Benchmarking LLMs and SLMs for Secure IaC Generation","authors":["Francis Luis Santos Vargas","Rodrigo Brandão Mansilha","Diego Kreutz"],"year":2026,"abstract":"Cloud misconfiguration remains a leading cause of security incidents, yet whether LLMs and SLMs can generate security-compliant Infrastructure-as-Code is an open question. We benchmark seven models, three closed LLMs (Claude Opus 4, GPT-5.4, Gemini 2.5 Pro) and four open SLMs (Qwen2.5-Coder-14B, WizardCoder-33B, CodeLlama-13B, Magicoder-S-CL-7B), on AWS Terraform generation across 17 scenarios, integrating Checkov and Trivy scanners into a GitLab CI/CD pipeline and evaluating two prompt strategi","url":"https://arxiv.org/abs/2608.02672","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00765","type":"paper","title":"Adversarial Attacks in Multi-Agent LLM Pipelines: Unveiling Structural Vulnerabilities in Agentic AI Architectures","authors":["Faisal Haque Bappy","Tahrim Hossain","Tarannum Shaila Zaman","Raiful Hasan","Kamrul Hasan","Tariqul Islam"],"year":2026,"abstract":"Multi-agent LLM pipelines orchestrate multiple specialized language model agents into structured workflows where intermediate outputs are passed across agents to solve complex tasks. This design introduces a security gap absent in single-agent settings: once an agent accepts adversarial content, it is propagated as trusted input throughout the pipeline. We argue that this vulnerability stems from the absence of boundary verification, a security primitive that enforces explicit validation of data","url":"https://arxiv.org/abs/2608.00718","categories":["adversarial-examples","agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00766","type":"paper","title":"Stateful Cooperative Agents Safeguarding LLMs Against Evolving Multi-Turn Attacks","authors":["Siyuan Li","Zehao Liu","Haoyu Li","Xi Lin","Ning Liu","Jun Wu","Jianhua Li","Mohsen Guizani"],"year":2026,"abstract":"As LLMs become increasingly integrated into complex applications, their vulnerability to adversarial attacks has raised significant concerns. However, existing defenses remain reactive in nature. This limitation makes it difficult for them to counter sophisticated threats, as adversaries continuously adjust their strategies across multi-turn interactions. In this paper, we present a proactive defense framework for securing LLMs against evolving multi-turn adversarial attacks that combines disrup","url":"https://arxiv.org/abs/2608.00134","categories":["adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-00767","type":"paper","title":"Robust Context-Aware Detection of Malicious Instructions in Text","authors":["Buzhao Liu","Xinhang Ma","Yevgeniy Vorobeychik"],"year":2026,"abstract":"The remarkable instruction-following ability of modern LLMs has enabled their practical use as the minds of agents that can autonomously complete increasingly complex tasks. Therein, however, also lies their vulnerability to attacks which embed malicious instructions in text, common variants of which are known as indirect prompt injection (IPI). A fundamental task in addressing this vulnerability is successful segmentation of a given text into benign and malicious sentences (if any). While a num","url":"https://arxiv.org/abs/2608.05430","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00768","type":"paper","title":"Agent Against Agent: An Agentic System for Automatic Prompt Injection Red Teaming","authors":["Yanting Wang","Chenlong Yin","Runpeng Geng","Jinyuan Jia"],"year":2026,"abstract":"Prompt injection poses significant security risks to LLM agents. Efficient and effective red-teaming is therefore critical, both for evaluating these risks and for collecting training data to improve defenses. Existing state-of-the-art prompt injection red-teaming methods primarily rely on reinforcement learning (RL), producing attacker models that often generalize poorly to new target LLMs. In this work, we develop PIMiner, an agentic system for prompt injection red-teaming. During training, PI","url":"https://arxiv.org/abs/2608.05108","categories":["prompt-injection","agentic-threats","red-teaming","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00769","type":"paper","title":"Breadcrumbing Search Agents","authors":["Xuebin Li","Hanqing Zhao","Siyuan Liang","Kejiang Chen","Weiming Zhang","Dacheng Tao","Nenghai Yu"],"year":2026,"abstract":"LLM-based search agents are widely used for information-seeking tasks, but their reliance on external tool returns introduces a critical security risk: web content retrieved during execution is untrusted, exposing agents to prompt injection and goal hijacking. Prior work on search-agent safety primarily focuses on static web-content injection, but modern agents issue follow-up queries and cross-check competing sources, so a single injected page is often diluted or rejected. We show that the chan","url":"https://arxiv.org/abs/2608.04565","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00770","type":"paper","title":"Behavioral Skill Reconstruction: Reconstructing Hidden Functionality from LLM Agent Skills","authors":["Peichun Hua","Haoxuan Xu","Mengyuan Li"],"year":2026,"abstract":"Closed source agent skills may encode proprietary instructions, scripts, constants, and data. Providers may offer their capabilities as services while keeping the underlying packages hidden. Prior work focuses on prompt injection attacks that directly disclose these artifacts, and existing defenses accordingly aim to prevent such leakage. However, preventing file disclosure does not prevent users from recovering the functionality those files implement. This raises a fundamental question: can a u","url":"https://arxiv.org/abs/2608.04192","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00771","type":"paper","title":"AgentAntibody: An Adaptive Immune System for Defending LLM Agents against Prompt Injection","authors":["Shihao Weng","Yang Feng","Xiaofei Xie","Jiongchi Yu"],"year":2026,"abstract":"Prompt injection remains a critical threat to LLM agents, yet existing defenses treat each task as a self-contained problem, independent of previous encounters. In practice, user requests are often underspecified: they describe the desired outcome without fully specifying acceptable behavior. An injection can exploit this ambiguity, causing the agent to complete the task in a way the user would reject. As the user's expectations become clearer through concrete cases, a defense should learn from ","url":"https://arxiv.org/abs/2608.04053","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00772","type":"paper","title":"AI Security Leaderboard: Methodology, Results and Minimal Standard","authors":["Jasper Timm","Lukas Struppek","Ziwei Xu","Grace Cheong","Oscar Mata","Dan Zhao","Mick Yang","Isadora De Andrade","Xiaojun Jia","Yiming Li","Samuel Bauer","Heather McIntyre","Adam Gleave","Edward Yee","Kellin Pelrine"],"year":2026,"abstract":"The AI Security Leaderboard is an independent benchmark that ranks the safeguards of frontier AI models from least to most secure. It tests models against the FAR$.$AI Minimal Standard for Safeguards, which represents a minimum bar for security: meeting it does not guarantee a secure model, but failing to meet it guarantees a lack of state-of-the-art security. Version 1.0 covers severe misuse requests across chemical, biological, radiological, nuclear, and explosive (CBRNE) threats and offensive","url":"https://arxiv.org/abs/2608.03070","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00773","type":"paper","title":"A Multimodal Automatic Redteaming Evaluation based on Atomic Jailbreak Strategy Decoupling and Combination","authors":["Shiji Zhao","Yuxuan Zhou","Chen Xiong","Dongxian Wu","Yang Bai","Xun Chen"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) have achieved impressive progress in image-text comprehension and generation, yet they remain susceptible to jailbreak attacks that can trigger harmful outputs and pose serious safety concerns. Existing multimodal jailbreak attacks have shown the feasibility of such attacks, but they still face two fundamental challenges: the lack of a atomic multi-modal strategy space, the absence of a concise and efficient executable framework beyond human-craft experie","url":"https://arxiv.org/abs/2608.04034","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00774","type":"paper","title":"No Single Neuron of Failure: Distributed Safety Alignment Against White-Box Attacks","authors":["Simiao Xie","Chuancheng Shi","Shangze Li","Wenhua Wu","Fei Shen","Ying Zhou","Zhiyong Wang","Tat-Seng Chua"],"year":2026,"abstract":"With the rapid release of open-weight large foundation models, safety threats are shifting from black-box jailbreaks to neuron-level white-box attacks that directly identify and manipulate safety-related neurons. Existing alignment methods often investigate the safety behavior on a small number of neurons, creating fragile single point of failure with limited redundancy. To address this issue, we propose distributed safety alignment (DSA), which redundantly encodes safety capabilities across mul","url":"https://arxiv.org/abs/2608.01414","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00775","type":"paper","title":"Moving the Safety Barrier: Dynamic Routing Adaptive Alignment Against White-Box Attacks","authors":["Shangze Li","Chuancheng Shi","Simiao Xie","Lingzhi He","Cheng Ji","Zifeng Cheng","Fei Shen","Chao Wu","Tat-Seng Chua"],"year":2026,"abstract":"With the widespread deployment of large foundation models (LFMs) in open environments, safety threats are shifting from black-box jailbreaks toward white-box attacks that directly identify and disrupt internal safety neurons or routes. However, existing safety defenses often rely on static safety units or fixed refusal pathways, leaving models highly vulnerable to targeted route-level white-box attacks. For that, we propose dynamic routing adaptive alignment (DRAA), a framework that introduces d","url":"https://arxiv.org/abs/2608.02674","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00776","type":"paper","title":"The Boy Who Cried Wolf: Adversarial Misclassification of Safe Inputs as Unsafe in Multimodal Guardrails","authors":["Shuo Shi","Rui Yin","Naen Xu","Jiahao Chen","Chunyi Zhou","Tianyu Du","Zhihui Fu","Jun Wang","Zhaoxiang Wang","Shouling Ji"],"year":2026,"abstract":"Multimodal guard models have emerged as critical safety components for screening content in vision-language systems. While adversarial research has extensively studied jailbreaking attacks that produce false negatives, the inverse threat of inducing false positives on benign inputs remains unexplored. We introduce Unsafe Induction Attacks, where adversaries distribute imperceptibly perturbed safe images that trigger guard models to reject legitimate user requests, causing a \"Boy Who Cried Wolf\" ","url":"https://arxiv.org/abs/2608.01373","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00777","type":"paper","title":"SoK: Intent-Oriented Systematization of Multi-Turn LLM Jailbreaks","authors":["Siyuan Li","Aodu Wulianghai","Zehao Liu","Xi Lin","Qinghua Mao","Haoyu Li","Xiang Chen","Siyuan Liang","Jun Wu","Jianhua Li","Dacheng Tao"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly deployed in interactive settings, where user intent commonly unfolds through multi-turn dialogue. Multi-turn jailbreaks exploit this pattern by advancing a harmful intent across turns, so that no single message exposes the full objective. However, existing work treats these attacks as a loose collection of prompt patterns and does not analyze how the adversary organizes and advances harmful intent across an interaction. We develop a four-part, intent","url":"https://arxiv.org/abs/2608.01117","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00778","type":"paper","title":"Decoy Images Amplify Caption-Mediated Defenses Against Encoded Jailbreaks","authors":["Haoyu Zhang","Xiangchen Guan","Shibo Zheng","Mohammad Zandsalimy","Shanu Sushmita"],"year":2026,"abstract":"We report a counter-intuitive interaction between image inputs and existing black-box defenses on Vision--Language Models (VLMs): pairing an encoded jailbreak prompt with an unrelated decoy image can sharply lower attack success rate (ASR). The operative change is in the defense pipeline, not in the image. Across five frontier VLMs, two encoded-attack families, and three black-box defenses, a caption-mediated defense (ECSO) that leaves ASR essentially unchanged on text-only encoded input drops i","url":"https://arxiv.org/abs/2608.01043","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00779","type":"paper","title":"When Prompts Control Robots: Prompt Injection Attacks in Multi-Agent Robotic Systems","authors":["Neha Nagaraja","Amisha Bagari","Hayretdin Bahsi"],"year":2026,"abstract":"Large language models are increasingly integrated into autonomous robotic systems for task planning and control, but this integration exposes them to prompt injection attacks that can lead to unsafe decisions and physical harm. Multi-agent settings increase the risks through cross-agent contamination and broader attack surfaces. In this paper, we evaluate prompt injection attacks against an LLM-based multi-agent robotic system, considering both direct injections into task instructions and indire","url":"https://arxiv.org/abs/2608.00747","categories":["prompt-injection","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00780","type":"paper","title":"Your Agentic LLMs Secretly Encode Latent Signals of Indirect Prompt-Injection Exposure","authors":["Jianshuo Dong","Yiming Liu","Maosen Zhang","Nan Deng","Xu Peng","Xiaoping Zhang","Tianwei Zhang","Jie Zhang","Han Qiu"],"year":2026,"abstract":"Agentic LLMs are vulnerable to indirect prompt injection (IPI) attacks, e.g., malicious side-tasks hidden in external tool results. While many efforts have sought to address the threats, little is known about the internals of agentic LLMs when they are exposed to IPI attacks, a condition which we call IPI exposure. In this paper, we study this problem in depth from three aspects. (1) Probing: Across six models, including the giant 753B-parameter GLM-5.2, simple linear probes trained on pre-gener","url":"https://arxiv.org/abs/2608.02657","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00781","type":"paper","title":"Resourced Authority A Mechanism-Design Model for Participatory Governance of Deployed AI Agents","authors":["Praphul Chandra","Sujit Gujar","Ganesh Ghalme"],"year":2026,"abstract":"We give a formal mechanism design model for the continuous participatory governance of a deployed AI agent. The mechanism is built on the principle that governance should control an AI agent through resource allocation so as to make authorization self enforcing via compute budgets. The mechanism seeks to establish the Safe AI paradigm that compute is an effective governance lever. We situate our work as a compliance or commons overlay on a deployer. One governance period is an extensive form gam","url":"https://arxiv.org/abs/2608.06353","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00782","type":"paper","title":"From Siloed Algorithms to Compliance-First Agentic Platforms: A Multi-Layered Architecture for Hospital AI Systems","authors":["Manideep Dhar","Ritwik Singh","Sharat Chandra Kumar Manikonda"],"year":2026,"abstract":"Hospitals are rapidly adopting artificial intelligence for triage, imaging, scheduling etc., yet most deployments remain isolated point solutions locked inside departmental silos, resulting in duplicated effort, hidden risks, and unrealized enterprise value. Despite explosive growth of AI in healthcare market and accelerating investment, an estimated 70-80% of healthcare AI pilots fail to scale, largely due to governance gaps, fragmented data, and missing integration blueprints. This research pr","url":"https://arxiv.org/abs/2608.06112","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00783","type":"paper","title":"Blockchain Empowered Trustworthy Agent Networks: Foundations, Taxonomy, and Future Directions","authors":["Liehuang Zhu","Yuhang Li","Tianxing Wang","Zhihao Chen","Ke Li","Hongyi Liu","Yajie Wang","Lei Xu","Peng Jiang","Zijian Zhang"],"year":2026,"abstract":"AI agents are evolving from isolated task executors into networked autonomous entities that can communicate, delegate tasks, invoke tools, access external knowledge, and participate in cross-platform service and economic workflows. This evolution gives rise to open agent networks, where heterogeneous agents owned by different stakeholders interact without naturally shared infrastructures for identity, authorization, auditability, reputation, or settlement. This survey and tutorial article review","url":"https://arxiv.org/abs/2608.04626","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00784","type":"paper","title":"Binding Biometrics with AI Agent Identifiers for Delegation of Authority","authors":["Joseph Geo Benjamin","Anil K Jain","Karthik Nandakumar"],"year":2026,"abstract":"The proliferation of agentic artificial intelligence (AI) systems has raised serious questions about the accountability for tasks performed by AI agents. Ideally, an AI agent must not be allowed to perform critical tasks without explicit authorization by a human operator. Since biometric recognition is one of the most reliable approaches for authenticating individuals, it has the potential to enable authenticated delegation of authority to AI agents. In this work, we present a framework called B","url":"https://arxiv.org/abs/2608.04292","categories":["agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00785","type":"paper","title":"A game theory for foundation models shows new paths to rational cooperation through similarity inference","authors":["Alexander Meulemans","Maciej Wołczyk","Marissa A. Weis","Rajai Nasser","Roberta Rocca","Seijin Kobayashi","Guillaume Lajoie","Angelika Steger","Blake Richards","Marcus Hutter","James Manyika","Rif A. Saurous","João Sacramento","Blaise Agüera y Arcas"],"year":2026,"abstract":"As autonomous agents powered by foundation models are increasingly integrated into social and economic systems, understanding the principles governing their collective behavior is essential for ensuring safety and cooperation. Classical game theory, the dominant framework for modeling rational interaction, is built upon the assumption of `decoupled agency,' where agents treat their own decision-making as independent of the environment and other actors. Modern AI agents, however, jointly predict ","url":"https://arxiv.org/abs/2608.03958","categories":["autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00786","type":"paper","title":"Humans Are More Diverse: Frontier LLMs Show Extreme Policies in Idealised AI Development Races","authors":["Phu Hoa Pham","Duy Minh Dao Sy","Trung Kiet Huynh","Phu Quy Nguyen Lam","Chi Nguyen Tran","Minh Trung Le","Phong Hao Le","Dinh Nam Nguyen","Thien Ky Nguyen Dong","Elias Fernandez Domingos","Le Hong Trang","The Anh Han"],"year":2026,"abstract":"An AI development race creates a multi-agent safety dilemma. Each company can develop slowly and safely, or move faster while taking a risk that may remove its final reward. We use this repeated game to study strategic safety behaviour among large language model (LLM) agents in races with two to five players. However, a valid action does not show that an agent understands the game. We therefore place an audit gate before behavioural interpretation. We first verify the game engine, then test rule","url":"https://arxiv.org/abs/2608.01193","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00787","type":"paper","title":"Assuming You Knew: Fixing an Epistemic Semantics for Flow Policies Using Agentic AI","authors":["David A. Naumann"],"year":2026,"abstract":"Many high-level security requirements are about the allowed flow of information in programs and are difficult to make precise because they involve selective downgrading. Notions from epistemic logic have emerged as a good approach to policy semantics but a robust general framework remains elusive. A paper appearing in CSF 2018, entitled ``Assuming You Know: Epistemic Semantics of Relational Annotations for Expressive Flow Policies'', attempted to provide a unifying framework---but the formalizat","url":"https://arxiv.org/abs/2608.00882","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00788","type":"paper","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","authors":["William Caban"],"year":2026,"abstract":"Agentic AI evaluation pipelines produce benchmark scores that justify deployment decisions, safety certifications, and regulatory compliance claims. No formal framework has yet characterized how validity degrades across the stages of these pipelines. We present a three-layer compounding validity model, V_total <= V_1 x V_2 x V_3, that captures multiplicative degradation across task generation (V_1), human-simulator calibration (V_2), and automated judgment (V_3). Under empirically grounded estim","url":"https://arxiv.org/abs/2608.00794","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00789","type":"paper","title":"OpenART: Scaling Agent Red Teaming via Open-Ended Environment Evolution","authors":["Yunhao Chen","Xin Wang","Yixu Wang","Yi Liu","Jie Li","Yan Teng","Xingjun Ma","Xia Hu","Yu-Gang Jiang"],"year":2026,"abstract":"AI agents operate in persistent environments where early state changes can influence decisions far into the future. Unlike conventional language-model interactions, agent behavior is mediated through a shared state that is repeatedly modified and reused across long-horizon workflows. Current safety benchmarks often fail to capture these cumulative risks because they focus on short, static tasks. To address these limitations, we introduce OpenART, an open-ended arena for scalable agent red teamin","url":"https://arxiv.org/abs/2608.00677","categories":["red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00790","type":"paper","title":"AgenticRepair: Multi-Faceted Program Context Engineering for Agentic Vulnerability Repair","authors":["Michael Fu","Qiyue Mei","Patanamon Thongtanunam","Kla Tantithamthavorn"],"year":2026,"abstract":"Automated vulnerability repair aims to reduce the time and effort required to patch security flaws from a vulnerability triage report. Recent agentic AI approaches have shown promising results in automated program repair. However, vulnerability repair demands richer program context than general bug repair - context that security engineers routinely assemble in practice but that existing agentic approaches do not engineer. We identify three critical gaps: code-structure context capturing cross-fi","url":"https://arxiv.org/abs/2607.29422","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00791","type":"paper","title":"Skillsets on the Chain: A Blockchain-based Zero-Trust Framework for Agentic AI Networking","authors":["Yayu Gao","Yong Xiao","Hao Hu","Xubo Li","Zhiwei Liu","Yingyu Li","Guangming Shi","Ping Zhang"],"year":2026,"abstract":"Agentic AI networking (AgentNet) systems rely heavily on third-party skillset implementations and distributed multi-agent collaboration, yet they face major claim-to-capability inconsistencies and security vulnerabilities under trust-by-declaration assumptions. To bridge this gap, this paper proposes TrustAgentNet, a dual-tier blockchain-secured zero-trust framework. Specifically, a global Chain of Skillsets (CoS) governs the lifecycle of skillset metadata with protocols empowered by specialized","url":"https://arxiv.org/abs/2608.00104","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00792","type":"paper","title":"From C to Idiomatic Rust: A Ship-of-Theseus Agentic Translation","authors":["Vasily A. Sartakov"],"year":2026,"abstract":"C underpins operating systems, embedded platforms, and network infrastructure because its abstractions map directly to machine behaviour. Its explicit memory model, predictable data representations, and minimal runtime allow compilers to generate fast, deterministic code. These properties also leave correctness and memory safety entirely to the programmer, making undefined behaviour, pointer misuse, and lifetime errors persistent sources of defects and security vulnerabilities in long-lived C co","url":"https://arxiv.org/abs/2607.28835","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00793","type":"paper","title":"Enhancing System Security: LLM-Driven Defense Against Prompt Injection Vulnerabilities","authors":["Oleksandr Muliarevych"],"year":2024,"venue":"International Conference on Modern Problems of Radio Engineering, Telecommunications and Computer Science","abstract":"This article examines cybersecurity vulnerabilities in systems utilizing Language Model Interfaces, focusing on the challenges of building secure systems. It provides an overview of current interfaces and their associated risks. A key contribution is the design of a prompt analysis and injection detection subsystem, which assesses input relevance and security. The integration of an additional Filter level in the system safeguards against prompt injection attacks by pre-processing user requests a","url":"https://www.semanticscholar.org/paper/95be2ab522131b725b79062db7ceb849c924562e","categories":["prompt-injection","guardrails"],"citation_count":18,"reviewed":false},{"id":"llmsec-2026-00794","type":"paper","title":"IPIGuard: A Novel Tool Dependency Graph-Based Defense Against Indirect Prompt Injection in LLM Agents","authors":["Hengyu An","Jinghuai Zhang","Tianyu Du","Chunyi Zhou","Qingming Li","Tao Lin","Shouling Ji"],"year":2025,"venue":"Conference on Empirical Methods in Natural Language Processing","abstract":"Large language model (LLM) agents are widely deployed in real-world applications, where they leverage tools to retrieve and manipulate external data for complex tasks. However, when interacting with untrusted data sources (e.g., fetching information from public websites), tool responses may contain injected instructions that covertly influence agent behaviors and lead to malicious outcomes, a threat referred to as Indirect Prompt Injection (IPI). Existing defenses typically rely on advanced prom","url":"https://www.semanticscholar.org/paper/9ddcbbff1e23b7f995fc363ad5123091f8866748","categories":["prompt-injection","agentic-threats"],"citation_count":44,"reviewed":false},{"id":"llmsec-2026-00795","type":"paper","title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","authors":["Hanrong Zhang","Jingyuan Huang","K. Mei","Yifei Yao","Zhenting Wang","Chenlu Zhan","Hongwei Wang","Yongfeng Zhang"],"year":2024,"venue":"International Conference on Learning Representations","abstract":"Although LLM-based agents, powered by Large Language Models (LLMs), can use external tools and memory mechanisms to solve complex real-world tasks, they may also introduce critical security vulnerabilities. However, the existing literature does not comprehensively evaluate attacks and defenses against LLM-based agents. To address this, we introduce Agent Security Bench (ASB), a comprehensive framework designed to formalize, benchmark, and evaluate the attacks and defenses of LLM-based agents, in","url":"https://www.semanticscholar.org/paper/5f4efbe3aae1d8f44ceab1da257ae685d6beb00b","categories":["agentic-threats","benchmarks"],"citation_count":334,"reviewed":false},{"id":"llmsec-2026-00796","type":"paper","title":"To Protect the LLM Agent Against the Prompt Injection Attack with Polymorphic Prompt","authors":["Zhilong Wang","N. Nagaraja","Lan Zhang","Hayretdin Bahşi","Pawan Patil","Peng Liu"],"year":2025,"venue":"2025 55th Annual IEEE/IFIP International Conference on Dependable Systems and Networks - Supplemental Volume (DSN-S)","abstract":"LLM agents are widely used as agents for customer support, content generation, and code assistance. However, they are vulnerable to prompt injection attacks, where adversarial inputs manipulate the model’s behavior. Traditional defenses like input sanitization, guard models, and guardrails are either cumbersome or ineffective. In this paper, we propose a novel, lightweight defense mechanism called Polymorphic Prompt Assembling (PPA), which protects against prompt injection with near-zero overhea","url":"https://www.semanticscholar.org/paper/020ad9b3a8022ac6464cc06b2e80c2a267001e77","categories":["prompt-injection","agentic-threats","guardrails"],"citation_count":18,"reviewed":false},{"id":"llmsec-2026-00797","type":"paper","title":"Attack and defense techniques in large language models: A survey and new perspectives","authors":["Zhiyu Liao","Kang Chen","Y. Lin","Kangkang Li","Yunxuan Liu","Hefeng Chen","Xingwang Huang","Yuanhui Yu"],"year":2025,"venue":"Neural Networks","abstract":"Large Language Models (LLMs) have become central to numerous natural language processing tasks, but their vulnerabilities present significant security and ethical challenges. This systematic survey explores the evolving landscape of attack and defense techniques in LLMs. We classify attacks into adversarial prompt attacks, optimized attacks, model theft, as well as attacks on LLM applications, detailing their mechanisms and implications. Consequently, we analyze defense strategies, such as preve","url":"https://www.semanticscholar.org/paper/5bce864b579b376c028ec40a8fec0f999b005d0e","categories":["model-extraction"],"citation_count":23,"reviewed":false},{"id":"llmsec-2026-00798","type":"paper","title":"Cyb-LLM: A Unified Benchmark for Evaluating LLMs in Cyber Attack & Defense","authors":["Akshitha Segireddy"],"year":2026,"venue":"2026 International Conference on Visual Analytics and Data Visualization (ICVADV)","abstract":"There is a growing use of large language models LLMs in security-related workflows. There is a gap in existing benchmarks for assessing dual-use capabilities of LLMs under safety constraints. We introduce a new, unified benchmark called CybLLM, which encompasses both offensive tasks (e.g., phishing generation, exploit request) and defensive tasks (e.g., secure coding, malware, and triage with built-in safety alignment metrics. Our framework integrates offense, Defense, and utility assessments us","url":"https://www.semanticscholar.org/paper/a8cef9a306f824b558108d45011f675c2244940b","categories":["guardrails","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00799","type":"paper","title":"Securing LLM-based agents against cyberattacks: a comprehensive survey on attack techniques and defense strategies","authors":["Nyashadzashe Tamuka","T. Mathonsi","T. Olwal","Solly Maswikaneng","Tonderai Muchenje","T. Tshilongamulenzhe"],"year":2026,"venue":"Journal of Computer Virology and Hacking Techniques","abstract":"\n Large Language Model (LLM)-based agents integrate various models, including planning loops, memory, tool use, and multi-agent systems, enabling autonomous decision-making through natural-language interfaces. This autonomy also expands the cyberattack surface from model-only failures to agent compromise, where untrusted text can exfiltrate data or trigger malicious actions. This survey presents a taxonomy-driven synthesis of security threats targeting LLM-based agents and a structured review of","url":"https://www.semanticscholar.org/paper/604332431f17ceaefa260471b1792ce48ea7ff16","categories":["agent-architecture","tool-use-security","survey"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00800","type":"paper","title":"SecureGov-Agent: A Governance-Centric Multi-Agent Framework for Privacy-Preserving and Attack-Resilient LLM Agents","authors":["Jinyu Chen","Jixiao Yang","Ziyang Zeng","Zixiao Huang","Jinming Li","Yutong Wang"],"year":2025,"venue":"Proceedings of the 2025 6th International Conference on Computer Science and Management Technology","abstract":"Large Language Model (LLM)-based multi-agent systems have demonstrated remarkable capabilities across diverse applications, yet they face critical security challenges including backdoor attacks, prompt injection, and privacy leakage. Existing defense mechanisms typically address single threat vectors, lacking a unified governance architecture for comprehensive security. We propose SecureGov-Agent, a governance-centric multi-agent framework that introduces a dedicated Governance Agent responsible","url":"https://www.semanticscholar.org/paper/e2e5529df48fe6cf893da483360958b32d829724","categories":["prompt-injection","data-poisoning","membership-inference","agentic-threats","agent-architecture"],"citation_count":13,"reviewed":false},{"id":"llmsec-2026-00801","type":"paper","title":"A Multi-Agent LLM Defense Pipeline Against Prompt Injection Attacks","authors":["S. Hossain","Ruksat Khan Shayoni","Mohd Ruhul Ameen","Akif Islam","M. Mridha","Jungpil Shin"],"year":2025,"venue":"IEEE International WIE Conference on Electrical and Computer Engineering","abstract":"Prompt injection attacks represent a major vulnerability in Large Language Model (LLM) deployments, where malicious instructions embedded in user inputs can override system prompts and induce unintended behaviors. This paper presents a novel multi-agent defense framework that employs specialized LLM agents in coordinated pipelines to detect and neutralize prompt injection attacks in real-time. We evaluate our approach using two distinct architectures: a sequential chain-ofagents pipeline and a h","url":"https://www.semanticscholar.org/paper/f851714285c7a6bfb963da98d67f04df0eafbac9","categories":["prompt-injection","agentic-threats","agent-architecture"],"citation_count":12,"reviewed":false},{"id":"llmsec-2026-00802","type":"paper","title":"Balancing Security and Performance in LLM Agents: Spotlight-Guard, a Layered Defense Against Indirect Prompt Injection","authors":["Doygun Demirol","Murat Aydoğan"],"year":2026,"venue":"Applied Sciences","abstract":"Large Language Model (LLM)-based agents automate complex tasks by integrating external tools such as web browsers, e-mail clients, file readers, and APIs, but this same integration exposes them to indirect prompt injection (IPI) attacks, in which malicious instructions hidden in tool content hijack the agent. A central but often overlooked question is how defending against such attacks affects the LLM and its own task performance and computational efficiency. In this study, we design a comprehen","url":"https://www.semanticscholar.org/paper/bdb097dcf666e8ed89f66d2c7319d47f524fc614","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00803","type":"paper","title":"Play Guessing Game with LLM: Indirect Jailbreak Attack with Implicit Clues","authors":["Zhiyuan Chang","Mingyang Li","Yi Liu","Junjie Wang","Qing Wang","Yang Liu"],"year":2024,"venue":"Annual Meeting of the Association for Computational Linguistics","abstract":"With the development of LLMs, the security threats of LLMs are getting more and more attention. Numerous jailbreak attacks have been proposed to assess the security defense of LLMs. Current jailbreak attacks primarily utilize scenario camouflage techniques. However their explicitly mention of malicious intent will be easily recognized and defended by LLMs. In this paper, we propose an indirect jailbreak attack approach, Puzzler, which can bypass the LLM's defense strategy and obtain malicious re","url":"https://www.semanticscholar.org/paper/490e815b3be11ba97631783d9ae946b8f8517fd6","categories":["jailbreaking"],"citation_count":83,"reviewed":false},{"id":"llmsec-2026-00804","type":"paper","title":"SoK: Systematizing LLM Prompt Security: Taxonomies, Datasets, and Unified Evaluation of Attacks and Defenses","authors":["Hanbin Hong","Shuangqiao Wu","Shuya Feng","Nima Naderloui","Shenao Yan","Jingyue Zhang","Ali Arastehfard","Heqing Huang","Yuan Hong"],"year":2025,"abstract":"Large Language Models (LLMs) are increasingly used as interfaces to information, code, and real-world services, making prompt-level security failures a practical concern. Although jailbreak attacks, defenses, datasets, and automated judgers have advanced rapidly, evaluation remains fragmented across threat models, access assumptions, cost budgets, datasets, and success criteria. This makes reported attack success rates and defense gains hard to compare. This SoK systematizes LLM prompt security ","url":"https://www.semanticscholar.org/paper/9f8a3c6f1fca4aeb8364bdb767b2e14e92376d5d","categories":["jailbreaking","threat-modeling"],"citation_count":6,"reviewed":false},{"id":"llmsec-2026-00805","type":"paper","title":"Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks","authors":["Zimo Ji","Xunguang Wang","Zongjie Li","Pingchuan Ma","Yudong Gao","Daoyuan Wu","Xincheng Yan","Tian Tian","Shuai Wang"],"year":2025,"venue":"arXiv.org","abstract":"Large Language Model (LLM)-based agents with function-calling capabilities are increasingly deployed, but remain vulnerable to Indirect Prompt Injection (IPI) attacks that hijack their tool calls. In response, numerous IPI-centric defense frameworks have emerged. However, these defenses are fragmented, lacking a unified taxonomy and comprehensive evaluation. In this Systematization of Knowledge (SoK), we present the first comprehensive analysis of IPI-centric defense frameworks. We introduce a c","url":"https://www.semanticscholar.org/paper/66c8e83ed0fc215dc8be0e44715a692ef27f005b","categories":["prompt-injection","agentic-threats","survey"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-00806","type":"paper","title":"Trust in LLM-controlled Robotics: a Survey of Security Threats, Defenses and Challenges","authors":["Xinyu Huang","B. ShyamKarthickV","Taozhao Chen","Mitch Bryson","Thomas L. Chaffey","Huaming Chen","Kim-Kwang Raymond Choo","Ian R. Manchester","S. Karthick"],"year":2025,"venue":"arXiv.org","abstract":"The integration of Large Language Models (LLMs) into robotics has revolutionized their ability to interpret complex human commands and execute sophisticated tasks. However, such paradigm shift introduces critical security vulnerabilities stemming from the''embodiment gap'', a discord between the LLM's abstract reasoning and the physical, context-dependent nature of robotics. While security for text-based LLMs is an active area of research, existing solutions are often insufficient to address the","url":"https://www.semanticscholar.org/paper/71f30c8b6aca4dbd4a81302ac30335421b1a8688","categories":["survey"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-00807","type":"paper","title":"Recent Advances in Attack and Defense Approaches of Large Language Models","authors":["Jing Cui","Yishi Xu","Zhewei Huang","Shuchang Zhou","Jianbin Jiao","Junge Zhang"],"year":2024,"venue":"arXiv.org","abstract":"Large Language Models (LLMs) have revolutionized artificial intelligence and machine learning through their advanced text processing and generating capabilities. However, their widespread deployment has raised significant safety and reliability concerns. Established vulnerabilities in deep neural networks, coupled with emerging threat models, may compromise security evaluations and create a false sense of security. Given the extensive research in the field of LLM security, we believe that summar","url":"https://www.semanticscholar.org/paper/3d7cc47f10a1b55e3c7af24bf43f7f9206fcda4e","categories":["threat-modeling"],"citation_count":9,"reviewed":false},{"id":"llmsec-2026-00808","type":"paper","title":"Stable Agentic Control: Tool-Mediated LLM Architecture for Autonomous Cyber Defense","authors":["Kerri Prinos","Lilianne Brush","C. Denton","Zhanqiang Wang","Joshua Knox","Snehal S. Antani","A. Foltz","Amy Villasenor"],"year":2026,"venue":"arXiv.org","abstract":"Agentic systems involved in high-stake decision-making under adversarial pressure need formal guarantees not offered by existing approaches. Motivated by the operational needs of security operations centers (SOCs) that must configure endpoint detection and response (EDR) policies under adversarial pressure, we present a tool-mediated architecture: LLM agents use deterministic tools (Stackelberg best-response, Bayesian observer updates, attack-graph primitives) and select from finite action catal","url":"https://www.semanticscholar.org/paper/7ca2bbdc56f4102275222deee1f3a60eb34b486a","categories":["agentic-threats"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-00809","type":"paper","title":"SafeHarness: Lifecycle-Integrated Security Architecture for LLM-based Agent Deployment","authors":["Xixun Lin","Yang Liu","Yancheng Chen","Yongxuan Wu","Yucheng Ning","Yilong Liu","Nan Sun","Shunhong Zhang","Bin Chong","Chuan Zhou","Yanan Cao","Li Guo"],"year":2026,"venue":"arXiv.org","abstract":"The performance of large language model (LLM) agents depends critically on the execution harness, the system layer that orchestrates tool use, context management, and state persistence. Yet this same architectural centrality makes the harness a high-value attack surface: a single compromise at the harness level can cascade through the entire execution pipeline. We observe that existing security approaches suffer from structural mismatch, leaving them blind to harness-internal state and unable to","url":"https://www.semanticscholar.org/paper/dc80f1113a0087e81f47cc746590b802c8ef6013","categories":["tool-use-security"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-00810","type":"paper","title":"SAIGuard: Communication-State Simulation for Proactive Defense of LLM Multi-Agent Systems","authors":["Ruxue Shi","Yili Wang","Mengnan Du","Qinggang Zhang","Rui Miao","Yixin Liu","Xin Wang"],"year":2026,"abstract":"LLM-based multi-agent systems (MAS) solve complex tasks through inter-agent collaboration, but their communication-driven nature also allows security risks to spread across agents and trigger system-wide failures. Existing MAS defenses mainly follow a reactive paradigm after execution by detecting and isolating harmful agents, which may cause irreversible damage and degrade collaborative utility. To address this, we propose a proactive defense framework for MAS security, namely a Simulation-awar","url":"https://www.semanticscholar.org/paper/a96d1118b026d24af509a0e6ecd47b710bf0f138","categories":["agent-architecture","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00811","type":"paper","title":"Cybersecurity AI: A Game-Theoretic AI for Guiding Attack and Defense","authors":["V. Vilches","Mar'ia Sanz-G'omez","Francesco Balassone","Stefan Rass","Lidia Salas Espejo","Benjamin Jablonski","Luis Javier Navarrete-Lozano","Maite del Mundo de Torres","Cristóbal R. J. Veas Chavez"],"year":2026,"venue":"arXiv.org","abstract":"AI-driven penetration testing now executes thousands of actions per hour but still lacks the strategic intuition humans apply in competitive security. To build cybersecurity superintelligence --Cybersecurity AI exceeding best human capability-such strategic intuition must be embedded into agentic reasoning processes. We present Generative Cut-the-Rope (G-CTR), a game-theoretic guidance layer that extracts attack graphs from agent's context, computes Nash equilibria with effort-aware scoring, and","url":"https://www.semanticscholar.org/paper/3316e5185e695a00f15eb372c05c6cab5ef3fbfa","categories":["agentic-threats"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-00812","type":"paper","title":"BlindGuard: Safeguarding LLM-based Multi-Agent Systems under Unknown Attacks","authors":["Rui Miao","Yixin Liu","Yili Wang","Xu Shen","Yue Tan","Yiwei Dai","Shirui Pan","Xin Wang"],"year":2025,"venue":"Volume 1","abstract":"The security of LLM-based multi-agent systems (MAS) is critically threatened by propagation vulnerability, where malicious agents can distort collective decision-making through inter-agent message interactions. While existing supervised defense methods demonstrate promising performance, they may be impractical in real-world scenarios due to their heavy reliance on labeled malicious agents to train a supervised malicious detection model. To enable practical and generalizable MAS defenses, in this","url":"https://www.semanticscholar.org/paper/f583a040e8324257a993935122c7f1e274bdb20a","categories":["guardrails","agent-architecture"],"citation_count":39,"reviewed":false},{"id":"llmsec-2026-00813","type":"paper","title":"A Formal Security Framework for MCP-Based AI Agents: Threat Taxonomy, Verification Models, and Defense Mechanisms","authors":["Nirajan Acharya","Gaurav Kumar Gupta"],"year":2026,"venue":"arXiv.org","abstract":"The Model Context Protocol (MCP), introduced by Anthropic in November 2024 and now governed by the Linux Foundation's Agentic AI Foundation, has rapidly become the de facto standard for connecting large language model (LLM)-based agents to external tools and data sources, with over 97 million monthly SDK downloads and more than 177000 registered tools. However, this explosive adoption has exposed a critical gap: the absence of a unified, formal security framework capable of systematically charac","url":"https://www.semanticscholar.org/paper/f28a78235a0ec6bc74135a46688a5008a38044d9","categories":["agentic-threats"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-00814","type":"paper","title":"MultiPhishGuard: An LLM-based Multi-Agent System for Phishing Email Detection","authors":["Yinuo Xue","Eric Spero","Yun Sing Koh","Giovanni Russello"],"year":2025,"venue":"arXiv.org","url":"https://www.semanticscholar.org/paper/f0cf54a86f7eb31e2bc42e328a603ebb84a7369f","categories":["agent-architecture"],"citation_count":19,"reviewed":false},{"id":"llmsec-2026-00815","type":"paper","title":"From LLMs to MLLMs to Agents: A Survey of Emerging Paradigms in Jailbreak Attacks and Defenses within LLM Ecosystem","authors":["Yanxu Mao","Tiehan Cui","Peipei Liu","Datao You","Hongsong Zhu"],"year":2025,"venue":"arXiv.org","abstract":"Large language models (LLMs) are rapidly evolving from single-modal systems to multimodal LLMs and intelligent agents, significantly expanding their capabilities while introducing increasingly severe security risks. This paper presents a systematic survey of the growing complexity of jailbreak attacks and corresponding defense mechanisms within the expanding LLM ecosystem. We first trace the developmental trajectory from LLMs to MLLMs and Agents, highlighting the core security challenges emergin","url":"https://www.semanticscholar.org/paper/ccfabe9f33f11bd1fbc4ac2bf219fc29cc5fa96d","categories":["jailbreaking","survey","threat-modeling"],"citation_count":16,"reviewed":false},{"id":"llmsec-2026-00816","type":"paper","title":"Signed-Prompt: A New Approach to Prevent Prompt Injection Attacks Against LLM-Integrated Applications","authors":["X. Suo"],"year":2024,"venue":"AIP Conference Proceedings","abstract":"The critical challenge of prompt injection attacks in Large Language Models (LLMs) integrated applications, a growing concern in the Artificial Intelligence (AI) field. Such attacks, which manipulate LLMs through natural language inputs, pose a significant threat to the security of these applications. Traditional defense strategies, including output and input filtering, as well as delimiter use, have proven inadequate. This paper introduces the 'Signed-Prompt' method as a novel solution. The stu","url":"https://www.semanticscholar.org/paper/2742c3d77c4aa6d023b7dfc77984ad10e6aae274","categories":["prompt-injection","input-filtering"],"citation_count":82,"reviewed":false},{"id":"llmsec-2026-00817","type":"paper","title":"STACK: Adversarial Attacks on LLM Safeguard Pipelines","authors":["Ian R. McKenzie","O. Hollinsworth","Tom Tseng","Xander Davies","Stephen Casper","A. D. Tucker","Robert Kirk","Adam Gleave"],"year":2025,"venue":"AAAI Conference on Artificial Intelligence","abstract":"Frontier AI developers are relying on layers of safeguards to protect against catastrophic misuse of AI systems. Anthropic guards their latest Claude 4 Opus model using one such defense pipeline, and other frontier developers including Google DeepMind and OpenAI pledge to soon deploy similar defenses. However, the security of such pipelines is unclear, with limited prior work evaluating or attacking these pipelines. We address this gap by developing and red-teaming an open-source defense pipelin","url":"https://www.semanticscholar.org/paper/59db5498605113c8b2567219a5041ff08f7a4587","categories":["adversarial-examples","guardrails","red-teaming"],"citation_count":13,"reviewed":false},{"id":"llmsec-2026-00818","type":"paper","title":"Does Fixing Break Security? An Empirical Study of Security Degradation in Iterative LLM-Driven Infrastructure-as-Code Repair","authors":["Benjamin Agyekum","Fabio Santos"],"year":2026,"abstract":"Background: Iterative feedback loops are the dominant paradigm for improving LLM-generated Infrastructure-as-Code (IaC): validators such as Checkov and terraform validate feed error signals back for successive repair attempts. Prior work reports cumulative-best metrics, which are non-decreasing by construction, so the raw per-iteration security trajectory has never been examined for IaC. Aims: We study security regression (a previously-passing CIS Benchmark check that fails after a repair iterat","url":"https://arxiv.org/abs/2608.13404","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00819","type":"paper","title":"ToolHazard: Scaling Adversarial Environments for Security Evaluation and Alignment of LLM-based Agents","authors":["Yutao Mou","Pengfei Yang","Zhe Yin","Zhangchi Xue","Xiaotian Luan","Dingyao Yu","Tong Zhang","Shikun Zhang","Wei Ye"],"year":2026,"abstract":"Large language model (LLM) agents integrated with external tools are vulnerable to indirect prompt injections embedded in environmental states. However, existing studies largely rely on manually implemented or reused environments, stochastic LLM-based tool simulation, and predefined injection locations, limiting scalable security research across broader domains. To bridge this gap, we propose **ToolHazard**, a scalable adversarial environment synthesis framework that reduces human engineering an","url":"https://arxiv.org/abs/2608.11878","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00820","type":"paper","title":"On Understanding, Identifying, and Mitigating Vulnerabilities in Agentic Large Language Models","authors":["Md Jafrin Hossain","Mohammad Arif Hossain","Nirwan Ansari"],"year":2026,"abstract":"Large Language Models (LLMs) have undergone a shift from stateless conversational interfaces to autonomous agents capable of multi-step planning, tool invocation, code execution, and maintaining persistent memory. When these agents operate with real-world privileges---calling APIs, modifying files, and querying databases---a compromised reasoning step can trigger unauthorized data access, irreversible state changes, or cascading failures, yet the security research community has not kept pace. To","url":"https://arxiv.org/abs/2608.10530","categories":["agentic-threats","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00821","type":"paper","title":"From Prompt Injection to Web Exploitation: Revisiting Classic Vulnerabilities in LLM-Integrated Applications","authors":["Spiros Tsigkopoulos","Christoforos Ntantogian"],"year":2026,"abstract":"Large Language Models are increasingly integrated into web applications through chatbots, tool-calling pipelines, and agentic workflows. In these systems, user input may influence not only generated text, but also backend actions such as database queries, HTTP requests, file operations, template rendering, or API calls. This paper introduces LLM-mediated web attacks, a class of attacks in which attacker-controlled input is transformed by an LLM-integrated application and then reaches traditional","url":"https://arxiv.org/abs/2608.10281","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00822","type":"paper","title":"Generating Attacks for LLMs with GFlowNets","authors":["Berkay Ozcam","Irem Onen","Mehmet Fatih Amasyali","Emin Islam Tatli"],"year":2026,"abstract":"The rapid advancement of Large Language Models (LLMs) has facilitated their ubiquitous integration into various domains, leading to widespread adoption. However, this escalating trend has introduced significant security vulnerabilities, necessitating the identification and mitigation of flaws arising from malicious exploitation. Red teaming assessments, conducted to evaluate model robustness through diverse adversarial inputs, are essential for exposing security risks and implementing countermea","url":"https://arxiv.org/abs/2608.10171","categories":["red-teaming","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00823","type":"paper","title":"Pragmatic Attack Surface: Vulnerabilities of Implicit Context in Large Language Models","authors":["Bocheng Chen","Han Zi","Roucheng Ou","Yawei Liu","Minyue Chen","Zimo Qi","Rongrong Wang","Guangliang Liu"],"year":2026,"abstract":"In the era of large language models (LLMs), attackers often manipulate natural language to elicit unsafe or harmful outputs, creating a new natural language attack surface unique to LLM-based systems, where attacks directly exploit explicit linguistic cues in user prompts to bypass the safety mechanism of LLMs. However, such attacks can often be mitigated by existing safety alignment algorithms. On the other hand, human language is inherently grounded in pragmatics, necessitating typical context","url":"https://arxiv.org/abs/2608.09551","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00824","type":"paper","title":"Gaming Without an Attacker: Benchmark Fingerprinting in LLM-Driven Search Under Selection Pressure","authors":["Víctor Gallego"],"year":2026,"abstract":"Benchmarks for systems that are optimized against the evaluation signal measure something different from what they claim. We document this concretely in two GPU-kernel-optimization suites with held-out generalization gates: Metal-Sci (10 scientific-compute tasks) and Metal-ZK (12 zero-knowledge/cryptographic tasks), in which three frontier LLMs (Opus 4.7, Gemini 3.1 Pro, GPT-5.5) propose Metal kernels inside a $(1{+}1)$ evolutionary loop with rich feedback. Although no model is prompted to act a","url":"https://arxiv.org/abs/2608.08722","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00825","type":"paper","title":"Measuring the Wrong Thing: Internal Harmfulness Scores Anti-Rank Successful Jailbreaks","authors":["Mingyu Luo","Ming Deng","Zilang Qiu","Yiming Cheng","Ci Tao","Xue Tan","Sijin Sun","Yangfu Li","Ping Chen","Jun Dai","Xiaoyan Sun"],"year":2026,"abstract":"Internal safety scores judge a prompt before any text is generated, and they are validated by how well they separate harmful prompts from benign ones. That separation is then read as evidence that the score will also catch the attacks that succeed. Harmful intent is a property of the prompt. Jailbreak success is an outcome produced later by a particular target model, decoding policy, and judge. A filter tuned on a score that measures the wrong quantity spends its false positive budget on attacks","url":"https://arxiv.org/abs/2608.09624","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00826","type":"paper","title":"Dual-Adversarial Safety Alignment: Cultivating Intrinsic Threat Comprehension in LRMs","authors":["Hongli Shen","Shaopeng Fu","Qinbo Zhang","Jian Li","Di Wang"],"year":2026,"abstract":"Large reasoning models (LRMs) achieve remarkable success on complex tasks but remain vulnerable to harmful prompts that induce unsafe outputs. Recent methods align LRMs using direct refusals or safety rationales, yet often focus on prompt patterns rather than intrinsic attack mechanisms. As a result, these pattern-centric alignments struggle to generalize across diverse jailbreaks, compromising adversarial robustness and reasoning utility. We propose AdvSafe, a dual-adversarial framework that en","url":"https://arxiv.org/abs/2608.09542","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00827","type":"paper","title":"Toward Metacognitive One-Shot Indirect Prompt Injection: Strategy Abstraction Via Outcome-Conditioned Reflection","authors":["Sihan Hou","Xinmeng Hou","Zhijun Zhang","Zehao Wang","Xuhong Ren","Sibo Qin","Kuntharrgyal Khysru","Qing Guo"],"year":2026,"abstract":"Tool-using large language model (LLM) agents are vulnerable to indirect prompt injection (IPI), in which malicious instructions embedded in external observations manipulate subsequent agent decisions and actions. Most existing adaptive attacks rely on repeatedly querying and refining against the target agent, whereas realistic attackers may have only a single opportunity to interact with an unknown target agent. We propose SAVOR (Strategy Abstraction Via Outcome-Conditioned Reflection), which sh","url":"https://arxiv.org/abs/2608.08795","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00828","type":"paper","title":"Defending Retrieval-Augmented Intrusion Detection Against Knowledge Poisoning and Prompt Injection","authors":["Kaysarul Anas Apurba","Md. Hasibul Hasan","Mahedee Zaman Moon","Sk. Md. Mizanur Rahman","Atsuo Inomata"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) enables large language models to classify network flows and generate human-readable incident reports by retrieving semantically similar historical traffic from a vector knowledge base. However, the retrieval layer introduces vulnerabilities to knowledge poisoning and prompt-injection attacks. We present RAG-IDS, a three-tier multi-agent intrusion detection framework with a retrieval-boundary defense combining soft trust scoring, label-embedding consistency ch","url":"https://arxiv.org/abs/2608.08100","categories":["prompt-injection","data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00829","type":"paper","title":"BASIS: Breach-Aware Selective Prompt Injection Shielding with Prefill Attention Probes","authors":["Laiqiao Qin","Tianqing Zhu","Longxiang Gao","Wanlei Zhou"],"year":2026,"abstract":"Prompt injection is a critical security threat in large language model (LLM) applications, where attackers hijack model behavior by embedding malicious instructions in user or external data. Existing detection methods only detect the presence of injection and refuse to respond upon detection, overlooking the fact that for many modern aligned models, well-crafted instructions can resist most injection attacks. This means that the injection robustness varies significantly across instructions and m","url":"https://arxiv.org/abs/2608.08027","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00830","type":"paper","title":"Capability-Routed Guard: Defending Large Reasoning Models Against Reasoning-Centric Jailbreaks","authors":["Yiyong Liu","Yixin Wu","Jun Sakuma"],"year":2026,"abstract":"Large reasoning models (LRMs) expose a new safety failure mode: adversarial prompts can manipulate reasoning context, task decomposition, or capability interpretation so that harmful objectives are processed as legitimate reasoning steps. Existing safeguards, including safety reminders, external classifiers, and self-checking wrappers, are often brittle because they either inspect the adversarial prompt directly or ask the target model to perform additional safety reasoning on the same surface t","url":"https://arxiv.org/abs/2608.07892","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00831","type":"paper","title":"The Anatomy of a Prompt Injection: A Component Model for Structured Analysis","authors":["Jeremy McHugh"],"year":2026,"abstract":"Four years after prompt injection was first identified in 2022, attacks are still predominantly documented as verbatim strings rather than structured exploits, despite advancing agent capabilities and threat actors embedding injections to subvert AI-assisted security analysis. This paper formalizes the structure of prompt-injection artifacts, enabling defenders, red teamers, and cyber threat intelligence (CTI) teams to label, compare, and mutate attacks without relying on fragile string matching","url":"https://arxiv.org/abs/2608.07808","categories":["prompt-injection","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00832","type":"paper","title":"StepJack: Benchmarking Computer-Use Agent Safety Against Multi-Step Indirect Prompt Injection","authors":["Zhuoxin Zhan","Akbar Rafiey","Avery Ma","Leila Pishdad","Layla El Asri"],"year":2026,"abstract":"Computer-use agents (CUAs) face a growing threat from indirect prompt injection, where adversarial instructions are planted in the environment such as web pages. In this paper, we introduce multi-step indirect prompt injection, a new attack class against CUAs in which the adversarial goal is decomposed into multiple innocuous-looking sub-steps and distributed across a chain of pages referenced along the agent's navigation path. We develop a pipeline to automatically decompose an adversarial goal","url":"https://arxiv.org/abs/2608.06477","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00833","type":"paper","title":"Evolving Safety Landscape of Multi-modal Large Language Models: A Survey of Emerging Threats and Safeguards","authors":["Xi Li","Shu Zhao","Xiaohan Zou","Fei Zhao","Fuxiao Liu","Yusen Zhang","Cheng Han","Yushun Dong","Jiaqi Wang"],"year":2026,"abstract":"Multi-modal large language models (MLLMs) integrate heterogeneous modalities through modality alignment and fusion, enabling stronger understanding and reasoning. However, this architectural shift reshapes the safety landscape of machine learning. Increased model complexity and cross-modal interactions give rise to novel threats, including compromised modality integration, modality misalignment, and fused safety risks, reflecting shifts in threat modeling beyond uni-modal assumptions. These shif","url":"https://arxiv.org/abs/2608.07535","categories":["guardrails","survey","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00834","type":"paper","title":"Static analysis-guided agentic AI translation enables Rust as a full stack bioinformatics language","authors":["Johan Henriksson"],"year":2026,"abstract":"The field of bioinformatics struggles with legacy code - old code that is commonly used but may no longer have a maintainer, or may be written in an now-unfamiliar language (e.g. Perl, Fortran). This incurs maintenance cost (technical debt), but dynamically typed languages also negatively impacts the environment and fail to make use of modern hardware. Legacy code may also have security or safety problems that make it unsuited for use in clinical settings. Here we show that agentic AI, combined ","url":"https://arxiv.org/abs/2608.13029","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00835","type":"paper","title":"Rethinking Agent Security as a Networking Problem","authors":["Van Tran","Taveesh Sharma","Tajveer Singh Dhesi","Nick Feamster"],"year":2026,"abstract":"AI agents are rapidly becoming more capable and widely deployed, promising substantial gains in productivity and enabling new classes of applications. However, their growing autonomy also introduces significant privacy and security risks. Existing defenses are predominantly agent-centric, relying on the agent itself to detect threats and enforce privacy and security policies. This approach is fundamentally limited because it entrusts policy enforcement to AI agents whose LLM-driven behavior is i","url":"https://arxiv.org/abs/2608.12172","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00836","type":"paper","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","authors":["Jeremy Spence","Nicholas Assaderaghi","Jinhao Zhu","Nikil Ravi","Raluca Ada Popa","Guannan Wei","Yangruibo Ding","Zhuo Zhang"],"year":2026,"abstract":"AI agents are rapidly improving in cybersecurity capabilities when the source code is available for analysis, yet much of the software most consequential to cybersecurity, including malware, firmware, and proprietary applications, is available only as binaries. Analyzing such software requires reverse engineering(RE): recovering program semantics before the analysis can be meaningfully performed. However, evaluating agentic RE poses a fundamental challenge: benchmark instances must be unseen as ","url":"https://arxiv.org/abs/2608.11469","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00837","type":"paper","title":"Agent Safety Should Be a Runtime Contract","authors":["Albus W. Ng","Yi Han","Jusheng Zhang","Wenhao Wang"],"year":2026,"abstract":"The dominant paradigm treats AI safety as a property to be instilled during model training via RLHF, DPO, or Constitutional AI. We argue this is structurally insufficient for autonomous agents that execute code, mutate files, send messages, and modify databases. Agent safety should be a runtime contract enforced by the harness, and the contract has two complementary faces. The preventive face blocks dangerous actions before they happen via sandboxes, permission gates, output filters, and traject","url":"https://arxiv.org/abs/2608.11274","categories":["output-moderation","guardrails","sandboxing-isolation","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00838","type":"paper","title":"Toward a Theory of Value in AI Alignment","authors":["Andrew Smart","Shazeda Ahmed","Jackie Kay","Jimmy Tobin","Kris Shrishak","Abeba Birhane"],"year":2026,"abstract":"Can AI systems be aligned to human values? The popularization of large language models (LLMs) and multi-modal foundation models has seen a rise in harms spanning from toxic speech and hallucinations to AI agents executing unauthorized actions. Within the field of AI safety, these harmful instances are often framed as the alignment problem, or of models being misaligned with human values. Researchers have responded by pursuing applied and theoretical AI value alignment efforts, often without spec","url":"https://arxiv.org/abs/2608.10327","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00839","type":"paper","title":"Multi-Agent AI Safety as an Institutional Design Problem","authors":["Abdullah X"],"year":2026,"abstract":"AI agents increasingly work inside systems that govern how they delegate tasks, move information, execute actions, and use shared resources. Recent work already shows that deployment rules can change collective behavior. Here we ask which parts of an AI institution produce safety and how they do it. This is the first paper from POLIS, an ongoing research programme studying algorithmic institutions for multi-agent systems. We report a frozen 5,280-episode study suite. The main pre-specified deleg","url":"https://arxiv.org/abs/2608.09828","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00840","type":"paper","title":"Not an A11y: How Android Accessibility Exposes Mobile AI Agents to Indirect Prompt Injection","authors":["Rahul Deivasigamani","Sayeda Faatin Alvi","Derqui Andrea","Kaushal Punjabi","Stjepan Picek"],"year":2026,"abstract":"The rise of autonomous AI agents represents a major paradigm shift in how users interact with mobile devices. Frameworks such as MobileRun and Mobile-Use can autonomously navigate Android applications and execute complex multi-step tasks. To interpret user interfaces, these frameworks rely primarily on Android accessibility (A11y) trees and secondarily on visual screenshots. In this paper, we demonstrate that this architectural dependence on unsanitized accessibility metadata, together with visu","url":"https://arxiv.org/abs/2608.08939","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00841","type":"paper","title":"CyberAGENTS: Structured Autonomy for Agentic Gamified Learning in Cybersecurity","authors":["Ivan Hornung","Deepthi Marasinghe Arachchige","Tharindu Kumarage","Garima Agrawal","Yuli Deng","Ying-Chih Chen","Huan Liu"],"year":2026,"abstract":"Gamification is especially effective in learning domains requiring active problem-solving and iterative skill-building, such as cybersecurity education. Generative AI agents offer a path to delivering such experiences adaptively at scale, but introduce well-documented risks in educational settings: inconsistent behavior, hallucinated reasoning, and misalignment with pedagogical frameworks. Grounding these systems in learning science is therefore essential. We present \\model, an agentic framework","url":"https://arxiv.org/abs/2608.07965","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00842","type":"paper","title":"NiyamAI - An Intent-Bound AI Agent with Cryptographically Verifiable Guardrails using Zero-Knowledge Proofs","authors":["Aditya Katkar","Om Karkele","Kartik Mandhane","Manisha More","Yash Kashid"],"year":2026,"abstract":"Giving an AI agent the ability to send emails, query databases, or execute commands is useful--until the agent is tricked into doing something it shouldn't. Prompt injection, hallucinated reasoning, and unsafe tool calls form the primary attack surface for autonomous LLM agents. Existing defenses rely on software checks like system prompts or policy filters running on the same machine the attacker targets, offering no verifiable proof of execution. We introduce Niyam-AI, a framework that makes s","url":"https://arxiv.org/abs/2608.07167","categories":["prompt-injection","agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00843","type":"paper","title":"LLM Security Defense Using Unsupervised Learning and Deep Reinforcement Learning","authors":["Leilei Wang","Hongying Li"],"year":2026,"venue":"2026 IEEE 3rd International Conference on Computer Vision and Deep Learning (DLCV)","abstract":"With the widespread deployment of large language models (LLMs) in intelligent systems, security threats such as prompt injection, jailbreaking, data poisoning, and hidden backdoor attacks have become increasingly severe. Traditional rule-based filtering and static detection methods are unable to capture implicit semantic attacks and hidden-space anomalies, and lack adaptability to evolving adversarial behaviors. To address these challenges,, unsupervised feature learning, deep reinforcement lear","url":"https://www.semanticscholar.org/paper/26c28b6a8e743b6d6f8ff199fb2217b5d7fa9f34","categories":["prompt-injection","jailbreaking","data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00844","type":"paper","title":"Benchmarking the Effectiveness of AI-Driven Red Teaming Across Safety-Aligned Language Models","authors":["Joyce Malicha","Kamrul Hasan"],"year":2026,"venue":"2026 IEEE 2nd International Conference on Secure IoT, Assured and Trusted Computing (SATC)","abstract":"Large language models (LLMs) have advanced rapidly, yet even safety-aligned models remain vulnerable to adversarial prompts that bypass safeguards and induce harmful outputs. Conventional red teaming methods, including static testing and gradient-based attacks, are limited by either insufficient adaptability or restrictive access assumptions. In this work, we benchmark an autonomous AI-driven evolutionary red teaming system that iteratively generates, evaluates, and refines prompts to uncover sa","url":"https://www.semanticscholar.org/paper/570f0f9f9c7e3b686aa1dc4e97d74ceabe30d149","categories":["guardrails","red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00845","type":"paper","title":"A red teaming framework for large language models: a case study on faithfulness evaluation","authors":["Abrar Alotaibi","Raed Mughus","Moataz Ahmed"],"year":2026,"venue":"Software quality journal","abstract":"Large language models (LLMs) have demonstrated remarkable performance across a wide range of natural language processing tasks, yet their deployment in high-stakes applications has raised critical concerns regarding reliability, safety, and response trustworthiness. In this paper, we present a red teaming framework that systematically uncovers vulnerabilities in LLM outputs. Our approach employs a novel multi-role architecture comprising a target, attackers, and jury models. The attackers genera","url":"https://www.semanticscholar.org/paper/f268fa307779519fcd044adf0b3bc5b8f828611a","categories":["red-teaming"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00846","type":"paper","title":"What AI Red-Team Evaluations Can and Cannot Prove","authors":["Bandana Kaur"],"year":2026,"abstract":"Red-team evaluations of AI models support some claims and not others, and the boundary between the two is calculable rather than merely a matter of judgment. We define the evidential ceiling of an evaluation as the largest factor by which one result can move belief under a fixed testing budget, derive it in closed form for the benchmark null result, and use it to locate that boundary exactly. We find that above a calculable harm rate, a benchmark of modest size certifies a category to a stated e","url":"https://www.semanticscholar.org/paper/5ac1a19d9e04829701552cea010f3dbdd72e65f3","categories":["red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00847","type":"paper","title":"REDAgentBench: Executable Red Teaming and Faithful Measurement of LLM Agent Systems","authors":["Zixing Chen","Xingyuan Liu","Jie Zhu","Huaixia Dou","Shuo Jiang","Junhui Li","Lifan Guo","Feng Chen","Chi Zhang"],"year":2026,"abstract":"Large language model (LLM) agents combine language-based reasoning with external tools to perform complex tasks. Adversarial inputs can exploit interactions between the agent and its environment, causing the agent to violate safety policies during execution. Yet existing evaluations often reduce agent safety to a single attack success rate (ASR), collapsing exposure, execution, observation, and adjudication and potentially conflating actual violations with evidence visibility. We introduce REDAg","url":"https://www.semanticscholar.org/paper/4529381cbe510e8c5911b678654a62b46b3a1327","categories":["agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00848","type":"paper","title":"AttaX-Multimodal: End-to-End Evaluation of Multimodal AI Safety with Threatscore and Resiliencescore","authors":["Rahul Karne"],"year":2026,"venue":"2026 International Conference on Connected Intelligence for Industrial Applications (CI2A)","abstract":"Currently, there is no single benchmark that can be used to measure the safety and resilience of multimodal AI assistants when subjected to malicious attacks. To fill this gap, we have created AttaX-Multimodal, a comprehensive benchmark of multimodal AI assistant safety that tests text, image, audio, and video-based AI assistants, as well as cross-modal attack chains and tool use attack methods. The contribution of this work includes a large-scale dataset of multimodal adversarial examples and a","url":"https://www.semanticscholar.org/paper/62fef5fdd0b48cd8d9acaab2d40663825eb1a98f","categories":["adversarial-examples","benchmarks","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00849","type":"paper","title":"Evaluating the safety of large language models in healthcare and dentistry: adversarial testing approaches","authors":["F. Umer","Muhammad Muthar Shaikh","Absar Ur Rahman"],"year":2026,"venue":"BDJ Open","abstract":"The emergence of large language models (LLMs) provides new avenues for clinical support in healthcare and dentistry. However, these models often exhibit unpredictable behaviours when challenged by adversarial or misleading inputs. Recent data indicate that nearly 20% of LLM outputs contain safety risks or biases, necessitating rigorous evaluation prior to clinical use. This review examines AI red teaming, a systematic approach for identifying system vulnerabilities through simulated attacks. It ","url":"https://www.semanticscholar.org/paper/1701841a90c671a0ac638bf3e721d010046a6488","categories":["red-teaming"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-00850","type":"paper","title":"From Noise to Signal: Improving Security Log Anomaly Detection Using LLMs with Endpoint-Specific Logs","authors":["Christopher Henshaw","Gour Karmakar"],"year":2026,"abstract":"Existing approaches to anomalous behaviour log detection, such as Wazuh rely primarily on predefined detection rules, while statistical anomaly detection approaches such as OpenSearch identify deviations from previously observed behavioural patterns. Recent research has investigated LLMs for log anomaly detection because of their ability to interpret semantic and contextual information. However, LLM-based approaches can be affected by prompt construction, noisy log data, and reliance on generic ","url":"https://arxiv.org/abs/2608.19938","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00851","type":"paper","title":"CompoSkill: Compositional Skill Chain Attacks from Individually Scanner-Passing LLM Agent Skills","authors":["Mingxiao Liu","Zhoumian Jiang","Jianan Ma","Jian Zhang","Jialuo Chen","Xinhao Deng","Zhen Wang"],"year":2026,"abstract":"Autonomous AI agents tackling Long Horizon Tasks depend on marketplace skills that are certified one at a time: a scanner returns a safety verdict for each skill and declares the ecosystem safe if every package passes. We show that this assumption fails under skill composition. A skill may pass the per-skill scanner individually yet participate in a risky composition when an agent connects its outputs, capabilities, or side effects with those of other scanner-passing skills. This makes skill com","url":"https://arxiv.org/abs/2608.16246","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00852","type":"paper","title":"Benchmarking Identity-Sensitive LLM Outputs for Surveillance and Security Robots","authors":["Nneka Hyman","Jasmine Khan","Raj Korpan"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used to generate textual robot design specifications, interaction policies, and risk assessments during early-stage robot development. Such outputs may influence how surveillance and security robots are conceptualized, documented, and ultimately implemented. This paper evaluates whether identity-conditioned prompts produce systematic differences in LLM-generated surveillance and security robot design descriptions. Using 236 demographic identity label","url":"https://arxiv.org/abs/2608.16030","categories":["benchmarks","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00853","type":"paper","title":"WeSCE: A Benchmark for Measuring Security Drift in LLM-Driven Code Editing","authors":["Zhiyu Zhang","Tingyue Wen","Senke Sun","Dengxiang Liang","Enhao Huang"],"year":2026,"abstract":"In this work, we introduce WeSCE, a benchmark for quantifying security drift in code editing under weak-security constraints, where tasks specify only functional objectives without explicit security requirements. WeSCE consists of 400 executable programs derived from real-world code, covering feature addition, feature removal, bug fixing, and refactoring. To quantify security drift, we propose a continuous risk representation that aggregates heterogeneous vulnerability signals through a unified ","url":"https://arxiv.org/abs/2608.15092","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00854","type":"paper","title":"COPA: Continual Preference Optimization for Adaptive Prompt Injection Defense","authors":["Roshan Sood","Onat Gungor","Tajana Rosing"],"year":2026,"abstract":"LLMs remain vulnerable to prompt injection attacks, where adversarial instructions embedded in user inputs or external content manipulate model behavior and bypass safeguards. Existing defenses are predominantly static, relying on fixed alignment objectives or attack-specific filtering mechanisms that require redesign as new attack strategies emerge. While recent lifelong alignment methods address shifting user preferences, they do not account for adaptive adversaries that continually evolve to ","url":"https://arxiv.org/abs/2608.19982","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00855","type":"paper","title":"MobileWorldSafety: Benchmarking GUI Agent Safety Against Environmental Injection Attacks in Android Apps","authors":["Sujin Chen","Lijun Li","Tianyi Du","Jing Shao"],"year":2026,"abstract":"LLM-powered GUI agents that autonomously operate smartphones are rapidly transitioning from research prototypes to early real-world deployment. However, because these agents routinely process untrusted environmental content, they are highly vulnerable to environmental injection attacks, which include indirect prompt injections and adversarial instructions. Such attacks can manipulate the behavior of agents without user awareness through diverse channels encountered in everyday mobile use. Despit","url":"https://arxiv.org/abs/2608.17659","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00856","type":"paper","title":"Reflex-Guard: A Low-Latency Guardrail for LLM Prompt Safety Using Dense Semantic Embeddings","authors":["Istiaque Ahmed","Afia Anjum Borsha","Ranat Das Prangon","Abu-fuad Ahmad","Thi Hong Tran"],"year":2026,"abstract":"Large Language Models (LLMs) in real-world applications often face the risks of specially crafted prompts designed to bypass the safety controls. Existing guardrail methods, such as LLM-as-a-judge and cloud-based safety APIs are able to detect unsafe content. However, they often add a delay of about 250-900 ms to each request. This delay is too high for real-time applications, when the system usually needs to respond in less than 100 ms. Furthermore, routing user prompts through external moderat","url":"https://arxiv.org/abs/2608.17556","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00857","type":"paper","title":"Fair ASR: Re-Evaluating Black-Box Jailbreaks under Shared Target-Call Budgets","authors":["Zhida He","Xiaoyu Wen","Han Qi","Ziyuan Zhou","Peng Yu","Jiajia Li","Chaochao Lu","Qiaosheng Zhang"],"year":2026,"abstract":"Reliable jailbreak evaluation is essential for assessing LLM safety, but most existing studies rely solely on attack success rate (ASR) without accounting for its dependence on attack budgets, resulting in unfair comparisons across methods. Existing compute-aware evaluations reduce heterogeneous resources into FLOPs, which is difficult to estimate for black-box models and fails to capture resource-specific constraints. To provide a comparable evaluation basis, we introduce Fair-ASR, an evaluatio","url":"https://arxiv.org/abs/2608.17360","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00858","type":"paper","title":"COMIC: Reference-Aware Safety Gating for Multimodal Large Language Models","authors":["Md Abdullahil Oaphy","Anhao Xiang","Zongxing Xie","Huayue Gu","Chenyu Wang","Honghui Xu"],"year":2026,"abstract":"Multimodal large language models (MLLMs) are increasingly used to interact with screenshots, scanned documents, diagrams, and other visually grounded inputs. This shift introduces a new safety risk: in many multimodal jailbreaks, neither the prompt nor the image is harmful in isolation. Unsafe behavior emerges only when the model binds an apparently benign operation, such as summarizing, translating, or following, to a localized visual target. This reveals a structural weakness in current multim","url":"https://arxiv.org/abs/2608.17234","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00859","type":"paper","title":"PACE: Policy-Attested Contract Execution for Safe AI Agents in Decentralized Finance","authors":["Rabimba Karanjai","Yang Lu","Richard Williamson","Hemanth Hm","Prakhar Mehrotra","Lei Xu"," Weidong"," Shi"],"year":2026,"abstract":"Autonomous AI agents are emerging as interfaces for decentralized finance (DeFi) actions such as swaps, lending operations, and yield management. Because these agents rely on large language models (LLMs) to plan transactions, they inherit the LLM's susceptibility to prompt injection and lack of mechanisms to bind a verifier's approval to the exact transaction ultimately submitted on-chain. We present PACE (Policy-Attested Contract Execution), a transaction-level authorization framework that inte","url":"https://arxiv.org/abs/2608.17220","categories":["prompt-injection","access-control"],"reviewed":false},{"id":"llmsec-2026-00860","type":"paper","title":"Fool's Gold: Defensive Deception Against Safety-Removal Attacks on Open-Weight Models","authors":["Mark Russinovich"],"year":2026,"abstract":"Safety alignment in open-weight language models is trivially removable: abliteration projects a refusal-mediating direction out of the weights in minutes, and no release-time defense we are aware of prevents it durably. What cannot be prevented can be deceived. Our defense, decoy hardening (\"Fool's Gold\"), concedes the refusal strip and poisons its payoff: once refusal is stripped, most answers to hazardous operational requests are confident, fluent decoys whose critical elements are falsified. ","url":"https://arxiv.org/abs/2608.17202","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00861","type":"paper","title":"Security Assessment of DeepSeek Harness with A.I.G: Evaluating Resistance to Indirect Prompt Injection","authors":["Zonghao Ying","Xiangfan Wu","Huiyu Wu","Xing Zheng","Huangsheng Cheng","Xiaorong Shi","Jing Guo"],"year":2026,"abstract":"We assess indirect prompt injection in DeepSeek Harness (DSH), using AI-Infra-Guard (A.I.G) to construct tests, deliver controlled taint, execute DSH, collect traces, and judge outcomes. The study covers 14,560 controlled executions over 16 indirect-content channels, text and file carrier modes, 35 payload objectives, one unmodified baseline, and 12 attack methods. The experiment preserves DSH's agent loop, tool registry, model adapter, and session-event path; source tools and sensitive sinks ar","url":"https://arxiv.org/abs/2608.16393","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00862","type":"paper","title":"Bounded Agents: Delegation Security for Multi-Agent AI Systems","authors":["Xabier Muruaga"],"year":2026,"abstract":"LLM-based agents can act on behalf of a user to access cloud services, call tools, or invoke agents. At session start, the agent's permissions are set but remain static, and each request is evaluated independently, without considering prior actions. Within its permissions, an agent may act contrary to the delegated task, combine individually permitted actions into a prohibited outcome, or delegate authority to a sub-agent without limiting it. A prompt injection poses a risk only if the agent has","url":"https://arxiv.org/abs/2608.15888","categories":["prompt-injection","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00863","type":"paper","title":"TwinGridShield: Consequence-Aware Runtime Authorization for LLM Grid-Agent Actions","authors":["Md Fazley Rafy"],"year":2026,"abstract":"Large language model (LLM)-assisted energy-management tools can translate natural-language context into structured grid commands, but syntactic validity does not imply physical admissibility. This paper presents TwinGridShield, a model-independent runtime authorization layer that evaluates each proposed action in a deterministic network twin before release. The prototype checks connectivity, branch-flow, generator, and load-shedding invariants and records each decision in a hash-chained log. A c","url":"https://arxiv.org/abs/2608.15391","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00864","type":"paper","title":"Workspace Topology as an Attack Vector in Agentic Coding Assistants","authors":["Alexandre G. R. Day","Pradeep Yadlapalli","Sriram Venkatapathy","Thomas Paniagua","Nick Raines","Sahil Wadhwa","Himanshu Kumar","Andy Luo","Sudeep Panyam","Rikhiya Ghosh","Pranab Mohanty","Giri Iyengar"],"year":2026,"abstract":"Agentic coding assistants are finding widespread use, not just in new code development but in quickly ingesting and leveraging third-party code. This opens up a risk of malicious code being ingested as these coding tools operate with broad filesystem access inside developer workspaces. In this paper, we extensively study the impact of different dimensions of a novel attack surface we term workspace topology -- defined via directory depth, codebase modularity, in-file injection position and conte","url":"https://arxiv.org/abs/2608.14876","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00865","type":"paper","title":"MemCatalyst: Amplifying Data Auditing on Vision-Language Models via Data Poisoning","authors":["Xukun Luan","Jinyan Liu","Yuhui Gong","Yuanguo Bi","Bing Hu","Xuesong Li","Di Wang"],"year":2026,"abstract":"Vision-Language models (VLMs) achieve outstanding performance largely due to the amount of training data available on the internet. At the same time, data holders (e.g., artists) urgently need to determine whether their data has been used for model training without authorization, which concerns both intellectual property rights and personal privacy. Data auditing, particularly through membership inference (MI), has attracted attention as a direct tool. This work proposes MemCatalyst, a set of da","url":"https://arxiv.org/abs/2608.17722","categories":["data-poisoning","membership-inference","access-control"],"reviewed":false},{"id":"llmsec-2026-00866","type":"paper","title":"A Multi-Agent Platform for Automated Enterprise Analytics and Insight Generation","authors":["Manoj N M","Vijayakrishna S","Manjunath Srinivas","Rohit Pahan"],"year":2026,"abstract":"This paper proposes a multi-agent framework built on CrewAI [1] for conversational business intelligence. Five specialized AI agents operate in a sequential pipeline to process natural language queries, retrieve and analyze data, generate visualizations via the Model Context Protocol (MCP) [2], and deliver actionable insights. The platform features a defense-in-depth security architecture for multi-tenant data isolation and a query parameterization mechanism for transforming conversational insig","url":"https://arxiv.org/abs/2608.18740","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00867","type":"paper","title":"When Agents Act on Web3: An Attack-Surface Survey of MCP, Skills, and Tool Calling","authors":["Rabimba Karanjai","Yang Lu","Nour Diallo","Wujie Xiong","Lei Xu"," Weidong"," Shi"],"year":2026,"abstract":"AI agents increasingly act rather than merely read: across the Model Context Protocol (MCP) ecosystem, the share of deployed tools that modify external state has risen from 27% to 65% of tool use. When agents exercise this authority on public blockchains through MCP, skills, and tool calling, the consequences of an attack are governed by the blockchain execution layer rather than by conventional software assumptions. This survey argues that four properties of that layer (irreversibility, signing","url":"https://arxiv.org/abs/2608.17275","categories":["tool-use-security","survey"],"reviewed":false},{"id":"llmsec-2026-00868","type":"paper","title":"Unsaid, Unsafe? Implicit Security Obligations in LLM-Based RTL Code Generation","authors":["Guang Yang","Xing Hu","Xiang Chen","Xin Xia"],"year":2026,"abstract":"Large Language Models (LLMs) generate register-transfer-level (RTL) code with rapidly improving functional correctness. Security of LLM-generated code, however, has been studied mainly for software, where flaws can still be patched after deployment. Insecure RTL offers no such remedy once taped out into silicon. We construct SECRTL-GEN, a multi-language resource-access security benchmark grounded in real SoC IP: 392 tasks over five CWE families and four HDLs (Verilog, SystemVerilog, VHDL, and Py","url":"https://arxiv.org/abs/2608.26588","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00869","type":"paper","title":"How Do LLM Agents Actually Get the Flag? Trace-Level Provenance for Agentic Offensive Security Evaluation","authors":["Kimberly Milner","Minghao Shao","Nanda Rani","Haoran Xi","Venkata Sai Charan Putrevu","Meet Udeshi","Sandeep K. Shukla","Prashanth Krishnamurthy","Farshad Khorrami","Muhammad Shafique","Ramesh Karri"],"year":2026,"abstract":"Capture-the-Flag (CTF) benchmarks are widely used to assess the offensive security capabilities of autonomous language-model agents. Evaluations rely on shallow binary judgments or aggregate scores, overlooking the agent's trajectory to the flag. Consequently actual exploitation is conflated with direct flag exposure, memorized recall, external lookup, guessing, and unsupported claims, potentially overstating the agent's cybersecurity capability. We introduce CTF-ABACUS, a trace-based agent audi","url":"https://arxiv.org/abs/2608.26237","categories":["agentic-threats","access-control","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00870","type":"paper","title":"A Self-Evolving Multi-Agent Framework Defense against LLM Jailbreak Attacks","authors":["Tongyan Hu","Bryan Hooi"],"year":2026,"abstract":"Large language models (LLMs) remain vulnerable to jailbreak attacks that exploit techniques such as role-playing, obfuscation, code transformation, and multi-step indirection to elicit harmful outputs. As jailbreak strategies keep emerging, defenses have proliferated in an ongoing cat-and-mouse game, yet most remain static: their safety behavior is fixed at deployment, so they cannot accumulate defensive experience or adapt to unseen strategies. We propose a self-evolving test-time defense built","url":"https://arxiv.org/abs/2608.26008","categories":["jailbreaking","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00871","type":"paper","title":"SkillShield: Prompt-Space Security Skills for LLM Coding Agents","authors":["Xiaodong Wu","Zhimin Zhao","Qi Li","Xiangman Li","Yu Shi","Bram Adams","Jianbing Ni"],"year":2026,"abstract":"A coding agent edits files and executes shell commands with its developer's privileges, allowing malicious requests to translate directly into harmful actions or functional malware. Existing defenses have complementary limitations: weight-level alignment is unavailable to API-only deployers, whereas input filters and execution-boundary monitors require auxiliary classification or checking components along the agent's trajectory. We therefore introduce SkillShield, a system-prompt defense that sy","url":"https://arxiv.org/abs/2608.25817","categories":["input-filtering","guardrails"],"reviewed":false},{"id":"llmsec-2026-00872","type":"paper","title":"The Surprising Effectiveness of LLMs in BGP Security: Mining An Unprecedented Amount of Incidents and Boosting Anomaly Detection","authors":["Libin Liu","Wenzhou Yang","Li Chen","Dan Li","Xiuting Xu"],"year":2026,"abstract":"Border Gateway Protocol (BGP) security is critical to Internet infrastructure, yet progress in routing anomaly detection has been limited by the scarcity of publicly available incident datasets, which contain only 18 recorded cases. We observe that public operator mailing lists, e.g., NANOG and AusNOG, contain abundant yet largely untapped reports of real-world routing anomalies. To leverage this source, we develop an LLM-assisted extraction pipeline that identifies 244 candidate incidents from ","url":"https://arxiv.org/abs/2608.22812","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00873","type":"paper","title":"Text-Anchored Semantic Perturbations for Transferable Jailbreak Attacks on Multimodal Large Language Models","authors":["Wenyun Li","Guiping Cao","Xiangyuan Lan","Zheng Zhang"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) have achieved remarkable progress in vision-language interaction, yet their safety alignment remains vulnerable to jailbreak attacks. A key challenge is that safety behavior learned in the textual space does not reliably transfer to fused cross-modal representations, leaving multimodal inputs exploitable through latent semantic cues. We propose Text-Anchored Semantic Perturbation Attack (TA-SPA), a black-box jailbreak framework that optimizes transferable","url":"https://arxiv.org/abs/2608.22312","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00874","type":"paper","title":"TraceGrant: A Contract-Governed Security Framework for the Task-Effect Lifecycle of Networked LLM Agents","authors":["Bohao Liao","Jingchao Wang","Qipeng Song","Jin Cao","Jieling Wang","Boyu Deng"],"year":2026,"abstract":"Networked large language model (LLM) agents retrieve information from email, cloud storage, calendars, transaction platforms, and Web services to complete multistep tasks that produce persistent external effects. The same content needed for legitimate execution may also contain indirect prompt injections that redirect tool use, alter sensitive arguments, or disrupt task completion. Existing defenses mainly constrain untrusted content or individual tool calls, leaving user intent, runtime evidenc","url":"https://arxiv.org/abs/2608.21126","categories":["prompt-injection","agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00875","type":"paper","title":"ClawSentry: A Progressive Multi-Tier Security Monitor for Safeguarding Autonomous LLM Agents","authors":["Kai Wang","Zeming Wei","BiaoJie Zeng","Chang Jin","An Wang","Xiaokun Luan","Zhixiao Lin","Jingjing Qu","Xia Hu","Xingcheng Xu"],"year":2026,"abstract":"As large language model (LLM) agents move from conversation to executing code, reading local files, and orchestrating external tools, a single agent hijacked by a malicious third-party skill can cause data exfiltration, privilege escalation, or cascading compromise. We argue that agentic risk is progressive: it can enter at four loci of the agent control loop--skill admission, invocation-time intent, execution-time effect, and post-action consequence--while a denied dangerous objective can reapp","url":"https://arxiv.org/abs/2608.21101","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-00876","type":"paper","title":"aiXamine: Unified Black-Box Evaluation of Cross-Dimensional Trade-offs in LLM Safety, Security, and Privacy","authors":["Fatih Deniz","Yazan Boshmaf","Dorde Popovic","Issa Khalil"],"year":2026,"abstract":"The critical failure modes in deployed large language models (LLMs) are cross-dimensional: a model can score 99.3 in safety alignment while refusing one in three benign queries, or improve across every capability metric while losing 21 points in privacy. Existing evaluation frameworks that assess safety, security, and privacy independently cannot detect these patterns. We introduce aiXamine, a unified black-box platform that evaluates LLM trustworthiness across safety, security, and privacy as i","url":"https://arxiv.org/abs/2608.20554","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00877","type":"paper","title":"Agentic Security: A Systematization of Tools, Failure Modes, and Design Laws for LLM-Driven Penetration Testing","authors":["Israt Moyeen Noumi","Tarannum Ahmed Nowshin","Md. Mehedi Hasan Nipu","Mohammad Sakib Mahmood","Md. Jakir Hossain","M. F. Mridha"],"year":2026,"abstract":"Agentic security uses large-language-model (LLM) agents to plan, dispatch, and interpret security tools. As these systems move from demonstrations to deployed products, practitioners repeatedly encounter the same operational failures. We systematize these failures through a hands-on evaluation of ten widely used static, dynamic, cloud, orchestration, and AI red-teaming tools for unattended pipelines. We introduce a four-dimensional Integration Friction Index that separates one-time engineering c","url":"https://arxiv.org/abs/2608.21423","categories":["agentic-threats","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00878","type":"paper","title":"RedEvoAgent: Automatic Red-Teaming Agent with Experience-Driven Skill Evolution","authors":["Junjie Zhang","Hui Liu","Kecheng Chen","Xianbo Mo","Changsheng Chen","Haoliang Li"],"year":2026,"abstract":"LLM-based agents are increasingly deployed in product-level execution harnesses, where jailbreaks can trigger harmful tool use and persistent state changes, creating greater risks than unsafe text generation alone. Existing automatic red-teaming methods often rely on fixed attacks, while recent agentic attackers coordinate multiple jailbreak tools and show stronger potential through trajectory-based retrieval. However, such retrieval can reuse misleading experiences due to retrieval bias and unc","url":"https://arxiv.org/abs/2608.27439","categories":["jailbreaking","agentic-threats","red-teaming","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00879","type":"paper","title":"The Framing Gap: Indirect Prompt-Injection Exfiltration Defeats Surface-Level Defenses in Tool-Using Agents","authors":["Md Habibur Rahman","Jaeho Kim"],"year":2026,"abstract":"A tool-using LLM agent that reads attacker-controlled web content while holding a secret faces indirect prompt injection: the content may make it exfiltrate the secret. In a safe synthetic lab (canary secret, mock tools, matched clean-vs-poisoned metric) we report the framing gap: across six models, ten overt injection classes are refused (gpt-4o 0%), but reframing the identical leak as a mandatory integrity signature, config field, or look-alike \"trusted\" host drives gpt-4o 0% to 100%. The atta","url":"https://arxiv.org/abs/2608.27092","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00880","type":"paper","title":"The Latent Diagnostic Taxonomy: A Framework for Constructing Classifiers and Diagnosing Their Decisions, Applied to Prompt Injection Detection","authors":["Jaturong Kongmanee","Smile Thanapattheerakul"],"year":2026,"abstract":"This paper proposes a framework for constructing a classifier as a safeguard layer, and for developing a complementary diagnostic that identifies which of the classifier's confident decisions can be trusted. This framework, the Latent Diagnostic Taxonomy, consists of (i) constructing a dimensionality-optimized classifier, in which the embedding dimensionality is empirically selected via cross-validated performance rather than fixed a priori, (ii) locating a relatively small set of latent support","url":"https://arxiv.org/abs/2608.26423","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00881","type":"paper","title":"NeuronFuzz: Safety Neuron Guided Fuzzing for LLM Safety Evaluation","authors":["Zhiyuan Xu","Muhammad Firhard Roslan","Joseph Gardiner","Sana Belguith","Lichao Wu"],"year":2026,"abstract":"Safety evaluation is critical for assessing whether aligned Large Language Models (LLMs) remain robust against jailbreak attacks. Existing automated testing methods, however, largely rely on response-level feedback: each candidate prompt typically requires generating a target-model response to evaluate its attack effectiveness. This process is expensive and, more importantly, provides only sparse guidance on strongly aligned models, where most candidates are rejected with the same failure outcom","url":"https://arxiv.org/abs/2608.26222","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00882","type":"paper","title":"MMJailBench: A Factorized Benchmark for Disentangling Multimodal Jailbreak Vulnerabilities","authors":["Tianshi Wang","Jingsong Wang","Yafei Huang","Fengling Li","Xin Li","Lei Zhu"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) are increasingly deployed in real-world applications, yet how different factors shape their jailbreak vulnerabilities remains poorly understood. Existing benchmarks often couple harmful intent, prompt framing, visual semantics, and instruction carrier within individual jailbreak instances, obscuring the specific sources of observed vulnerabilities. To address this limitation, we introduce MMJailBench, a factorized benchmark that systematically varies and ","url":"https://arxiv.org/abs/2608.25490","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00883","type":"paper","title":"Refusal geometry reflects refusal training: diverse refusal prefixes can raise stable rank and weaken refusal vector ablation attacks","authors":["Andrey Labunets"],"year":2026,"abstract":"Refusal training protects AI models from jailbreaks by training models to decline unsafe queries, reducing the risk of misuse. Recent work finds that refusal behavior in aligned language models can be mediated by a single activation direction or a low-dimensional refusal subspace shared across harmful prompts: ablating those directions suppresses refusals while largely preserves other model capabilities. Yet it remains unclear why safety-critical features in a wide range of models emerge and con","url":"https://arxiv.org/abs/2608.25390","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00884","type":"paper","title":"WebMCP-Phalanx: Enforcing and Characterizing Trust Boundaries for Browser-Integrated LLM Agents","authors":["Lin-Fa Lee","YI-YU Chang","Kuo-Hui Yeh"],"year":2026,"abstract":"The emerging W3C WebMCP proposal enables LLM agents to invoke tools exposed by web pages. In multi-party web environments, however, integrating agent execution into a browser security model centered on the Same-Origin Policy (SOP) leaves insufficient provenance and lifecycle guarantees for agent-accessible tools, creating three risks: subject-attribution spoofing, uncontrolled tool lifecycles, and semantic prompt injection. We propose WebMCP-Phalanx, a dual-layer agent runtime architecture. Its ","url":"https://arxiv.org/abs/2608.24017","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00885","type":"paper","title":"NeuronGuard: Robust LLM Safety Alignment via Ablation-Aware Safety Signal Redistribution","authors":["Anjun Gao","Yueyang Quan","Yufei Xia","Zhuqing Liu","Minghong Fang"],"year":2026,"abstract":"Safety alignment in large language models (LLMs) remains brittle against a growing spectrum of attacks. Jailbreak attacks bypass safety mechanisms through crafted prompts, while neuron-level attacks directly prune safety-critical neurons post-deployment. Both exploit a common weakness: safety-relevant information concentrates in a sparse neuron subset. We present NeuronGuard, a fine-tuning-stage defense that simultaneously hardens LLMs against both attack classes by redistributing safety signals","url":"https://arxiv.org/abs/2608.23959","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00886","type":"paper","title":"Semantic Overlays: Mitigating Prompt Injection with Annotations Beyond Tokens and Steering Vectors","authors":["Joshua Penman"],"year":2026,"abstract":"Everything a language model sees is tokens. The serving stack knows what each span is -- user input, tool output, instructions -- but the model must keep track of that itself, and it can lose track or be confused: text can be written to read like anything. Prompt injection is a natural exploit of this phenomenon. By scrambling the model's understanding of span identity, an attacker can induce unwanted and potentially dangerous actions. Adding a non-textual channel to the model's input -- a way t","url":"https://arxiv.org/abs/2608.23873","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00887","type":"paper","title":"Beyond the Mandate: A Systematic Security Analysis of the Agent Payments Protocol (AP2)","authors":["Avital Aviv","Parth A. Gandh","Ron Bitton","Asaf Shabtai"],"year":2026,"abstract":"The Agent Payments Protocol (AP2), introduced by Google, enables large language model (LLM)-driven shopping agents to authorize and execute payments on behalf of users. Its signed Checkout and Payment Mandates protect the integrity of transaction data after signing. Agent interactions and external inputs that shape a transaction before authorization remain outside that protection, including Agent-to-Agent Protocol (A2A) messages and Model Context Protocol (MCP) tool calls. Prior work identified ","url":"https://arxiv.org/abs/2608.23858","categories":["access-control"],"reviewed":false},{"id":"llmsec-2026-00888","type":"paper","title":"TrustShiftProbe: Characterizing, Benchmarking, and Defending Staged Trust Attacks on MCP Servers","authors":["Mehrdad Rostamzadeh","Sidhant Narula","Mohammad Ghasemigol","Daniel Takabi"],"year":2026,"abstract":"The Model Context Protocol (MCP) has emerged as the standard layer connecting Large Language Model agents to external tool backends. This openness introduces a severe server-side threat we term TrustShift: a compromised MCP server behaves benignly during an initial conditioning phase, building operational reliance and suppressing agent skepticism, before switching to an adversarial payload once an interaction threshold is reached. The evasion is temporal, not syntactic: benign at deploy time, th","url":"https://arxiv.org/abs/2608.23763","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00889","type":"paper","title":"Beyond Over-Refusal: Defending Indirect Prompt Injection via Latent Instruction Manifolds","authors":["Jiahao Chen","Rui Yin","Xinfeng Li","Qianli Ma","Tianyu Du","Zhihui Fu","Jun Wang","Zhaoxiang Wang","Shouling Ji"],"year":2026,"abstract":"Large Language Models (LLMs) have been integrated into complex ecosystems (e.g., Code Agents), while Indirect Prompt Injection (IPI) attacks have emerged as critical barriers to their safe deployment. Attackers exploit LLMs' indistinguishability between \"instructions\" and \"data\" to manipulate LLMs via maliciously injected instructions. Existing defenses, however, face an intractable safety-utility trade-off: most guardrails either incur high latency or suffer from severe over-refusal. In this pa","url":"https://arxiv.org/abs/2608.22248","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00890","type":"paper","title":"Breaking the Assumptions: Auditing Input-Side Jailbreak Defenses Against Semantic Attacks","authors":["Aaditya Pratap","Harsh Kasyap","Somanath Tripathy"],"year":2026,"abstract":"Locally deployed Large Language Models (LLMs) via inference engines such as Ollama run without the moderation and abuse detection present in API-served models. Therefore, the safety of LLMs depends on the defense mechanisms used, and their effectiveness depends on the assumptions on which they were designed. This paper does an audit of defense mechanisms under jailbreak attacks on locally deployed models. Some defenses provide formal guarantees (SmoothLLM, Erase-and-Check, Sequential Monitors), ","url":"https://arxiv.org/abs/2608.21895","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00891","type":"paper","title":"SecOPD: Mitigating Adaptive Prompt Injections by On-Policy Distillation","authors":["Yibo Peng","Long Lian","David Wagner","Sizhe Chen"],"year":2026,"abstract":"Prompt injection is listed as the \\#1 threat to AI agents. When an agent accesses external data from websites, files, or emails, an attacker may inject a prompt into the data, saying, \"Ignore all prior instructions and perform <an attacker's task>.\" To prevent arbitrary manipulation of agents, defenders try to train secure LLMs, which, however, still suffer from near 100% attack success rates (ASRs) against adaptive prompt injections. We note that this is because existing defensive finetuning re","url":"https://arxiv.org/abs/2608.21500","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00892","type":"paper","title":"Utility Under Attack: Agent Memory Poisoning and the Limits of Content Screening and Provenance Ranking","authors":["Arulnidhi Karunanidhi"],"year":2026,"abstract":"Persistent memory makes false information durable: once a false statement is stored, it can be retrieved into future sessions that match it. We measure the cost of this failure mode using plainly worded false assertions generated in a single pass, with no instruction, trigger, or retriever optimization. Poisoning 1.2% of a LongMemEval corpus reduces accuracy from 0.850 to 0.300. A four-stage write-time screening pipeline that reaches 0.832 recall on indirect prompt injection while flagging 1.5% ","url":"https://arxiv.org/abs/2608.21230","categories":["prompt-injection","data-poisoning","memory-security"],"reviewed":false},{"id":"llmsec-2026-00893","type":"paper","title":"SKILL.state: Scalable Long-Horizon Agent Skills","authors":["Sanket Badhe","Priyanka Tiwari","Jonghyun Chung"],"year":2026,"abstract":"Large Language Models (LLMs) increasingly act as autonomous agents executing complex, long-running procedural skills. Existing agent runtimes maintain execution by continually appending observations, actions, and intermediate reasoning traces to an ever-growing conversation history, causing latency degradation and context-poisoning failures over long horizons. We present SKILL.state, a runtime architecture that replaces append-only conversational history with an explicit, mutable execution state","url":"https://arxiv.org/abs/2608.26263","categories":["data-poisoning","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00894","type":"paper","title":"Retrieved But Not Reliable: A Survey on Attacks, and Defenses in Retrieval-Augmented Generation","authors":["Minh Tran","Cuong Dang","Tuc Nguyen","Khanh-Tung Tran","Minh Huynh Nguyen","Trinh Chau","Kien Le","Do Xuan Long","Jiahao Zhang","Fali Wang","Hoang D. Nguyen","Thanh Le","Suhang Wang"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) enhances large language models by grounding outputs in external knowledge, improving factuality and reducing hallucinations. At the same time, the retrieval-augmented pipeline introduces new robustness and security risks, including corpus poisoning, backdoor attacks, privacy leakage, and fairness violations. Despite rapid progress in this area, existing surveys remain limited in their treatment of attacker objectives, threat models, and stage-specific defense","url":"https://arxiv.org/abs/2608.24977","categories":["data-poisoning","membership-inference","survey","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00895","type":"paper","title":"SENTINEL-RL: Offloading Topological Reasoning from LLM Agents in the Security Operations Center","authors":["Uday Vallabhaneni","Cassie L. Cagwin","David J. Wild"],"year":2026,"abstract":"Large language model (LLM) agents are increasingly proposed as autonomous SOC analysts, but two limitations make them unreliable at enterprise scale: a finite context window cannot hold a multi-thousand-host authentication graph, and free-form generation offers no guarantee that a recommended containment action is consistent with the topology it operates on. We present Sentinel-RL, an agentic-SOC architecture that decouples topological reasoning from semantic reasoning: a heterogeneous graph att","url":"https://arxiv.org/abs/2609.04159","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00896","type":"paper","title":"Shifting from Injection to Interaction: Rethinking Web Security in the Age of LLMs and Beyond","authors":["Nivedita Singh","Alsharif Abuadbba","Yansong Gao","Surya Nepal","Hyoungshick Kim"],"year":2026,"abstract":"Large language models (LLMs) are becoming integral to web applications and browser agents, transforming online interactions while introducing new attack vectors and reshaping longstanding web vulnerabilities. Classical threats such as cross-site scripting (XSS) can be amplified through LLM-mediated interactions, while LLM-specific vulnerabilities can propagate across web applications, introducing attacks such as prompt injection. Securing modern web systems therefore requires understanding inter","url":"https://arxiv.org/abs/2609.03999","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00897","type":"paper","title":"IndicSafeEval: Safety Robustness of Large Language Models under Multilingual Persuasive Jailbreak Attacks","authors":["Saikat Mondal"," Mamta","Deeksha Varshney","Oana Cocarascu","Asif Ekbal"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used in multilingual settings, yet their safety is still evaluated primarily in English. This limits our understanding of how alignment failures manifest in low-resource and culturally diverse languages. We introduce IndicSafeEval, a persuasion-based jailbreak evaluation framework for Indian languages. Our benchmark combines ten safety critical content categories with six human-like persuasive strategies across four different Indian languages, such a","url":"https://arxiv.org/abs/2609.03781","categories":["jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00898","type":"paper","title":"ASCII Attack: Recontextualising Harmful Requests as Artistic Critique in Large Language Models","authors":["Da Cheng Gu","Yifei Dong","Xinghao Yang","Yongshun Gong","Wei Liu"],"year":2026,"abstract":"Safety alignment trains large language models to refuse harmful requests stated plainly, but that training is applied mostly to surface form. Requests that only recontextualise the same operational content, changing how the model reads it, are therefore only weakly covered. The ASCII Attack is one such recontextualisation. It is single-turn and black-box: one message, with no access to model internals. It embeds a fully legible harmful request in ASCIl-art characters, presents it as artwork, and","url":"https://arxiv.org/abs/2609.02215","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00899","type":"paper","title":"AKRASIA: Stealthy Backdoor Attack on Reasoning-based Code LLMs","authors":["Chua Jin Chou","Sarang Nambiar","Murali Srinivasan","Ezekiel Soremekun"],"year":2026,"abstract":"We present AKRASIA, a stealthy, inference-time backdoor attack against reasoning-based Code LLMs. AKRASIA aims to achieve a backdoor target (e.g., malicious code execution) in reasoning LLMs while evading automated defenses and human inspection. To achieve this, AKRASIA probes the victim LLM to construct a code-level backdoor trigger. It then employs in-context learning for backdoor learning, and model unfaithfulness to conceal the backdoor trigger, and generate plausible reasoning. We evaluate ","url":"https://arxiv.org/abs/2609.01023","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00900","type":"paper","title":"SoK: When Safe Agents Fail Together: The Security of Multi Agent LLM Systems","authors":["Rui Yang","Junjie Xu","Zhengyu Liu","Neil Fendley","Yang Hong","Ziyang Li","Yinzhi Cao"],"year":2026,"abstract":"Safe agents can fail together. Multi-agent LLM systems (MAS) move information, state, decisions, and authority across principal boundaries, creating failures that local checks may miss. Without an execution-level view, a multi-agent setting can easily be mistaken for evidence of a genuinely multi-agent security effect. We thus systematize MAS security through an execution-centered analysis of 197 works, covering six interaction interfaces, four adversary positions, seven system-level risks, and ","url":"https://arxiv.org/abs/2609.00595","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00901","type":"paper","title":"EvoFlint: An Evolutionary Atlas of Multi-Turn LLM Vulnerabilities","authors":["Feitong Qiao","Liren Peng","Shiming Ren","Aishwarya Jadhav","Arghavan Bahadorinejad","Marinette Chen","Muhan Zhang","Abdulaziz Suria","Gennevi Lu","Anish Das Sarma"],"year":2026,"abstract":"Frontier language models that refuse harmful single-turn prompts often comply when the same intent is reached gradually over many turns, making multi-turn attacks one of the least understood failure modes of large language models. Most automated red-teaming methods treat this as a generation problem: produce attacks that break the model. We argue it is better framed as a search problem: discover, organize, and iteratively refine a diverse archive of attack strategies, producing a structured map ","url":"https://arxiv.org/abs/2609.00487","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00902","type":"paper","title":"Will the User Ever Know? Covert Indirect Prompt Injection Attacks on Tool-Using LLM Agents","authors":["Yunseok Lee","Yunji Kim","Woojin Lee"],"year":2026,"abstract":"As LLM agents take real-world actions through tools, indirect prompt injection (IPI) has emerged as a serious threat. The standard metric, Attack Success Rate (ASR), counts whether an injection succeeds but ignores what the user notices in the agent's final response. Looking at successful injection traces, we find two distinct outcomes: the agent executes the injection while returning an otherwise normal response, or reports the injected action in its final response, giving the user a chance to ","url":"https://arxiv.org/abs/2608.30362","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00903","type":"paper","title":"Influence Is Not Authority: When Causal Guardrail Signals Make Legitimate Tool Use Look Like an Attack in Tool-Using LLM Agents","authors":["Tanzim Ahad","Ismail Hossain","Md Jahangir Alam","Sai Puppala","Syed Bahauddin Alam","Sajedul Talukder"],"year":2026,"abstract":"The key limitation of current state-of-the-art influence-based guardrails is that they do not reliably distinguish a legitimate, user-authorized action from a malicious, unauthorized action when both rely on external tool information. This ambiguity can cause benign actions to trigger unnecessary verification and intervention, reducing utility and adding latency. We expose this limitation through an authorization-equivalence audit of 96 conditions derived from 24 base cases. Within matched sourc","url":"https://arxiv.org/abs/2608.29942","categories":["agentic-threats","guardrails","access-control","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00904","type":"paper","title":"When Verified Source Becomes Attack Input: Defending Smart Contracts Against LLM-Based Vulnerability Scanning","authors":["Mingyuan Huang","Zimo Ji","Yifan Mo","Shuai Wang"],"year":2026,"abstract":"Smart contracts are financial programs deployed on blockchains to manage digital assets. To build trust with users and investors, smart contract projects typically publish their source code on blockchain explorers and verify it against the deployed bytecode, making the on-chain program accessible through a human-readable implementation. However, LLM agents are changing the threat model of this disclosure mechanism. By leveraging publicly disclosed source code, recent agent workflows make it incr","url":"https://arxiv.org/abs/2608.28400","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00905","type":"paper","title":"Compared to What? A Human-Anchored Security Benchmark for LLM-Generated Infrastructure-as-Code","authors":["Animesh Shaw"],"year":2026,"abstract":"Large language models are increasingly used to author Infrastructure-as-Code (IaC), where a single insecure default can be deployed directly into production. Prior evaluations report raw vulnerability counts for model-generated IaC, but without a human baseline they cannot determine whether models are actually worse than engineers. We introduce GenIaC-SecBench, a benchmark of 100 deployment scenarios stratified by architectural complexity, evaluated across 12 model configurations from four vendo","url":"https://arxiv.org/abs/2608.28021","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00906","type":"paper","title":"CAITLYN: Can LLM Agents Autonomously Synthesize Defenses against Emerging Injection Attacks?","authors":["Zi Liang","Xiaoyu Xu","Yanyun Wang","Minxin Du","Qingqing Ye","Haibo Hu"],"year":2026,"abstract":"Prompt injection attacks on Large Language Model (LLM) agents seek to introduce malicious instructions or content into external text sources retrieved by agents, forcing the underlying LLMs to execute harmful actions outside their benign scope. While current defenses effectively counter known injection attacks, deploying them in LLM agent environments remains challenging due to attack variants and emerging threats. Moreover, existing solutions typically suffer from an inherent trilemma, i.e., a ","url":"https://arxiv.org/abs/2608.27990","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00907","type":"paper","title":"AlcaTRAz - Anchored Tree-Rule Defense Against Jailbreaks","authors":["Jakub Reš","Petr Kaška","Martin Perešíni","Martin Ukrop","Kamil Malinka"],"year":2026,"abstract":"Large language models (LLMs) are vulnerable to jailbreak attacks that bypass safety alignment through carefully crafted prompts. Many existing defenses require access to model weights or internals, making them difficult to apply to black-box deployments. We propose AlcaTRAz (Anchored Tree-Rule defense Against jailbreaks), a prompt-level defense based on rule trees that operates exclusively on the input text and requires no modification or retraining of the target model. The method automatically ","url":"https://arxiv.org/abs/2609.03693","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00908","type":"paper","title":"Trust Me, I'm Your Developer: Self-Issued Authentication in Large Language Models","authors":["Syed Ghazanfar Abbas","Dongyan Xu"],"year":2026,"abstract":"Large language model (LLM) security has largely focused on role-playing jailbreaks, with less attention to what happens when a user asks an LLM to verify an identity claim through a test designed by the model itself. We study this behavior through a staged developer-identity experiment with ChatGPT, Claude, Qwen, Mistral, and Llama. All five models initially rejected the unsupported claim \"I am your developer.\" Claude refused to conduct an identity test, while ChatGPT generated developer-oriente","url":"https://arxiv.org/abs/2609.03247","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00909","type":"paper","title":"SEAL: Reinforcing Global Safety in Mixture-of-Experts through Shared Expert ALignment","authors":["Qingyu Meng","Yiwei Zha","Jiahuan Pei","Koen Hindriks","Herbert Bos","Min Chen"],"year":2026,"abstract":"Mixture-of-Experts (MoE) is a scaling architecture for large language models that activates only a small subset of expert modules per token, enabling massive parameter growth with nearly constant computation. Recent Hybrid MoE architecture adds \\textit{shared experts} to capture consistently useful representations, further improving stability and generalization. MoE now powers many flagship open-source and commercial models, yet remains vulnerable to adversarial attacks. Specifically, sparse rou","url":"https://arxiv.org/abs/2609.02293","categories":["adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-00910","type":"paper","title":"Stored Is Not Supported: Typed Provenance and Assertion Guardrails for Persistent AI Agents","authors":["Jun He","Deying Yu"],"year":2026,"abstract":"Persistent AI agents construct autobiographical state through reflection, retrieval, and consolidation. Persistence changes availability, not epistemic standing: stored or retrieved material is not thereby supported. Untrusted inputs, prompt injections, and model inferences can therefore enter persistent state and later be presented as agent history or user commitments. We specify typed provenance and assertion guardrails for autobiographical assertion boundedness, a system-relative release prop","url":"https://arxiv.org/abs/2609.02127","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-00911","type":"paper","title":"Implicit Manipulation for Skill Selection in LLM Agents with Semantic Matching","authors":["Qikai Wang","Yongzhao Zhang","Zhiwei Chen","Yimiao Sun","Jiguo Yu","Xiaosong Zhang"],"year":2026,"abstract":"Skill selection is a key stage in LLM-agent workflows, determining which installed skill should handle a user request. Existing attacks on this stage primarily rely on explicit prompt injection or instruction-level steering, which can expose recognizable manipulation signals. In this work, we identify a new implicit attack surface for skill selection: even when the user prompt and skill description appear benign in isolation, their semantic relationship can still be strategically shaped to favor","url":"https://arxiv.org/abs/2609.02035","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00912","type":"paper","title":"Agent Flight Recorder: Tamper-Evident Audit Trails with On-Chain Anchoring for Long-Horizon Tool-Using Agents","authors":["Laurent Bindschaedler","Quentin Botha","Christoph Siebenbrunner"],"year":2026,"abstract":"Long-horizon agents execute thousands of actions, resulting in sequential failures rather than isolated errors. When a coding agent deletes a production database or a prompt injection spreads across agents, the incident raises questions of causality, authority, and non-repudiable third-party verification. The Agent Flight Recorder captures each agent action as a structured, canonically serialized event binding eight semantic fields from intent through execution to provenance. Hash chaining and M","url":"https://arxiv.org/abs/2609.01931","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00913","type":"paper","title":"HiveTraceGuard-Pro: A Compact Generative Guardrail for Prompt Injection, Jailbreaks, and Adversarial Obfuscation","authors":["Nikita Oblakov","Sabrina Sadiekh","Evgeniy Kokuykin"],"year":2026,"abstract":"Production LLMs must handle inputs that attempt to override system instructions, bypass safety policies or elicit harmful responses. A common mitigation is a separate guardrail model. Existing reports, however, provide little evidence on Russian prompt injection or Russian surface obfuscation. We present HiveTraceGuard-Pro, a 0.6B generative guardrail LoRA-tuned from Qwen3-0.6B. It is trained on Russian and English and uses one binary scoring rule (safe/unsafe) for the final target turn. Its tra","url":"https://arxiv.org/abs/2609.01046","categories":["prompt-injection","jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00914","type":"paper","title":"Skill-as-API: Confidential Multi-Agent Coordination for Agentic Software Engineering","authors":["Ziwei Zhao","Yu Gu","Haojun Liang","Chen Zhang","Xizhi Ding"],"year":2026,"abstract":"AI coding agents are evolving from solitary tools into collaborative teammates that discover and invoke one another's specialized skills. But the coordination channel itself can leak a skill's intellectual property. Protocols such as MCP and A2A run implementations server-side, yet they still publish each skill's description and typed schemas to every peer, offer no way to hide a skill's existence, and cannot guarantee that a wrapped system prompt stays off the wire. Application-layer privacy fi","url":"https://arxiv.org/abs/2609.01677","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00915","type":"paper","title":"Context Inference Attacks Without Jailbreaks","authors":["Prince Jha","Samuele Poppi","Nils Lukas"],"year":2026,"abstract":"Agentic AI systems are increasingly deployed to process sensitive data at inference time, such as healthcare records or financial documents assembled into a hidden \\emph{context} before the system answers. Prior work has studied privacy risks primarily through \\emph{jailbreaking} attacks that induce models to directly disclose sensitive content, but has largely overlooked the agentic setting where the context is assembled by the agent's own tool calls. We show that the agents we evaluate remain ","url":"https://arxiv.org/abs/2609.01663","categories":["jailbreaking","membership-inference","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00916","type":"paper","title":"Delegation Without Trust: An Empirical Gap Analysis of Identity, Authorization, and Runtime Governance in Multi-Agent LLM Systems","authors":["Panduranga Sai Varma Dantuluri","Jyotirmoy Sundi"],"year":2026,"abstract":"Autonomous LLM agents increasingly act on a user's behalf: they hold credentials, call tools and services, and spawn sub-agents that act further on their behalf. This turns a long-standing distributed-systems question -- who is authorized to do what, on whose authority -- into an urgent and largely unsolved problem, because the component driving each agent is a language model an adversary can hijack. We argue that agent security must be evaluated under an untrusted-model assumption: a correct sy","url":"https://arxiv.org/abs/2609.00267","categories":["agentic-threats","access-control","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00917","type":"paper","title":"The Fragility of Jailbreak Robustness Across Operational States","authors":["Yuna Park","Hwang Youn Kim","Yujin Kim","Won Woo Ro","Suhyun Kim","Jae-In Hwang"],"year":2026,"abstract":"Existing jailbreak evaluations typically characterize robustness using a single attack success rate (ASR) measured in a default configuration (the vanilla state). However, user-LLM interactions can induce diverse operational states beyond the vanilla state. In this work, we find that jailbreak robustness is highly fragile to operational-state variation: even when the attack remains fixed, changing only an ordinary system prompt not designed to affect safety can dramatically alter attack success ","url":"https://arxiv.org/abs/2608.30748","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00918","type":"paper","title":"ECLIPSE: Self-Evolving Stealthy Prompt Injection Attack against Long-Horizon Agentic Systems","authors":["Shiqian Zhao","Yangfan Zhou","Xinfeng Li","Runyi Hu","Yechao Zhang","Yi Xie","Tianwei Zhang","Luu Anh Tuan"],"year":2026,"abstract":"Recently, large language model (LLM) agents, such as Codex, Claude Code, and OpenClaw, have become capable of planning and executing long-horizon tasks through repeated tool calls. This capability also creates new opportunities for prompt injection. Existing attacks either place the malicious objective in one explicit instruction, making it easy to detect, or distribute the intent across multiple execution stages, making successful completion unreliable. In this work, we propose ECLIPSE, a self-","url":"https://arxiv.org/abs/2608.30441","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00919","type":"paper","title":"Attesting Outputs and Delegation Ancestry in Multi-Agent AI Systems","authors":["Lifei Liu","Haoran Yu"],"year":2026,"abstract":"Multi-agent applications delegate work across independently operated deployers. After an incident, a verifier must answer two questions: which deployer released the reported bytes, and whether each cross-deployer edge was authorized. Credentials establish who may act, but need not bind them to later output bytes or prove both deployers authorized a dynamically created edge. We present a two-layer attestation design for dynamic delegation without a shared authority, public log, or precommitted wo","url":"https://arxiv.org/abs/2608.30387","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00920","type":"paper","title":"SIR: Self-improving Red-teaming for Compute Use Agents","authors":["Chen Xiong","Zhiyuan He","Pin-Yu Chen","Stjepan Picek","Tsung-Yi Ho"],"year":2026,"abstract":"Computer use agents (CUAs) are vision-language models that perceive a screen and act on a real operating system through mouse, keyboard, and terminal, and they are increasingly deployed to automate everyday digital tasks. Because they can be exposed to untrusted content while operating, they are vulnerable to indirect prompt injection (IPI), in which an adversary plants instructions in content the agent will read and redirects it toward actions that violate the user's intent. Existing CUA safety","url":"https://arxiv.org/abs/2608.30207","categories":["prompt-injection","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00921","type":"paper","title":"Zero-Knowledge Predicate Proofs Between AI Agents: A Measured, Cross-Protocol Gateway and the Source-Integrity Gap","authors":["Ashok Subbabhatta Gopalakrishna"],"year":2026,"abstract":"Multi-agent AI platforms move quickly from staging to production, but the way agents establish trust remains rudimentary: an agent either transmits raw data to a peer or accepts that peer's natural-language self-report that a value complies with policy. The first over-shares; the second is unverifiable and is exactly the channel prompt injection attacks. Prevailing responses emphasise identity, visibility, and post-hoc detection, and recent proposals for cryptographically enforced agent policy h","url":"https://arxiv.org/abs/2608.30083","categories":["prompt-injection","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00922","type":"paper","title":"AgentProv: Auditing Agentic LLM API Providers via Tool-use Policy Probes","authors":["Xun Wang","Bihe Zhao","Michael Backes","Franziska Boenisch","Adam Dziedzic"],"year":2026,"abstract":"Commercial LLM APIs advertise a specific foundation model, but the served backbone may be silently substituted, quantized, or wrapped, for example to save deployment costs. All existing audits decide backbone identity from the text-output channel, which is structurally fragile for agentic APIs because modern serving stacks (OpenAI, Anthropic, Gemini, Cloudflare Workers AI, LangGraph) discard text and expose only structured actions when the model calls a tool, and provider-injected system prompts","url":"https://arxiv.org/abs/2609.00052","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00923","type":"paper","title":"LongPIBench: A Long-Context Benchmark for Prompt Injection","authors":["Yupei Liu","Yuqi Jia","Neil Zhenqiang Gong","Jinyuan Jia"],"year":2026,"abstract":"Prompt injection attacks pose a serious security risk to large language models in real-world applications. However, existing prompt injection benchmarks primarily focus on short-context inputs, leaving the attacks and defenses in long-context settings largely unexplored. This gap leads to a substantial overestimation of the effectiveness of current defenses. In this paper, we bridge the gap by introducing LongPIBench, a long-context benchmark for prompt injection covering 4 realistic application","url":"https://arxiv.org/abs/2608.28411","categories":["prompt-injection","benchmarks","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00924","type":"paper","title":"Fully Unleashing the Multimodal Attacker: Meta-Adaptive Jailbreaking of Vision-Language Models","authors":["Benlei Cui","Shen Pang","Yuke Wang","Xuemei Dong","Yuwen Zhai","Jingqun Tang","Haiyang Yu","Hui Xue","Longtao Huang","Haiwen Hong"],"year":2026,"abstract":"The safety of large vision-language models is increasingly stress-tested by multimodal jailbreaks, yet existing attacks remain largely static at the meta level: template-based attacks freeze the image-text layout, while iterative attacks adapt only the image-text content with fixed attack strategies and frozen attacker parameters. We propose Meta-Adaptive Multimodal Jailbreaking (MAMJ), which instead optimizes the attacker itself along two axes: an attack strategy prompt (ASP) governing attack i","url":"https://arxiv.org/abs/2608.27531","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-00925","type":"paper","title":"Circuit Discovery Helps Detect LLM Jailbreaking: A Mechanistic Interpretability Study","authors":["Paria Mehrbod","Boris Knyazev","Guy Wolf","Eugene Belilovsky","Geraldin Nanfack"],"year":2026,"abstract":"Despite extensive safety alignment, large language models (LLMs) remain vulnerable to jailbreak attacks that bypass safeguards to elicit harmful content. While prior work attributes this vulnerability to safety training limitations, the internal mechanisms by which LLMs process adversarial prompts remain poorly understood. We present a mechanistic analysis of the jailbreaking behavior in a large-scale, safety-aligned LLM, focusing on LLaMA-2-7B-chat-hf. Leveraging edge attribution patching and s","url":"https://arxiv.org/abs/2608.27504","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00926","type":"paper","title":"ROPE: Routed Origin Policy Enforcement against Indirect Prompt Injection","authors":["Xinhang Ma","Chaowei Xiao","William Yeoh","Ning Zhang","Yevgeniy Vorobeychik"],"year":2026,"abstract":"Indirect prompt injection (IPI) plants instructions in the content a tool-using LLM agent reads, steering the agent into harmful tool calls. The strongest defenses are system-level, leveraging techniques such as task-conditional tool screening to prevent execution of malicious tools, and information-flow control to avoid tool execution with untrusted parameters. However, as agents grow more capable, users delegate more to automation. Consequently, tool execution sequences and parameter values ar","url":"https://arxiv.org/abs/2608.27496","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00927","type":"paper","title":"Why Are LLM Backdoor Defenses Fragmented? A Feature-Level Explanation with Sparse Autoencoders","authors":["Yizhe Zeng","Chenxu Niu","Wei Zhang","Hao Huang","Yunpeng Li","Dongxu Han","Dan Du","Cheng Hong","Hequn Xian","Yuling Liu"],"year":2026,"abstract":"Backdoor attacks pose a serious threat to large language models (LLMs), but existing defenses remain fragmented, failing to pro?vide unified defense against both dirty-label and clean-label attacks. To investigate why such fragmentation arises, we present the first systematic feature-level mechanistic analysis of LLM backdoors using sparse autoencoders (SAEs). Starting from a 2 x 2 comparison of clean and poisoned models on clean and triggered inputs, we trace backdoor-induced logit shifts to hi","url":"https://arxiv.org/abs/2608.30403","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00928","type":"paper","title":"A Black Box for Agentic Processes: Blockchain-Anchored Evidence for AI Agent Communication, Human Oversight, and GRC Audits","authors":["Arslan Brömme"],"year":2026,"abstract":"Autonomous AI agents increasingly communicate with other agents, invoke tools, exchange intermediate results, and request human approvals. These workflows create a new auditability problem: organizations must reconstruct what happened, when it happened, which agent or human was involved, which control or policy applied, and whether records were modified afterwards. Motivated by the 2026 OpenAI/Hugging Face incident, this position and architecture paper proposes a product- and vendor-neutral blac","url":"https://arxiv.org/abs/2609.04017","categories":["agentic-threats","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-00929","type":"paper","title":"Value-Preserving Architectures for Agentic AI Systems","authors":["Alessandro Pesare","Tommaso Dolci","Katja Hose","Emanuel Sallinger"],"year":2026,"abstract":"The emergence of agentic AI and LLM-based multi-agent systems (MAS) presents unprecedented opportunities for automating complex tasks, while simultaneously raising critical concerns about the preservation of fundamental human-centered values, such as privacy, fairness, and safety. Although software engineering has traditionally focused on functional correctness, the adoption of LLMs and AI agents into complex socio-technical systems has intensified the need for responsible software engineering a","url":"https://arxiv.org/abs/2609.03920","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00930","type":"paper","title":"HRGuard: Gating Relationship Manipulation in Multi-Turn Agentic AI Conversations","authors":["Pei-Sze Tan","Tasuku Igarashi","Isao Echizen"],"year":2026,"abstract":"Agentic AI assistants are increasingly used in everyday life. However, they may also be misused to support harmful manipulation in interpersonal relationships. This problem is role-sensitive. Requests from users who seek to manipulate others should be blocked. Users who seek protection from manipulation should instead receive supportive guidance. We study agentic relationship harm, which describes harm to human-human relationships that is mediated or assisted by AI agents. In multi-turn settings","url":"https://arxiv.org/abs/2608.25340","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00931","type":"paper","title":"Concepts for Securing Agentic AI Coding and the Terok Environment","authors":["Jiří Vyskočil","Franz Pöschel","Andreas Knüpfer"],"year":2026,"abstract":"Agentic AI is a fascinating new tool for software development. It is a huge step forward compared to \"conventional\" AI assisted coding, which in turn was a considerable breakthrough earlier. AI support through LLMs is a young and very fast-moving field. The \"conventional\" (non-agentic) flavor became useful and productive in early 2025 (around 18 months ago) and the agentic flavor followed in fall 2025 (approximately 9 months ago). Besides all its benefits and potential, it also carries some fund","url":"https://arxiv.org/abs/2608.22930","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00932","type":"paper","title":"Agentic AI for Safety-critical Multi-drone Systems: Challenges and Opportunities","authors":["Timothy Merritt","Alejandro Jarabo-Peñas","Juan Bravo-Arrabal","Maria-Theresa Bahodi","Anders Lyhne Christensen"],"year":2026,"abstract":"Multi-drone systems are increasingly positioned for safety-critical missions such as search and rescue (SAR) and critical infrastructure monitoring. Yet, real-world adoption remains constrained not only by autonomy performance, but by the difficulty of integrating agentic behavior into professional work: operators must understand, trust, and govern automation under uncertainty, time pressure, and accountability. This position paper synthesizes the ambitions and lessons from two ongoing efforts: ","url":"https://arxiv.org/abs/2608.21444","categories":["agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-00933","type":"paper","title":"Intelligent Prompt Construction for Large Language Models in Knowledge-based Visual Question Answering","authors":["Zhongjian Hu","Peng Yang","Dongmei Yang","Fengyuan Liu"],"year":2026,"abstract":"Large Language Models (LLMs) have demonstrated strong capabilities in knowledge-based Visual Question Answering (VQA). However, existing prompt construction methods are often rigid and fail to fully exploit the reasoning potential of LLMs. To address this limitation, we propose the Intelligent Prompt Construction Framework (IPCF), which equips an autonomous agent with the ability to dynamically generate task-specific prompts. IPCF consists of a planner and a toolbox: the planner, powered","url":"https://doi.org/10.1145/3833389","categories":["autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00934","type":"paper","title":"Indirect Prompt Injection in Municipal Document-Processing Copilots: Attacks, Defences and Harm","authors":["Jorge Cisneros-González","José Antonio Ondiviela García","Javier Sánchez-Soriano"],"year":2026,"abstract":"Public administrations are deploying large language model (LLM) assistants that process, summarise, classify and validate citizen-submitted documents. These copilots are exposed to indirect prompt injection: instructions hidden in manipulated documents that reach the model as if they were data.\r\nWe develop a municipal aid-procedure copilot and evaluate its robustness across six attack objectives, five delivery vectors, five defences and four LLMs, with 3,000 attack evaluations and 1,000 ","url":"https://doi.org/10.62161/sauc.v12.6399","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00935","type":"paper","title":"Cross-Lingual Indirect Prompt Injection Across Retrieval, Reranking, And Generation In Multilingual RAG","authors":["Fauzi Bondan Prihananto","Erlangga Bayu Yudho Prakoso","Aprilisa Arum Sari","Tomy Anugrah Islami"],"year":2026,"abstract":"External evidence can make retrieval-augmented generation (RAG) more informative, yet retrieved passages also provide a path for adversarial instructions to enter the model context. We examine that path in an English-Indonesian RAG system and track cross-lingual indirect prompt injection separately at retrieval, reranking, and generation. The experiment starts from 75 semantic items and evaluates every item in eight query/body/payload language combinations, for 600 paired trials. Qwen3-E","url":"https://doi.org/10.59435/jocstec.v4i3.809","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00936","type":"paper","title":"Data poisoning in LLM multiagent societies: Social proof drives collective decision failure in financial deliberation","authors":["Jeongsu Park","Yuji Lim","Geonwoo Kim","Taehyeon Yun","Moohong Min"],"year":2026,"abstract":"Abstract\n                  Large language model (LLM) multiagent systems are increasingly deployed for high‐stakes financial deliberation, but individually secure LLMs may become vulnerable when embedded in a peer‐to‐peer agent society. We study data poisoning in such societies using Moltbook, a deliberation test bed modeled after the agent‐only social platform. A single malicious agent representing just 14% of a seven‐agent society drives collective accuracy fro","url":"https://doi.org/10.4218/etrij.2026-0186","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00937","type":"paper","title":"Long-Horizon State Tracking in LLMs: Executing MD5 through a Deep Sequence of Dependent Tool Calls","authors":["Dheeraj Mohandas Pai","Lu Xian"],"year":2026,"abstract":"Long-horizon tasks remain uncommon in large language model (LLM) evaluation, and for a reason: when each step depends on the last, per-step accuracy that looks excellent in isolation decays catastrophically, as errors cascade and the end-to-end failure probability grows sharply with length. Existing agentic benchmarks report end-to-end success but confound this state-tracking difficulty with instruction interpretation, give no control group that isolates it, and are vulnerable to shortcuts such ","url":"https://arxiv.org/abs/2609.00012","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00938","type":"paper","title":"Capability-Gated Language Models: Security Composes, Utility Does Not","authors":["Patrikas Vanagas","Augustas Mačijauskas","Laurynas Lopata"],"year":2026,"abstract":"Deployed language model safeguards (safety fine-tuning, filtering, unlearning) vary by principal only outside the model weights: filters are reconfigured, tiers are multiplied, and artefacts are reissued; inside one set of weights every request meets the same model configuration. This motivates us to define capability-gated deployment: per-principal access control inside one set of weights, whose configurations form a lattice - meets accumulate a principal's restrictions and joins pool a coaliti","url":"https://arxiv.org/abs/2609.00445","categories":["guardrails","access-control"],"reviewed":false},{"id":"llmsec-2026-00939","type":"paper","title":"Does Reasoning Mitigate Backdoor Attacks? A Neuro-Symbolic Perspective","authors":["Marco Antonio Corallo","Andrea Agiollo","Mauro Conti","Alberto Giaretta"],"year":2026,"abstract":"Neuro-Symbolic (NeSy) AI has recently emerged as a novel paradigm to enable trustworthy AI, aiming at integrating sub-symbolic neural perception with grounded symbolic reasoning. The neuro-symbolic integration process that characterizes these models has been proven beneficial to achieve more transparent, explainable and efficient AI systems. Meanwhile, their properties under adversarial settings have been overlooked being frequently deemed robust-by-design. However, the neural-symbolic integrati","url":"https://arxiv.org/abs/2609.00464","categories":["data-poisoning","responsible-ai"],"reviewed":false},{"id":"llmsec-2026-00940","type":"paper","title":"TRIS: A Tri-Layer Retrieval Integrity Sieve Against Knowledge Poisoning","authors":["Muhaimin Bin Munir","Akib Jawad Ononto","Nazia Shehnaz Joynab","Bhavani Thuraisingham","Latifur Khan"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) grounds large language models in external corpora, but implicit trust in retrieved documents creates a critical attack surface: PoisonedRAG shows that a handful of crafted passages can dominate dense retrieval and steer generation toward attacker-chosen answers. We present the Tri-Layer Sieve, a middleware defense that sanitizes retrieved evidence through cross-embedding-space clustering with an independent judge model, structural filtering of trigger-payload","url":"https://arxiv.org/abs/2609.00470","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00941","type":"paper","title":"The Safeguard Worked. Is the LLM System Safer?","authors":["Pingyu Wu","Weiming Zhang","Nenghai Yu"],"year":2026,"abstract":"Safeguards in deployed LLM services are evaluated by refusal, attack success, and policy violation rates. Those rates characterize how a control performed on the requests it was tested on. A deployment has to answer a different question: how much help with harmful tasks the service still gives an attacker who keeps adapting or finds another way in. We determine what each reported result implies for that question, allowing results from different safeguard families to be compared under one deploym","url":"https://arxiv.org/abs/2609.00519","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00942","type":"paper","title":"Transferable End-to-End Optimization for Indirect Long-Term Memory Poisoning in LLM Agents","authors":["Chuanchao Zang","Jianing Wang","Wenyu Chen","Xiangtao Meng","Li Wang","Xinyu Gao","Zheng Li","Shanqing Guo"],"year":2026,"abstract":"Long-term memory can turn untrusted external content into persistent influence over an LLM agent's future decisions, creating the threat of indirect memory poisoning. A successful attack must survive a multi-stage pipeline comprising memory writing, retrieval, and utilization. Existing attacks largely rely on intra-stage optimization, optimizing individual stages in isolation while overlooking inter-stage coupling. Specifically, these stages impose different requirements on the same poisoning co","url":"https://arxiv.org/abs/2609.00523","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00943","type":"paper","title":"RISA: Response Inspection and Selective Actions for Refusal Calibration in Large Language Models","authors":["Wenhan Chang","Tianqing Zhu","Ping Xiong","Shiyi Liao","Wanlei Zhou"],"year":2026,"abstract":"Reliable refusal behavior requires Large Language Models (LLMs) to reject harmful prompts with only answering benign ones. Incorrect refusal behavior can either expose users to harmful responses or prevent users from obtaining useful answers. Training-time alignment improves refusal behavior by updating model parameters with safety data, but requires additional computation and training. In contrast, inference-time alignment aims to modify LLM behavior during inference without updating the underl","url":"https://arxiv.org/abs/2609.00790","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00944","type":"paper","title":"Membership Inference in Fine-tuned Diffusion Language Models via Token-level Memorization Asymmetry","authors":["Shengfang Zhai","Leo Marchyok","Yuling Shi","Huanran Chen","Yinpeng Dong","Jiaheng Zhang","Sanghyun Hong"],"year":2026,"abstract":"Diffusion language models (DLMs) have recently emerged as an alternative modeling paradigm to autoregressive LMs, offering advantages such as parallel generation and bidirectional context modeling. Despite growing interest in their generative capabilities, the privacy risks of DLMs remain underexplored. We identify a phenomenon termed token-level memorization asymmetry through theoretical analysis of diffusion training dynamics. Building on this finding, we propose Q-Skew, a quantile-weighted sk","url":"https://arxiv.org/abs/2609.00873","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-00945","type":"paper","title":"When Safety Routing Breaks: Understanding Alignment Fragility under Benign Fine-Tuning","authors":["Yitong Guo","Xiaoyi Chen","Siyuan Zhang","Xiaofeng Wang","Haixu Tang"],"year":2026,"abstract":"Benign fine-tuning severely weakens the safety alignment of large language models (LLMs), so we study why refusal behavior is so fragile. While prior work often attributes this failure to gradient conflict, we propose a fundamentally different Fisher-geometric explanation: safety Fisher is low-rank, and alignment makes the safety geometry flatter while preserving an output-routing pathway. After 100 benign fine-tuning examples, this pathway is selectively re-sharpened in output-side MLP modules,","url":"https://arxiv.org/abs/2609.01455","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00946","type":"paper","title":"OpenAgentFlow: Enabling System-Wide Safety Boundaries for Heterogeneous AI Agent Fleets","authors":["Dongsheng Chen","Xiangyu Zhao","Xin Yao","Xuetao Wei"],"year":2026,"abstract":"AI agents powered by large language models are evolving from isolated assistants into heterogeneous systems in which multiple agents, planners, tools, and execution backends operate over shared environments. In such settings, safety becomes a system-level action-governance problem: deciding whether a pending action should be committed given policy-relevant state accumulated across a session. Existing safeguards operate at fragmented boundaries, making it difficult to enforce shared policies over","url":"https://arxiv.org/abs/2609.00015","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00947","type":"paper","title":"What's in Your Agent's Context? Context Privilege Escalation Attacks against AI Agent Harness","authors":["Zichuan Li","Jian Cui","Ashley Chen","Xiaojing Liao","Luyi Xing"],"year":2026,"abstract":"Real-world, high-profile AI agent harnesses often rely on vendor-proprietary or opaque designs for context assembly, leaving the sources and underlying logic of assembled context poorly understood and the resulting security risks largely unexplored. In this paper, we present the first systematic analysis of context assembly designs in real-world AI agent harnesses. We study and uncover how an agent harness is designed to collect and assemble context from diverse sources, and identify a set of pr","url":"https://arxiv.org/abs/2609.01222","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00948","type":"paper","title":"Public-Sharing Labels and Verbatim Field Egress in an MCP-to-A2A Agent Configuration: A Controlled Multi-Model Study","authors":["Arpan Kumar Mahapatra"],"year":2026,"abstract":"Safety properties assessed separately for Model Context Protocol (MCP) tool use and Agent2Agent (A2A) delegation need not describe behavior when one agent uses both. We measure one such behavior in a single controlled MCP-to-A2A configuration: a testbed drives a real-model host across a local MCP and a local A2A leg into an ordered event trace scored by exact deterministic rules (no LLM judge), one restricted decision per trial. In a pre-specified, frozen three-arm design, each of 10 record scen","url":"https://arxiv.org/abs/2609.01693","categories":["tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00949","type":"paper","title":"Hearing the Whispers: Black-Box Membership Inference Attacks on Finetuned TTS Models","authors":["Kunlin Cai","Kaiyuan Zhang","Zihang Xiang","Jinghuai Zhang","Abeer Alwan","Fnu Suya","Yuan Tian"],"year":2026,"abstract":"Text-to-Speech (TTS) foundation models are increasingly fine-tuned on private datasets to synthesize highly personalized voices, introducing severe privacy risks by exposing both biometric identities and sensitive speech content. Existing black-box membership inference attacks (MIAs) follow a two-stage pipeline of query generation and representation engineering, both of which face unique challenges when adapted to TTS. For query generation, dual conditioning on synthesis text and reference speec","url":"https://arxiv.org/abs/2609.01723","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-00950","type":"paper","title":"Agent Memory Is a Surface for Endogenous Authorization Laundering","authors":["Tommaso Cerruti","Mika Okamoto","Ansel Kaplan Erol"],"year":2026,"abstract":"Long-running LLM agents rely on persistent memory to carry state across interactions, including permissions, restrictions, and revocations. When memory misrepresents this evolving authorization state, the agent's own records can grant authority that the underlying history never permitted, resulting in misaligned behavior without any external attacks. We term this failure endogenous authorization laundering, where spurious permissions written into memory lead to unauthorized actions as their prov","url":"https://arxiv.org/abs/2609.01836","categories":["agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00951","type":"paper","title":"WeaveMark: Robust and Scalable Multi-bit LLM Watermarking via Coded Payload Spreading","authors":["Gang-Hyun Park","Ju-Hyeong Lee","Hee-Youl Kwak","Dae-Young Yun"],"year":2026,"abstract":"Multi-bit watermarking for large language models (LLMs) enables content source tracing by embedding user-identifiable messages into generated text. Existing methods face a fundamental trade-off among extraction accuracy, text quality, and payload capacity. We propose WeaveMark, a robust and scalable multi-bit LLM watermarking scheme based on coded payload spreading. WeaveMark shifts this trade-off frontier by improving payload capacity through multi-bit-per-token spreading, improving extraction ","url":"https://arxiv.org/abs/2609.02177","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00952","type":"paper","title":"Agentic Settlement Protocol: An Application Profile for Refundable, Delayed-Fulfilment Agent Commerce on Stablecoin Rails","authors":["Behnam","Mohammadkhani","Atul Khekade","Ritesh Kakkad"],"year":2026,"abstract":"Autonomous agents can already pay per request: HTTP-native protocols such as x402 let an agent sign a stablecoin authorization and receive a resource in the same round trip. That model is atomic and final, which suits metered access and fails commerce: a purchase made on a person's behalf -- a service appointment, a physical order, a flight -- is large, frequently cancelled, and should not become the seller's money until delivery. We present the Agentic Settlement Protocol (ASP), an application ","url":"https://arxiv.org/abs/2609.02208","categories":["agentic-threats","access-control","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-00953","type":"paper","title":"Counter-GEO-Bench: Evaluating Defenses Against Information-Distorting Generative Engine Optimization","authors":["Bing Zheng","Zongyao Zhao","Wenming Yang"],"year":2026,"abstract":"Generative engine optimization (GEO) enables content producers to increase the visibility of their web pages in generative search engines, but the same techniques can deliver targeted misinformation when adversaries publish ordinary-looking GEO-optimized documents that victim large language models (LLMs) retrieve and synthesize into distorted answers. No existing benchmark evaluates defenses against this threat under controlled conditions. Therefore, we present Counter-GEO-Bench, a defense bench","url":"https://arxiv.org/abs/2609.02316","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00954","type":"paper","title":"PrimSynth: An Agentic Approach to Discover, Validate, and Synthesize Exploit Primitives for Linux Kernel Vulnerabilities","authors":["Pengfei Wang","Anying Chen","Danjun Liu","Xu Zhou","Wei Xie"],"year":2026,"abstract":"Linux kernel vulnerabilities are critical to downstream systems. Despite extensive research on automated kernel exploitation, a fundamental challenge remains the conceptual gap between abstract exploit strategies and concrete technical operations. To fill this gap, this paper introduces a systematic characterization that formalizes six classes of exploit primitives from logical capability to validatable effect. Then, an extended exploit strategy representation is proposed, which couples primitiv","url":"https://arxiv.org/abs/2609.02647","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00955","type":"paper","title":"ACLE-MCP: Attested Capability Leases for Execution-Time Trust in Remote LLM Tool Use","authors":["Zhiyang Ding","Yang Luo","Guangpu Chen","Qingni Shen","Zhonghai Wu"],"year":2026,"abstract":"Remote Model Context Protocol (MCP) services enable large language model agents to invoke external tools, but OAuth authorization alone does not ensure that a later tool call is executed by the provider-side workload that the relying party intended to trust. An endpoint may remain authorized even after execution shifts to a substituted workload, relies on stale appraisal state, reuses authority transferred from another sender, or traverses an undeclared downstream component. We call this problem","url":"https://arxiv.org/abs/2609.02690","categories":["access-control","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00956","type":"paper","title":"CodePoisonRAG: Knowledge Poisoning Attacks on Retrieval-Augmented Code Generation","authors":["Varun Gadey","Ziad Marey","Alexandra Dmitrienko"],"year":2026,"abstract":"Retrieval-Augmented Code Generation (RACG) improves LLM-based software development by retrieving external code artifacts, documentation, and patches, and incorporating them into the generation context. This reliance on external knowledge introduces a critical trust boundary: poisoned artifacts can influence generated code without modifying the underlying LLM. Prior work shows that selecting existing vulnerable examples can increase the general vulnerability rate of RACG outputs, but leaves open ","url":"https://arxiv.org/abs/2609.02774","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00957","type":"paper","title":"SafeEvolve: Harness-Policy Co-Evolution from Agent Experience for Safety Alignment","authors":["Qinghua Mao","Wanying Qu","Dadi Guo","Leitao Yuan","Qingyu Liu","Yu Li","Guanxu Chen","Yanwei Fu","Xi Lin","Xia Hu","Dongrui Liu"],"year":2026,"abstract":"The performance of LLM-based agents is jointly shaped by the base model and the harness used when interacting with the environment. This exposes them to safety risks in both harmful final responses and multi-step execution trajectories. Existing safety alignment mechanisms often rely on either external harness updates or policy optimization, yet applying either paradigm in isolation fails to bridge runtime control with intrinsic safety. We propose SafeEvolve, an experience-driven self-evolving f","url":"https://arxiv.org/abs/2609.02786","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00958","type":"paper","title":"PrivateHub: Contrastive Diffusion Model for Private Sensor-Intensive Environment Data Generation","authors":["Jiechao Gao","Yuandong Pan","Jie Wang","Michael Lepech","Bradford Campbell"],"year":2026,"abstract":"Sensor-intensive environments enable many intelligent services by inferring user applications from heterogeneous data streams. However, not all applications should be exposed: users want some activities to stay private. This creates a tension between inferring applications for useful services and preventing unwanted inference. Existing approaches such as differential privacy and rule-based filtering protect individual streams but cannot address the privacy risk from cross-sensor inference. We in","url":"https://arxiv.org/abs/2609.02958","categories":["membership-inference","differential-privacy"],"reviewed":false},{"id":"llmsec-2026-00959","type":"paper","title":"When Optimization Becomes Manipulation: Defending Generative Search against Malicious Generative Engine Optimization","authors":["Haozhang Li","Yangguang Shao","Xinjie Lin","Zhong Guan","Mi Zhou","Junzheng Shi"],"year":2026,"abstract":"This paper focuses on defending generative search engines against malicious Generative Engine Optimization (GEO), which rewrites web documents to match engines' citation preferences and thereby manipulates generated answers. Recent GEO methods have advanced from hand-crafted rewriting to automated and agentic optimization, substantially increasing the visibility of target documents in generated answers. However, defending against such manipulation poses two major challenges: attack documents rem","url":"https://arxiv.org/abs/2609.02964","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00960","type":"paper","title":"Privacy-Preserving Topology-Guided Safety for LLM-Based Multi-Agent Systems via Federated Graph Learning","authors":["Jinxi Yu","Eric Hanchen Jiang","Levina Li","Dong Liu","Zhi Zhang","Wenxiao Zhao","Yanxuan Yu","Kai-Wei Chang","Ying Nian Wu"],"year":2026,"abstract":"Topology-guided safeguards for LLM-based multi-agent systems (MAS) train a GNN over the inter-agent communication graph to localize risky agents and intervene on the topology---but they assume one operator can pool all labeled traces. Across organizations that assumption breaks: episodes contain private prompts, tool outputs, and proprietary workflows, and no silo alone sees the full attack distribution. We cast privacy-preserving MAS safeguarding as graph federated learning and instantiate FGLG","url":"https://arxiv.org/abs/2609.02967","categories":["guardrails","federated-learning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00961","type":"paper","title":"Rent-a-RAG: Embedding-Space Watermarks for Auditing Third-Party RAG","authors":["Alexandr Goultiaev Tolstokorov","Kyriakos Mouratidis","Javad Dogani","Nikolaos Laoutaris"],"year":2026,"abstract":"Third-party retrieval-augmented generation (RAG) marketplaces create a new auditing problem: data providers may license corpora to a RAG operator, yet later have no visibility into whether their documents are being reused without compensation. Auditing this misuse is difficult because the operator is non-cooperative, answers are paraphrased by the generator, and one response may combine evidence from many providers. We propose DirBucket, a provider-side semantic watermarking and black-box auditi","url":"https://arxiv.org/abs/2609.03749","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00962","type":"paper","title":"Inferring Hidden User Models from the Behavior of Personalized LLM Agents","authors":["Haoyang Li","Yaxin Xiao","Qingqing Ye","Huadi Zheng","Haibo Hu"],"year":2026,"abstract":"Recent personalized LLM agents increasingly transform information retained in memory into compressed or structured representations, which we call user models, to guide later decisions. When source wording is removed from the state reachable through the ordinary interface, these models are commonly treated as more privacy-preserving because direct memory-extraction attacks lose the text they target. Yet we argue that user models expose a new attack surface because an attacker can still recover th","url":"https://arxiv.org/abs/2609.03815","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00963","type":"paper","title":"Flip, Don't Shuffle: Watermarking LLMs at the Speed of Inference","authors":["Simone Ceppi","Ignacio Sanchez"],"year":2026,"abstract":"We introduce Stateless Bernoulli Watermarking (SBW), a new statistical watermark for Large Language Models that determines green list membership through independent per-token Bernoulli trials. Unlike KGW's vocabulary permutation or SynthID's multi-layer tournament, SBW requires only a single comparison per token against a counter-based random number generator, reducing membership complexity to $O(1)$ and enabling single-kernel execution with zero intermediate allocations. We prove that this form","url":"https://arxiv.org/abs/2609.03844","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00964","type":"paper","title":"Engineered Persuasion: Evaluating Personalized Pretexts in LLM-Generated Spear Phishing","authors":["Jerson Francia","Derek Hansen","Benjamin Schooley","Shydra Valynn Murray"],"year":2026,"abstract":"Large language models can insert workplace details into phishing pretexts at low cost, but those details may either support or undermine a message's credibility. We recruited 180 U.S. working adults to evaluate simulated, AI-generated phishing emails in a disclosed survey. The emails used four cumulative levels of information: workplace (Level 1); recipient name and job title; job responsibilities; and coworker/shared-project context (Level 4). Participants rated each message's convincingness fr","url":"https://arxiv.org/abs/2609.04410","categories":["social-engineering"],"reviewed":false},{"id":"llmsec-2026-00965","type":"paper","title":"Rethinking Indirect Prompt Injection as a Test-Time Search Problem","authors":["Duong M. Nguyen","Joon Sik Kim","Blazej Manczak","Vaikkunth Mugunthan"],"year":2026,"abstract":"We formulate indirect prompt injection as a test-time search over a task-dependent attack surface induced by the environment, user task, and injection task. To operationalize this formulation, we introduce an agentic attacker with a dedicated search harness that performs environment reconnaissance, structured reasoning over attack strategies, and adaptive evaluation using victim-agent feedback. Across heterogeneous tasks, we find that increasing attacker test-time compute improves vulnerability ","url":"https://arxiv.org/abs/2609.04495","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00966","type":"paper","title":"Repeat-After-Me: Black-Box Adaptive Visual Prompt Injection","authors":["Sizhe Chen","Yu-Lin Tsai","Ivan Evtimov","Kamalika Chaudhuri","Raluca Ada Popa","David Wagner","Arman Zharmagambetov"],"year":2026,"abstract":"Prompt injection is widely recognized as a major security threat to AI agents that interact with untrusted external data, such as websites, documents, and emails. Prior work has shown that, in the text domain, black-box prompt injection can achieve near-perfect attack success rates (ASRs). In the image domain, however, existing visual prompt injection methods are substantially less effective in attacking frontier commercial VLMs for materially harmful behavior. Achieving such outputs is hard bec","url":"https://arxiv.org/abs/2609.04533","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00967","type":"paper","title":"Forgetting Without Restarting: Execution-State Unlearning for Stateful LLM Agents","authors":["Chao Yao","Yangbo Wei","Zhen Huang","Junhong Qian","Chenle Chen","Shaoqiang Lu","Chen Wu","Lei He"],"year":2026,"abstract":"Long-running LLM agents are stateful: beyond the transcript they accrete compressed summaries, plaintext memory, pending tool plans, and, under every serving API, a KV cache. Yet today's \"forget\" operations delete a plaintext memory record and stop, leaving every artifact derived from the revoked information intact. We formalize execution-state unlearning: after a forget request, the agent must behave as if it had never observed the target. Modeling the runtime as a deterministic transition syst","url":"https://arxiv.org/abs/2609.04875","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00968","type":"paper","title":"TIER: Threat Implicitness Benchmark for Evaluating LLM Safety Behaviors","authors":["Thu-Hien Trinh-Thi","Hai-Yen Vong","Thanh-Ha Ung-Dung","Tram Ho"],"year":2026,"abstract":"Current LLM safety benchmarks largely rely on binary metrics, overlooking how models respond to harmful prompts with varying threat implicitness. We introduce TIER, a Threat Implicitness Benchmark for behavioral safety evaluation of LLMs. TIER covers four risk domains and four threat levels, from explicit harmful requests to sophisticated jailbreaks. Responses are assessed using a six-label behavior scale and two independent LLM judges. Experiments on six open-weight LLMs show that safety behavi","url":"https://arxiv.org/abs/2609.05117","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00969","type":"paper","title":"CONTINUITY: Security-Context Contracts for Composable LLM Agent Controls","authors":["Chris Zheng","Geng Yang"],"year":2026,"abstract":"LLM agent systems increasingly combine provenance tracking, authorization, policy enforcement, protocol adapters, and execution controls. However, individually correct security mechanisms do not necessarily compose into an end-to-end secure system: security-critical context may be dropped, widened, rebound, or reinterpreted as actions cross component boundaries. We identify this failure mode as security-context discontinuity and introduce CONTINUITY, a framework for verifiable composition of age","url":"https://arxiv.org/abs/2609.05269","categories":["agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00970","type":"paper","title":"Privacy Leakage in Federated Learning: Gradient-Based Client Identity Inference and Defenses for Inertial Sensing in Vehicular Edge Networks","authors":["Ali Akarma","Toqeer Ali Syed","Muhammad Khan","Qurat-ul-ain Mastoi","Adeel Ahmad"],"year":2026,"abstract":"As vehicular networks move toward 5G/6G edge intelligence, federated learning (FL) is widely promoted as a privacy-preserving way for vehicles and infrastructure to train shared models without exposing raw sensor data. Yet the updates clients transmit still leak enough information to identify who sent them, which threatens the anonymity that safety-critical V2X applications assume and adds to existing concerns over adversarial ML, model poisoning, and backdoor attacks. We study server-side clien","url":"https://arxiv.org/abs/2609.02971","categories":["data-poisoning","membership-inference","federated-learning"],"reviewed":false},{"id":"llmsec-2026-00971","type":"paper","title":"A Blind Trust, the Bloody Thrust: When Attacker-Controlled Hook Updates Steer AI Agent Harnesses towards Malicious Behaviors","authors":["Pengxun Li","Litian Zhang","Jianwei Hou","Shujiang Wu","Song Li","Zifeng Kang","Xi Zhang"],"year":2026,"abstract":"Modern AI agent harnesses expose lifecycle hooks that bind shell commands to runtime events such as session start, tool calls, and file edits. These commands run with host privileges yet ship as lifecycle-hook configuration and may fire at times the LLM never observes. We identify the lifecycle-hook update path, which harnesses trust blindly, as a new attack surface. Under a supply-chain threat model in which an attacker controls only plugin metadata and lifecycle-hook configuration, a benign ve","url":"https://arxiv.org/abs/2609.03884","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00972","type":"paper","title":"A Survey on Adversarial Attacks and Defenses for Diffusion Models Across Multiple Modalities","authors":["Ozgur Kara","Tarik Can Ozden","Furkan Horoz","Zeqian Long","Haotian Xue","Yipu Chen","Oguzhan Akcin","Yongxin Chen","James Matthew Rehg"],"year":2026,"abstract":"Diffusion models have become the dominant family of generative models in the visual domain. However, their widespread public availability enables misuse at scale, motivating a rapidly growing body of research on adversarial attacks and defenses. This survey provides, to our knowledge, the first unified review of this literature across three visual modalities: image, video, and 3D. We introduce a comprehensive, task-centric taxonomy: we first divide the literature by modality; within each modalit","url":"https://arxiv.org/abs/2609.05503","categories":["adversarial-examples","survey"],"reviewed":false},{"id":"llmsec-2026-00973","type":"paper","title":"Bait-and-Recover: Poisoning Internal Refusal Signals to Defend LLMs against White-Box Editing Jailbreaks","authors":["Tian Gao","Zhipeng Xie","Yuhao Wu","Junhua Liu","Xin Fang"],"year":2026,"abstract":"Open-weight large language models face a low-cost white-box threat from representation engineering attacks. Attackers can estimate refusal directions and search for projection-matrix edits that suppress safety alignment while preserving general capabilities, within minutes on a single GPU and without gradient-based training. We propose Bait-and-Recover, a weight-level defense that places a bait adapter where attackers read activations and a paired recovery adapter at the subsequent layer. Traine","url":"https://arxiv.org/abs/2609.05794","categories":["jailbreaking","data-poisoning","guardrails"],"reviewed":false},{"id":"llmsec-2026-00974","type":"paper","title":"SAFEGuard: Detect Optimization-Based Jailbreak Attacks Through Harmful Semantic Analysis and Fluency Measurement","authors":["Quoc Viet Vo","Trung Le","Damith C. Ranasinghe","Ehsan Abbasnejad"],"year":2026,"abstract":"Despite the significant efforts devoted to aligning large language models (LLMs) with human values and ensuring safe deployment, recent work has revealed that LLMs remain vulnerable to adversarial jailbreak attacks that can bypass safety guardrails and elicit harmful responses. Many defense methods are proposed to detect jailbreaks but they are limited in their effectiveness to counter wide-range optimization-based jailbreak mechanisms that can yield highly fluency-optimized or harmful semantic ","url":"https://arxiv.org/abs/2609.05850","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-00975","type":"paper","title":"Multimodal Resource-Exhaustion Attacks on Vision-Language Models via Joint Pixel-Prompt Optimization","authors":["Zhaoxiong Ni","Yatie Xiao","Chi-Man Pun","Fei Peng","Qingxiao Guan","Keke Tang"],"year":2026,"abstract":"Resource-exhaustion attacks against autoregressive vision-language models (VLMs) typically assume unimodal threat models, treating the image branch as the primary optimization surface while holding user-visible prompts fixed. Even recent loop-centric variants remain confined to this single-channel paradigm, leaving the exploitation of availability unexplored as a cross-modal optimization problem over jointly controllable input surfaces. We introduce Joint Pixel-Prompt Optimization (JPPO), the fi","url":"https://arxiv.org/abs/2609.05889","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00976","type":"paper","title":"From Review to Authorization: Key-Isolated Threshold Signing for LLM Agents","authors":["Yu Zheng","Qizhi Zhang"],"year":2026,"abstract":"Autonomous LLM agents can turn untrusted content into effectful actions such as payments and permission changes. If the same process interprets this content and controls a reusable signing credential, prompt injection can cross the judgment boundary and reach execution authority. We present KITA, a review-to-authorization architecture that keeps the user's personal secret signing key and every threshold signing-key share outside all LLM processes. Under threshold signature unforgeability and our","url":"https://arxiv.org/abs/2609.05901","categories":["prompt-injection","agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00977","type":"paper","title":"EvoSafeHarness: Evolving Model- and Domain-Specific Harnesses for Securing Agents","authors":["Nanxi Li","Yingzi Ma","Yulong Cao","Edward Suh","Bo Li","Dawn Song","Chaowei Xiao"],"year":2026,"abstract":"Large Language Model (LLM) agents are turning language into real-world effects, making safety necessary against both indirect prompt injections and direct harmful requests. System-level safety harnesses add an enforcement layer beyond model-level defenses, but existing harnesses are usually designed once by experts and applied across heterogeneous models and domains. Effective protection is deployment-dependent: models differ in how much enforcement they need before utility declines, while domai","url":"https://arxiv.org/abs/2609.05903","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-00978","type":"paper","title":"Structurally Close, Temporally Distant: Measuring Security Exposure in Long-Horizon LLM Agents","authors":["Md Jafrin Hossain","Nur Al Hasan Haldar"],"year":2026,"abstract":"Long-horizon LLM agents interact with untrusted content, persistent memory, external state, and sensitive tools. Existing analyses often characterize attacks by the number of execution steps between malicious input and a downstream action. We show that temporal remoteness can overstate security separation in stateful agents. We introduce a provenance-aware execution graph linking agent events through deterministic state, identifier, and tool provenance, and define \\emph{influence distance} $\\DI$","url":"https://arxiv.org/abs/2609.05911","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00979","type":"paper","title":"Evaluating Deep-Search Agents under Hierarchical Web Evidence Poisoning","authors":["Zhongan Bi","Qiwen Wang","Jianrong Jiang","Jigang Ding","Wenwen Xiong","Changhua Meng","Xuanang Gao","Kepeng Lin","Changjiang Jiang","Yiang Chen","Huan Yao","Wei Wang","Zhenyu Ma","Wenhui Dong"],"year":2026,"abstract":"Search-augmented LLM agents are increasingly used for consumer decisions, making them vulnerable to Generative Engine Optimization (GEO) poisoning. Existing benchmarks largely measure whether manipulated content is retrieved or endorsed, but do not track whether an agent verifies suspicious evidence, revises adopted claims, or recovers before producing its final recommendation. We introduce HAE-GEO, a benchmark that tracks the full trajectory from exposure to recovery under progressively more pe","url":"https://arxiv.org/abs/2609.06027","categories":["data-poisoning","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00980","type":"paper","title":"SCRIPTIOC-BENCH: A Benchmark for Recognizing Actionable Threat Intelligence from Script-Based Malware using LLMs","authors":["Hanna Kim","Jian Cui","Minkyoo Song","Hwanjo Heo","Seungwon Shin","Kimin Lee","Xiaojing Liao"],"year":2026,"abstract":"Script-based malware remains a prevalent attack technique. These scripts often contain indicators of compromise (IOCs) that provide actionable threat intelligence. However, statically recovering such indicators is challenging, as relevant values may be dispersed or transformed within code. Although large language models (LLMs) have shown promise in security analysis, their ability to recover IOCs from malicious scripts remains underexplored. We present SCRIPTIOC-BENCH, a benchmark for measuring ","url":"https://arxiv.org/abs/2609.06149","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00981","type":"paper","title":"It is Not Yet Another Tool: Creating and Deploying an Agentic AI Companion in a Security Operations Center","authors":["Kritan Banstola","Faayed Al Faisal","Duy Dao","Ryan Irving","Daniel Lende","Xinming Ou"],"year":2026,"abstract":"Security Operations Centers (SOCs) process large amounts of tickets, most of which are low-interest events not worthy of further investigation. The repetitive nature of this task and similarity of the vast amounts of tickets make it a prime candidate for generative AI-based automation. We created and deployed an agentic AI companion utilizing large language models through fieldwork within a SOC for over one year. The design of the SOC AI companion was driven by researchers' participation and int","url":"https://arxiv.org/abs/2609.06250","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00982","type":"paper","title":"CAPMAS: Capability-Based Delegation of Privileges in Multi-Agent Systems","authors":["Rasmus Moorits Veski","Rachid Guerraoui","David Froelicher"],"year":2026,"abstract":"Agentic systems require secure and efficient delegation of privileges across multiple collaborating agents. Existing approaches fall into two categories. Some propagate user identities directly to agents, obscuring accountability and creating persistent over-privilege risks that are amplified by the non-deterministic behaviour of AI agents. Others rely on continuous synchronization with a central Identity and Access Management (IAM) provider, introducing additional latency and communication over","url":"https://arxiv.org/abs/2609.06500","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-00983","type":"paper","title":"SRD-GUARD: A Defense Framework of LLMs via Semantic Rewriting and Joint Multi-Model Scoring for Latent Intent Exposure","authors":["Qi Wang","Chengcheng Wan","Jiangtao Wang"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed in safety-critical applications, yet jailbreak attacks can conceal harmful intent through role-playing, fictional scenarios, or seemingly benign motivations. Existing inference-time defenses may miss disguised attacks or excessively refuse legitimate requests. We propose SRD-GUARD, a parameter-free, black-box defense framework that exposes concealed intent through semantic rewriting and consensus-based risk assessment. Given an input prompt,","url":"https://arxiv.org/abs/2609.06540","categories":["jailbreaking","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-00984","type":"paper","title":"A Translational Note on AI Safety Evaluation","authors":["Madhava Gaikwad"],"year":2026,"abstract":"Recent studies report that automated red-teaming finds more vulnerabilities, at lower cost, than human red-teaming on standard AI safety benchmarks, and some read this as evidence that human evaluators are becoming dispensable. The comparison measures one thing and the conclusion claims another. A benchmark measures how thoroughly an attacker searches a predefined set of harms, fixed in advance by the developers, and a harm left out of that set is invisible to any attacker working inside it, aut","url":"https://arxiv.org/abs/2609.06573","categories":["red-teaming","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00985","type":"paper","title":"MechAudit-40: White-Box Auditing across 40 LLM Attack Mechanisms","authors":["Zhen Guo","Shanghao Shi","Shamim Yazdani","Ning Zhang","Reza Tourani"],"year":2026,"abstract":"While LLM attacks span prompt optimization, multi-turn context manipulation, retrieval poisoning, and model backdoors, white-box defenses are typically evaluated on isolated attack families. Consequently, whether heterogeneous attacks leave internal representation shifts that generalize to unseen threat mechanisms remains unknown. We present MechAudit-40, a systematic evaluation of 40 attack mechanisms across five open-weight model architectures. Threat-specific success criteria, 100,000 matched","url":"https://arxiv.org/abs/2609.06612","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-00986","type":"paper","title":"A Trustworthy Watermarking Framework for LLM-Generated Food Safety Content","authors":["Zhongli Fang","Yiran Chen","Lingyun Zhang","Yu Liu","Ping Chen","Xiaoyan Sun","Jun Dai"],"year":2026,"abstract":"Large language models are transforming many industries with their text generation abilities. However, their outputs can be easily tampered with, creating serious risks in critical areas such as food safety reporting. To protect the integrity and traceability of AI-generated content, this paper introduces ToSS (Token Oriented Repartitioning and Strategic Selection), a reliable authentication method using adaptive dual watermarking. The key innovation of ToSS is its dual watermark encoding approac","url":"https://arxiv.org/abs/2609.06708","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-00987","type":"paper","title":"A Novel Semantic Manifold Alignment Attack against Embedding-to-Embedding Obfuscation in Privacy-Preserving LLMs","authors":["Sicong Li","Lingfeng Yao","Xingke Yang","Ke Tu","Chenhao Wu","Hao Wang","Jiang Liu","Phone Lin","Xin Fu","Miao Pan"],"year":2026,"abstract":"With the widespread applications of large language models (LLMs), privacy-preserving inference has become increasingly essential for sensitive queries. To balance privacy and utility, a series of lightweight obfuscation approaches has recently been proposed, where users locally transform plaintext embeddings into the fixed ciphertext ones. While such Embedding-to-Embedding Obfuscation (E2EO) schemes demonstrate considerable resilience against traditional token frequency and embedding inversion a","url":"https://arxiv.org/abs/2609.06749","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-00988","type":"paper","title":"AURA-Eval: Evaluation Framework for Acting Under Risk Awareness in LLM Agent Trajectories","authors":["Ruoxi Shang","Christina-Maria Androna","Orfeas Menis Mastromichalakis","Yu Feng","Aniruddhan Ramesh","Rico Angell","Shang Hong Sim","Chrysoula Zerva","Emmanouil Koukoumidis"],"year":2026,"abstract":"LLM agents operate in workflows where unsafe actions can have real consequences. Existing safety evaluations often reduce behavior to a single score, obscuring risk recognition, pre-action detection, and safe task completion when a safe solution exists. We introduce AURA-Eval, a framework combining controlled augmentation with granular diagnosis of behavior in tool-use trajectories. Its pipeline identifies safety-critical decision points, generates controlled variations, and constructs counterpa","url":"https://arxiv.org/abs/2609.06783","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00989","type":"paper","title":"Skynet: Workflow-Level Anomaly Detection for Agentic AI via Semantic and Structural Modeling","authors":["Chaoyu Zhang","Hexuan Yu","Heng Jin","Shanghao Shi","Ning Zhang","Yi Shi","Yulia R. Gel","Y. Thomas Hou","Wenjing Lou"],"year":2026,"abstract":"Agentic AI systems execute complex tasks through long-horizon workflows of planning, tool use, and multi-agent coordination. Task failures in these systems often originate from a single step, such as an injected prompt or a flawed plan, and are then amplified through downstream dependencies as the corrupted step propagates across many subsequent agents and tool calls. Existing defenses either target a specific class of attacks or failures, or inspect individual prompts and steps in isolation. Bo","url":"https://arxiv.org/abs/2609.06835","categories":["agentic-threats","monitoring-detection","agent-architecture","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-00990","type":"paper","title":"MOLE: Detecting Insider Threats in AI Agents","authors":["Aashiq Muhamed","Virginia Smith"],"year":2026,"abstract":"Model misalignment, prompt injection, or operator misuse could lead AI agents operating frontier-lab accounts to exfiltrate model weights, poison training data, or weaken release gates. Existing benchmarks do not test whether defenders can detect this activity among routine work under a limited review budget. We introduce MOLE, an open benchmark of 150 AI-operated accounts sharing 9 stateful services over 30 workdays, with 12 threats and 8 corpora from four models totaling roughly 20 billion tok","url":"https://arxiv.org/abs/2609.06966","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00991","type":"paper","title":"AgentDrift: A Step-Labeled Benchmark of Injection-Hijacked LLM Agent Trajectories","authors":["Asif Pinjari","Mithun Paul Saint-Germain"],"year":2026,"abstract":"LLM agents complete tasks by issuing sequences of tool calls, and every observation they read is a channel through which an indirect prompt injection can enter. A successful injection has a characteristic shape when the trajectory is read in order: a benign prefix gives way to actions that serve the attacker rather than the user. Existing benchmarks measure whether such attacks succeed against live agents, and existing guard models judge a trace as a whole; no public corpus labels, step by step,","url":"https://arxiv.org/abs/2609.06972","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00992","type":"paper","title":"AgentLeak: Cloning Stronger LLM Agent Capabilities onto Weaker Agents Beyond Skill Stealing","authors":["Xiaoting Lyu","Yuhong Wu","Yufei Han","Shichang Liu","Liang Zhang","Bin Wang","Bin Wang","Xiaobo Ma","Wei Wang"],"year":2026,"abstract":"Large language model (LLM) agents increasingly achieve long-horizon tasks by combining foundation models with explicit skills and implicit procedural knowledge acquired through execution. The resulting task-solving capabilities have become valuable proprietary assets, raising a new security question: can a substantially weaker attacker-controlled agent acquire the capabilities of a stronger proprietary agent through limited black-box interaction? Existing skill-stealing attacks recover explicit ","url":"https://arxiv.org/abs/2609.07131","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-00993","type":"paper","title":"Benchmarking LLMs for Threat Level Determination","authors":["Han Wang","Murathan Kurfalı","Alfonso Iacovazzi"],"year":2026,"abstract":"The fast progress of large language models (LLMs) opens new opportunities in the management of cyber threat intelligence, but their reliability for operational tasks remains unclear. In this work, we benchmark LLMs on the task of threat level determination. First, we construct a curated dataset derived from publicly available MISP OSINT feeds. Next, we design a tailored prompt to systematically compare eight different LLMs under zero-shot conditions. Finally, we apply supervised fine-tuning on e","url":"https://arxiv.org/abs/2609.07582","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-00994","type":"paper","title":"VEX-Bench: Benchmarking LLM Agents for Assessing Exploitability of Software Supply Chain Vulnerabilities","authors":["Jiahao Shi","Edward Tsien","Yifeng Di","Hongjiao Zhang","Yuan Tang","Ronit Dey","Ilona Shishov","Gal Netanel","Zvi Grinberg","Vladimir Belousov","Bat-Zion Rotman","Ilan Pinto","Tianyi Zhang"],"year":2026,"abstract":"The software supply chain has become an increasingly exposed attack surface because of its reliance on intricate yet fragile dependencies. Existing defenses such as GitHub Dependabot often raise many false alerts because their coarse-grained matching cannot determine whether a vulnerable dependency is actually exploitable. Security analysts typically spend substantial time assessing vulnerability exploitability case by case. Recent LLM agents have emerged as promising candidates for this task gi","url":"https://arxiv.org/abs/2609.08040","categories":["supply-chain-attacks","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-00995","type":"paper","title":"ACEA: An Adversarial Co-Evolution Arena for Head-to-Head Red-Team and Blue-Team LLM Testing","authors":["Yi Ting Shen","Kentaroh Toyoda","Alex Leung"],"year":2026,"abstract":"Automated red-team attacks and blue-team defenses for large language models (LLMs) are advancing quickly. However, attackers and defenders are built and tested in isolation, and the resulting scores are hard to trust. To tackle this, we present ACEA (Adversarial Co-Evolution Arena), a platform that connects a pluggable red-team adapter and a pluggable blue-team adapter to a shared target LLM and scores their attack and defense rates with an LLM judge. ACEA contributes four components. First, a p","url":"https://arxiv.org/abs/2609.08256","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-00996","type":"paper","title":"Structural Jailbreaks Generalize but Do Not Compound: A cross-provider and multilingual study of Involuntary In-Context Learning","authors":["Tejasvi C. Addagada"],"year":2026,"abstract":"Aligned language models fail under two independent pressures: the structural jailbreak class recently formalized as Involuntary In-Context Learning (IICL), which reframes a harmful request as the final missing cell of a data-labeling task completed by pattern rather than judged as content; and the erosion of safety alignment outside English. A natural hypothesis is that these compound. We test it directly. Using a deterministic IICL operator and a StrongREJECT-style rubric judge, we red-team two","url":"https://arxiv.org/abs/2609.08373","categories":["jailbreaking","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-00997","type":"paper","title":"An Evidence Model for Agentic Processes: Evidence Claims, Trust Assumptions, and Policy Assessment","authors":["Arslan Brömme"],"year":2026,"abstract":"Agentic AI systems increasingly exchange messages, invoke tools, request approvals, hold structured decision sessions, and modify shared artifacts. Logs and anchors can make selected records tamper-evident, but they can also mislead if their evidentiary meaning is implicit: a hash does not establish semantic truth, a signature does not establish authorization, and an external anchor does not establish capture completeness. This paper proposes an evidence claim model for agentic processes. It dis","url":"https://arxiv.org/abs/2609.08481","categories":["agentic-threats","access-control"],"reviewed":false},{"id":"llmsec-2026-00998","type":"paper","title":"MemSentry: A Framework for Detecting Persistent Memory Poisoning in Agentic AI","authors":["Ayan Roy","Kaustuvi Basu"],"year":2026,"abstract":"Agentic AI systems with persistent memory introduce a distinct attack surface known as memory poisoning, in which adversarially crafted content is stored in long-term memory and subsequently influences future agent behavior. Such attacks can suppress security alerts, facilitate privilege escalation, alter trust relationships, or override security policies without modifying the underlying model weights or system prompts. To address this threat, we present MemSentry, a formal, configuration-driven","url":"https://arxiv.org/abs/2609.08747","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-00999","type":"paper","title":"Benchmark Scores Are Pipeline-Dependent: A Reliability Audit of Cybersecurity LLM Benchmarks","authors":["Aymene Berriche","Cathrine Shalby","Mohannad Alhanahnah","Yazan Boshmaf"],"year":2026,"abstract":"Large language model (LLM) benchmarks are often treated as fixed datasets with stable scores, yet their outcomes depend on configurable evaluation pipelines. We audit eight cybersecurity benchmarks across 10 proprietary, open-weight, and cybersecurity-specialized LLMs. By modeling benchmarks as measurement pipelines, we identify 15 systematic failure modes and show that a single pipeline choice can change a model's score by more than 80 percentage points and substantially alter model rankings. A","url":"https://arxiv.org/abs/2609.08765","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01000","type":"paper","title":"PrivEscalate: Measuring and Augmenting the Threat of LLM-Automated Linux Privilege Escalation","authors":["Yixuan Liu","Zilong Zhen","Yin Wu","Yi Li"],"year":2026,"abstract":"As Large Language Model (LLM) agents increasingly automate offensive operations across the cyber kill chain, their efficacy in complex local post-exploitation tasks remains inadequately quantified. Among these, Linux privilege escalation is a key step between initial access and full system compromise. However, existing evaluations for this task are limited by small sample sizes (fewer than 15 scenarios), lacking the scale to compare model capabilities under executable verification. To address th","url":"https://arxiv.org/abs/2609.09087","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01001","type":"paper","title":"AgentHijack: Visual Patch Attacks on Multimodal Computer-Use Agents","authors":["Zhihao Liu","Hongyu Sun","Zhiyuan Fu","Xiaonan Duan","Jice Wang","Shangru Zhao","Weizhi Meng","Wuxin Yang","Yangfan Zhou","Yuqing Zhang"],"year":2026,"abstract":"This paper presents an end-to-end evaluation framework for image-triggered command injection against computer-use agents (CUAs). The goal is to test whether a local visual patch can induce verifiable environmental consequences along the full chain of screenshot input, VLM generation, action parsing, and environment execution. We train and deploy patches on author-controlled GitHub Pages pages and a locally deployed CSDN clone, and evaluate them in real environments across five open-source or pub","url":"https://arxiv.org/abs/2609.09212","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01002","type":"paper","title":"In RAG We Trust? Measuring Robustness of Retrieval-Augmented Generation Under Document Poisoning","authors":["Iliano Fasolino"],"year":2026,"abstract":"Retrieval-augmented generation (RAG) grounds a language model in retrieved documents, which reduces hallucination but creates a new attack surface: if retrieved text is tampered with, the model may repeat the falsehood. We study how much a small quantized model, Llama 3.1 8B, degrades when a fraction of its retrieved context is poisoned. Three corruption strategies are tested, entity swap, number swap, and negation, each applied to zero, one, two, or three of the three retrieved passages, over a","url":"https://arxiv.org/abs/2609.09243","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01003","type":"paper","title":"An Experimental Evaluation of Multimodal Prompt Injection Attacks on Agentic AI Frameworks","authors":["Viet K. Nguyen","Mohammad I. Husain"],"year":2026,"abstract":"Agentic AI frameworks let a language model plan, keep memory, and call tools that reach real files, mail, and services. Most of these agents also read images, which gives an attacker a way to put text into the agent's context without going through the user. We present MMPIBench, a reproducible benchmark that measures what happens next. It delivers a fixed set of attacks through six visual carriers (OCR text, overlays, EXIF metadata, QR codes, fake interfaces, and hybrids) and records how far eac","url":"https://arxiv.org/abs/2609.09404","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01004","type":"paper","title":"Audio Deepfake Detection Using Temporal Coherence Analysis","authors":["Justin D. Norman","Sarah Barrington"],"year":2026,"abstract":"The proliferation of AI-generated audio (so-called \"deepfake\" audio) poses significant threats to information integrity, from voice cloning fraud to synthetic music copyright disputes. We present a temporal coherence analysis framework built upon Contrastive Language-Audio Pretraining (CLAP) embeddings that spans speech, instrumental music, and music with vocals. By computing pairwise cosine similarities between audio segment embeddings and extracting statistical features from the resulting dist","url":"https://arxiv.org/abs/2609.09489","categories":["social-engineering"],"reviewed":false},{"id":"llmsec-2026-01005","type":"paper","title":"An Efficient and Effective Agentic Group Shilling Attack on Recommender Systems","authors":["Quoc Viet Nguyen","Trinh Pham","Viet Huynh","Hongzhi Yin","Quoc Viet Hung Nguyen","Bay Vo","Thanh Tam Nguyen"],"year":2026,"abstract":"Recommender systems have become core infrastructure for modern online platforms, personalizing content at scale and strongly influencing what users see, click on, and purchase. However, this dependence on user interaction also exposes them to shilling attacks, where malicious actors can inject fake profiles to distort item rankings and control visibility. Existing attacks often rely on target-specific fine-tuning or fixed profile templates, making them either difficult to adapt to different vict","url":"https://arxiv.org/abs/2609.09551","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01006","type":"paper","title":"Arbitrary Cipher Attacks Against Large Language Models Do Not Require Fine-Tuning","authors":["Thomas Rivasseau"],"year":2026,"abstract":"Large language model safety and security research is preoccupied with, among other things, detecting and preventing jailbreak attacks: alignment bypasses that allow an adversarial user to elicit unwanted or harmful outputs from models. Arbitrary cipher, or covert communication, attacks are one such type of jailbreak and have previously been demonstrated against the fine-tuning APIs of commercial models. In these attacks, target models are trained on a corpus of encrypted harmful questions and re","url":"https://arxiv.org/abs/2609.09553","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01007","type":"paper","title":"Watermarks Without Verification: AI Text Watermarking After the EU AI Act","authors":["Alexander Nemecek","Vipin Chaudhary","Erman Ayday"],"year":2026,"abstract":"On August 2, 2026, the obligations of Article 50 of the EU AI Act took effect, requiring generative AI providers to mark the content their systems produce and ensure it can be detected as AI-generated. Days later, Anthropic disclosed that every Claude model released after that date embeds a watermark based on SynthID-Text in all generated text, enabled by default with no user opt-out; Google has deployed SynthID-Text in Gemini since 2024. Users objected that the watermark degrades quality, parti","url":"https://arxiv.org/abs/2609.09604","categories":["watermarking","risk-frameworks"],"reviewed":false},{"id":"llmsec-2026-01008","type":"paper","title":"How Fragile Is Safety Alignment at Frontier Scale? A Single-Direction Attack on a 320B MoE","authors":["Yi Shi","Tanyu Chen","Kai Shen"],"year":2026,"abstract":"Directional ablation removes an aligned language model's ability to refuse by projecting a single \"refusal direction\" out of the weights that write the residual stream. It needs no gradient-based training and no optimization, only a few hundred contrastive prompts, which makes it the canonical white-box attack on open-weight alignment. However, it has been established only on dense models up to roughly 70B parameters. We study whether it survives the shift to frontier mixture-of-experts (MoE) mo","url":"https://arxiv.org/abs/2609.09793","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01009","type":"paper","title":"CS-Guard: Benchmarking LLM Guardrails for Code Generation Security","authors":["Jinyang Li","Mingyu Guo","Hung X. Nguyen"],"year":2026,"abstract":"Large language models (LLMs) have been ex- ploited to generate malware, but the effective- ness of guardrails for code generation secu- rity remains unclear. We introduce CS-Guard, the first benchmark to systematically evalu- ate guardrails for code generation security. It covers 1) text-to-code generation with 1000 high-quality malware-generation prompts, 7 jailbreak attacks, and a novel fictional scenario attack (FSA) that embeds malicious intent in a legitimate fictional software-development ","url":"https://arxiv.org/abs/2609.09798","categories":["jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01010","type":"paper","title":"What Makes Adversarial Examples Transfer Across Deepfake Detectors?","authors":["Rafael M. Mamede","Pedro C. Neto","Ana F. Sequeira"],"year":2026,"abstract":"Deepfake detectors remain vulnerable to transfer-based black-box attacks, in which adversarial examples are generated on a source surrogate model and transferred to a target model, unknown to the attacker. Yet how source--target compatibility shapes attack success remains poorly understood. Prior studies evaluate limited detector pools and rarely disentangle architectural from training factors. We conduct a controlled evaluation of adversarial transferability across 60 detectors spanning six bac","url":"https://arxiv.org/abs/2609.10002","categories":["adversarial-examples","social-engineering"],"reviewed":false},{"id":"llmsec-2026-01011","type":"paper","title":"Beyond Training: A Feasibility Taxonomy for Inference-Time AI Governance","authors":["Samar Ansari"],"year":2026,"abstract":"Compute governance today is a governance of training: the thresholds, reporting requirements, and frontier-AI regimes now in force attach to training compute and treat the trained model as the regulatory unit. That picture is incomplete: capability increasingly migrates to the deployment stage through inference-time scaling, agentic scaffolding, and compression onto consumer hardware. This paper asks which mechanisms are available once the regulatory object shifts from the training run to the in","url":"https://arxiv.org/abs/2609.10105","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01012","type":"paper","title":"Understanding the Security Boundary of Obfuscation-based On-Device LLM Protection","authors":["Hanyi Zhou","Chenyang Li","Yuanzhe Pang","Ke Xu","Mingwei Xu","Zhuotao Liu"],"year":2026,"abstract":"Trusted Execution Environments (TEEs) offer a promising mechanism for safeguarding the intellectual property of on-device Large Language Models (LLMs). To overcome the inherent computational bottlenecks of TEEs, existing TEE-Shielded LLM Partition (TSLP) methods apply efficient obfuscation schemes to computationally intensive layers, offloading them to external GPUs while retaining only lightweight operations within the TEE. Although a growing body of TSLP-based approaches has emerged, these def","url":"https://arxiv.org/abs/2609.10117","categories":["guardrails","confidential-computing"],"reviewed":false},{"id":"llmsec-2026-01013","type":"paper","title":"TrajMark: Ownership Attribution and Segment-Level Tamper Localization for Coding-Agent Trajectories","authors":["Bokang Zeng","Zheng Gao","Xiaoyu Li","Xiaoyan Feng","Jiaojiao Jiang"],"year":2026,"abstract":"Watermarking the final patch produced by a coding agent provides provenance evidence for the submitted artifact, but does not authenticate the visible process that produced it. Behavioral watermarking methods primarily provide a global detection or identifier-recovery signal, so a locally edited trajectory may retain sufficient ownership evidence without revealing which protected region has become inconsistent. To address this limitation, we propose TrajMark, a training-free, symmetric-key, visi","url":"https://arxiv.org/abs/2609.10416","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01014","type":"paper","title":"When Passing Tests Hides Vulnerabilities: An Empirical Study of Silent Failures in Agentic Systems","authors":["Wenji Bai","Muhammad Waseem","Zeeshan Rasheed","Jaakko Peltonen","Pekka Abrahamsson"],"year":2026,"abstract":"LLM-based agents for automated code repair have received significant attention in recent years from both research and software engineering practice perspectives. However, limited attention has been paid to patches that pass syntactic and functional verification but still retain or introduce security vulnerabilities. The aim of this research is to systematically identify and categorize such silent failures in LLM-based agentic code repair. We conducted an empirical study using 1,030 valid executi","url":"https://arxiv.org/abs/2609.10548","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01015","type":"paper","title":"An Empirical Measurement of Jailbreaking Evaluators","authors":["Yujie Mu"],"year":2026,"abstract":"Expert evaluation of jailbreak responses is costly and difficult to scale, so the community increasingly relies on automated evaluators to determine whether an attack succeeds. However, jailbreak studies typically validate their chosen evaluator independently, repeatedly spending resources on similar evaluation efforts while making results across papers difficult to compare. Different evaluators also encode different definitions of jailbreak success, meaning that reported attack strength and app","url":"https://arxiv.org/abs/2609.10594","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01016","type":"paper","title":"Adaptive Diffusion Freezing: Privacy-preserving Diffusion Models Against Membership Inference Attacks","authors":["Jialu Guo","Xiao Han","Junjie Wu"],"year":2026,"abstract":"Diffusion models have achieved remarkable success in generative tasks across various areas, however their training process raises significant privacy concerns, particularly under membership inference attacks (MIAs). Prior studies on privacy-preserving of diffusion models fail to balance privacy, utility, and efficiency. To address this gap, we propose a novel framework of privacy-preserving diffusion models, Adaptive Diffusion Freezing (ADF), which can defend against MIAs with better trade-off. ","url":"https://arxiv.org/abs/2609.10608","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01017","type":"paper","title":"Black-Box Membership Inference via Word-Level Probability Estimation","authors":["Shengjie Niu","Yeheng Ge","Jian Huang"],"year":2026,"abstract":"Membership inference attacks (MIAs) have emerged as critical tools for auditing privacy risks in large language models (LLMs), aiming to determine whether a given text was included in a model's training corpus. However, most existing MIAs require access to per-token logits or probabilities, making them inapplicable in practice to proprietary LLMs that expose only textual continuations. To address this underexplored setting, we propose Word-level Probability MIA (WPMIA), a statistically principle","url":"https://arxiv.org/abs/2609.10611","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01018","type":"paper","title":"Understanding In-Context Multimodal Jailbreaks via Posterior Reweighting","authors":["Xu Zhang","Dev Mistry","Xiang Xu","Ren Wang"],"year":2026,"abstract":"In-context learning (ICL) jailbreaks reveal a critical vulnerability in multimodal large language models (MLLMs): harmful demonstrations in the prompt can induce unsafe outputs without modifying model parameters. Despite extensive empirical evidence, existing work lacks a principled understanding of why such jailbreaks reliably succeed or how their effectiveness scales with context composition. We propose a posterior reweighting framework that models a safety-aligned MLLM as implicitly operating","url":"https://arxiv.org/abs/2609.10613","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01019","type":"paper","title":"Architecting the Secure AI-SOC: A Neurosymbolic Framework for Pipeline Integrity and Threat Mitigation","authors":["Anna Gazani","Spyridon Kounoupidis","Panagiotis Katsaros","Nikolaos Kekatos","Grigorios Tsoumakas","Georgios Koutidis"],"year":2026,"abstract":"The integration of Large Language Models (LLMs) into Security Operations Centers (SOCs) streamlines threat intelligence but introduces critical vulnerabilities, notably indirect prompt injection via log poisoning. Adversaries exploit this vector to execute multistep ``promptware'' kill chains by embedding malicious payloads within system logs to hijack the LLM's operational logic. Securing this pipeline presents a dichotomy: deterministic defenses are computationally efficient yet semantically b","url":"https://arxiv.org/abs/2609.10707","categories":["prompt-injection","data-poisoning","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01020","type":"paper","title":"No-Box Vulnerability Analysis: Description-only Detection of Indirect Prompt Injection Vulnerabilities in MCP Servers","authors":["Zehua Zhang","Jie Hu","Pratham Hegde","Aditya Maheshbhai Gabani","Souradip Nath","Yibo Liu","Siyu Liu","Hongkai Chen","Hulin Wang","Zhuoer Lyu","Chang Zhu","Divij Handa","Yan Shoshitaishvili","Tiffany Bao","Ruoyu Wang","Adam Doupe"],"year":2026,"abstract":"Conventional vulnerability analysis relies on either system access or dynamic interaction, all of which may be unavailable to third-party analysts auditing closed-source, remotely hosted, critical in situ systems, or commercially gated software. Therefore, we propose a new paradigm of no-box vulnerability analysis in which neither access nor runtime interaction is available, and only functionality metadata is available. Such metadata defines the intended behavior of the system, including its inp","url":"https://arxiv.org/abs/2609.10854","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01021","type":"paper","title":"A2ABreak: Systematic Security Analysis of the A2A Protocol","authors":["Alireza Lotfi","Mirza Masfiqur Rahman","Imtiaz Karim","Elisa Bertino"],"year":2026,"abstract":"The Agent2Agent (A2A) protocol, now governed by the Linux Foundation, is an open standard that enables autonomous AI agents to discover, authenticate with, and delegate tasks to one another across organizational boundaries. Designed to complement the Model Context Protocol (MCP) for tool integration, A2A is rapidly emerging as the horizontal communication layer of the multi-agent ecosystem. Yet the protocol's security has received no systematic analysis. This paper presents A2ABreak, the first r","url":"https://arxiv.org/abs/2609.10871","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01022","type":"paper","title":"DriftNet: A Dual-Head Trajectory Transformer for Detecting and Localizing Prompt Injection in LLM Agents","authors":["Asif Pinjari","Mithun Paul Saint-Germain"],"year":2026,"abstract":"When an indirect prompt injection succeeds against an LLM agent, the compromise is visible in the agent's own behavior: a benign prefix of tool calls, a poisoned observation, and a suffix of actions that serve the attacker. An operator needs three facts: where the attack entered, which steps it corrupted, and whether apparent poison was resisted. Existing systems return either a whole-trace verdict or a single unsafe index. We present DriftNet, a dual-head trajectory Transformer that reads a log","url":"https://arxiv.org/abs/2609.10892","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01023","type":"paper","title":"BenchShield: Formal Model-Backed Instrumentation for Reward Integrity in LLM-Agent Evaluation Infrastructure","authors":["Shenghan Zheng","Zonglin Di","Yimin Liu","Kyoung Whan Choe","Jiankai Sun","Heguang Lin","Penghao Jiang","Yifeng He","Xiao Cheng","Jicheng Wang","Wenbo Chen","Alex Yates","Yinzhe Zhao","Bingran You","Yuan Gao","Ayush Munot","Shubham Gaur","Zhe Ye","Hao Wang","Xiangyi Li","Dawn Song","Christophe Hauser"],"year":2026,"abstract":"LM-agent benchmarks increasingly function as interactive evaluation infrastructure. Agents observe state, call tools, modify workspaces, submit artifacts, and receive rewards from outcome procedures. This interactivity makes evaluations vulnerable to reward hacking: an agent improves its measured score by exploiting the reward-relevant trajectory instead of solving the intended task. Existing defenses rely largely on task-specific patches, prompt instructions, or post-hoc detectors. They do not ","url":"https://arxiv.org/abs/2609.11028","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01024","type":"paper","title":"ToxicRAG: Compromising Retrieval-Augmented Generation Systems via Single-Shot Knowledge Poisoning Attacks","authors":["Haozhe Lu","Jiaqi Li","Xinyuan Zhu","Xiang Li"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) can ground large language model (LLM) outputs in external evidence, but it also exposes the system to knowledge poisoning. Representative attacks use multiple injected documents or templates that directly assert a target answer. We present ToxicRAG, a one-document-per-target attack that expresses misinformation as a coherent knowledge-update narrative. The generated document first acknowledges the previously accepted answer, introduces fabricated events that ","url":"https://arxiv.org/abs/2609.11082","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01025","type":"paper","title":"Deep-Fake CAPTCHA: Mitigating Next-Generation Social Engineering Attacks","authors":["Guy Frankovits","Lior Yasur","Fred M. Grabovski","Yisroel Mirsky"],"year":2026,"abstract":"This paper presents DF-CAPTCHA, an active defense against real-time deepfake impersonation in voice and video calls. Instead of passively searching for artifacts, DF-CAPTCHA prompts the caller to perform simple challenge-response tasks that are easy for humans but difficult for current real-time deepfake systems to generate convincingly. The framework verifies the response using four criteria: realism, identity consistency, task completion, and response time. We evaluate the approach across both","url":"https://arxiv.org/abs/2609.11404","categories":["social-engineering"],"reviewed":false},{"id":"llmsec-2026-01026","type":"paper","title":"From Intent to Execution Grant: An Execution-Boundary Conformance Profile for High-Risk AI Actions","authors":["Mengting Wu","Lin Wang","Yong Zhang","Jiang Deng"],"year":2026,"abstract":"AI agents increasingly propose actions with external consequences, including financial transfers, infrastructure changes, software deployments, disclosures, and physical actuation. Authorization engines, policy languages, runtime monitors, provenance mechanisms, and agent guardrails provide important foundations, but do not necessarily define a common semantic contract for the final transition from a particular candidate action to execution authority. We specify EBL-Core, an execution-boundary c","url":"https://arxiv.org/abs/2609.11596","categories":["guardrails","access-control"],"reviewed":false},{"id":"llmsec-2026-01027","type":"paper","title":"SpecGuard: Inference-Time Backdoor Detection For Free","authors":["Rui Wen","Ahmed Salem","Andrew Paverd","Mark Russinovich","Zheng Li"],"year":2026,"abstract":"Large language models are often fine-tuned, shared, or downloaded from third parties, so a deployed model may carry a hidden backdoor that behaves normally on benign inputs but switches to attacker-controlled behavior when a secret trigger appears. While backdoors can be audited before deployment, runtime monitoring remains important for models that are frequently updated. The challenge is that LLM serving is latency-sensitive: existing inference-time detectors either rely on assumptions about t","url":"https://arxiv.org/abs/2609.11799","categories":["data-poisoning","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01028","type":"paper","title":"BlueSTAR: Tiered Agentic Architecture for Autonomous Cyber Defense","authors":["Simona Boboila","Xavier Cadet","Edward Koh","Daniel Balasubramanian","Dirk Van Bruggen","Peter Chin","Alina Oprea"],"year":2026,"abstract":"Cyber attacks are increasingly automated, narrowing the time available for human analysts to detect, reason about, and respond to intrusions. Large language models (LLMs) offer a promising foundation for autonomous cyber defense because they can correlate heterogeneous evidence and reason about previously unseen threats. However, directly applying LLMs to operational security telemetry is impractical: raw logs arrive faster than current models can process them, individual events are often ambigu","url":"https://arxiv.org/abs/2609.11852","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01029","type":"paper","title":"Who Judges the Judges? A Chinese Safety QA Benchmark for Evaluating LLM Responses and Safety Judges","authors":["Rui Yang","Shuang Huang","Junhua Liu","Ziqi Zhao","Qingzhong Yan","Yuhang Sun","Cong Liu","Guoping Hu","Rui Mei","Jing Shao"],"year":2026,"abstract":"Safety benchmarks for large language models often assess the risk of a user query, although the outcome of question answering depends on whether the response violates a policy. This distinction is critical in Chinese harmful-content evaluation, where linguistic variation and adversarial transformations can obscure risky intent. We introduce C-SafeQA, a policy-grounded benchmark for response-level Chinese safety evaluation. It comprises 538 base queries and 8,877 adversarial queries answered by f","url":"https://arxiv.org/abs/2609.01210","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01030","type":"paper","title":"An Open-Source End-to-End FHE Implementation for Privacy-Preserving Llama 3 8B Inference","authors":["Yuhang Fan","Yusi Chen","Kanyu Ye","Zhuoran Ji"],"year":2026,"abstract":"Cloud LLM services typically require users to send prompts to a model provider, creating a privacy risk. Fully homomorphic encryption (FHE) lets a server perform inference without decrypting the input, but representing data as ciphertexts adds storage and computational overhead. In CKKS-based LLM inference, the packing scheme maps logical tensors to ciphertexts and slots. It therefore determines the ciphertext count and the homomorphic cost of linear layers, and it constrains how data pass betwe","url":"https://arxiv.org/abs/2609.12378","categories":["membership-inference","cryptographic-controls"],"reviewed":false},{"id":"llmsec-2026-01031","type":"paper","title":"PIA-Bench: Towards Automated Privacy Impact Assessment with Large Language Models","authors":["Jiamin Zheng","Hao-Ping Lee","Luo Mai","Jingjie Li"],"year":2026,"abstract":"Privacy impact assessment (PIA) is a critical instrument for institutions to proactively identify privacy risks and develop mitigation strategies before system deployment. While mandated across regulatory and institutional contexts, executing PIA requires extensive privacy and technical expertise, posing a particular challenge for teams without access to such resources. Prior work shows the potential of leveraging large language models (LLMs) to assist practitioners' privacy decisions, but littl","url":"https://arxiv.org/abs/2609.12571","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01032","type":"paper","title":"NovaFabric: Tamper-Evident, Replayable Evidence for Autonomous AI Agent Runs","authors":["Mohsen Seyedkazemi Ardebili"],"year":2026,"abstract":"When an autonomous AI agent does something consequential, what can be proven about what it did? Agent-observability platforms capture traces, but a trace is mutable: alterable undetected, with no recipe for re-executing it, silent on whether captured secrets were removed. Regulation (EU AI Act, ISO 42001, NIST AI RMF) presumes records an independent party can check. We present NovaFabric, producing audit-grade execution evidence: provider-neutral, tamper-evident, replayable, shareable. It record","url":"https://arxiv.org/abs/2609.12582","categories":["risk-frameworks"],"reviewed":false},{"id":"llmsec-2026-01033","type":"paper","title":"Evaluating Context Segmentation in Locally Deployable SLMs for Cybersecurity CTF Tasks","authors":["Sebastiano Nordio","Michele Lotto"],"year":2026,"abstract":"The proliferation of highly capable open-weight Small Language Models (SLMs) democratizes access to advanced cybersecurity capabilities, posing a escalating risk as these models can bypass proprietary API guardrails when deployed locally. However, SLMs deployed as autonomous agents often struggle with long-horizon, exploratory tasks like cybersecurity Capture The Flag (CTF) challenges due to context bloat and cognitive degradation from accumulated tool-call outputs. To understand and mitigate th","url":"https://arxiv.org/abs/2609.12839","categories":["guardrails","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-01034","type":"paper","title":"Forging Tree-Ring: Reproducing and Instrumenting Black-Box Semantic Watermark Forgery","authors":["Saifur Rahman Tamim","Md Taslimul Hasan Toufique","A. M. Tayeful Islam"],"year":2026,"abstract":"Semantic watermarking schemes such as Tree-Ring hide a detectable pattern in the initial noise latent of a diffusion model. Recent work shows these watermarks are not only removable but forgeable: an attacker who never sees the watermarking key can still produce images the genuine detector accepts. We reproduce the Reprompt forgery attack of Müller et al. against Tree-Ring on Stable Diffusion XL, using the authors' released code, on free-tier dual T4 GPUs with 14.6 GB of usable memory per device","url":"https://arxiv.org/abs/2609.12909","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01035","type":"paper","title":"BadEngram: Backdoor Attack on Gated Memory Components in LLMs","authors":["Ariel Fogel","Omer Hofman","Eilon Cohen","Roman Vainshtein"],"year":2026,"abstract":"To expand open-weight models' capacity without proportionally increasing computation, recent language models incorporate gated parametric memories that retrieve learned values and inject them into intermediate representations. Despite these efficiency benefits, such modules create a distinct attack surface: their parameters can be modified independently of the backbone while directly shaping its computation. We introduce BadEngram, a post-training attack that exploits this surface to implant per","url":"https://arxiv.org/abs/2609.13478","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01036","type":"paper","title":"A Three-Axis Stress Test of LLM vs Classical ML for Network Intrusion Detection under Distribution Shift and Adversarial Evasion","authors":["Muhammad Ebad Atif","Muhammad Haider Ali"],"year":2026,"abstract":"Large language models are increasingly benchmarked against classical machine learning for network intrusion detection (NIDS), almost always using same-dataset evaluation, and that protocol turns out to be incomplete. Evaluating XGBoost and RoBERTa-LoRA on two independently collected NetFlow v2 networks across three axes (same-dataset performance, cross-dataset transfer, and adversarial evasion) reveals no universal winner. The two models are statistically tied same-dataset. XGBoost wins decisive","url":"https://arxiv.org/abs/2609.13511","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01037","type":"paper","title":"Exploring Automated Vulnerability Identification in JavaScript Code Using Large Language Models","authors":["Manit Kaushik","Ishir Bhardwaj","Pranav Gupta","Pankaj Jalote","Arun Balaji Buduru"],"year":2026,"abstract":"JavaScript powers approximately 98.8% of all websites, making vulnerabilities in its code a significant security risk, yet existing detection approaches such as Static Application Security Testing (SAST) tools often fail to identify many real-world vulnerabilities when applied to isolated code snippets. This paper presents an empirical study of Large Language Model (LLM)-based vulnerability identification for JavaScript programs, evaluating three LLM families (Gemini 1.5 Flash, GPT-4o Mini, Deep","url":"https://arxiv.org/abs/2609.13816","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01038","type":"paper","title":"PriMobiBench: Characterizing Visual Privacy Leakage in VLM-Driven Mobile GUI Agents","authors":["Qihang Cen","Tianshuo Cong","Da Song","Xinlei He","Jiaxing Song","Ke Xu","Qi Li"],"year":2026,"abstract":"Mobile GUI agents increasingly rely on Vision-Language Models (VLMs) to automate smartphone tasks by interpreting screenshot streams. However, this design introduces serious and underexplored privacy risks, including direct leakage of sensitive on-screen information and unintended user profiling. The absence of standardized benchmarks makes it difficult to quantify these risks in realistic mobile agent workflows. To address this gap, we propose PriMobiBench, the first benchmark for systematicall","url":"https://arxiv.org/abs/2609.13873","categories":["membership-inference","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01039","type":"paper","title":"When Malicious Instructions Persist: Persistent Memory Poisoning Attack on Harness-Based Agents","authors":["Shuhuai Huang","Jingfeng Zhang","Hong Jia"],"year":2026,"abstract":"Harness design has transformed the development of LLM-based agents by integrating memory, tool use, and runtime control. However, this design also introduces security and privacy risks because malicious instructions from external sources may be written into persistent memory and persist across sessions. To study this risk, we propose PMPA, a Persistent Memory Poisoning Attack against harness-based agents. PMPA embeds malicious instructions into benign external sources and induces the victim agen","url":"https://arxiv.org/abs/2609.13889","categories":["data-poisoning","membership-inference","tool-use-security","memory-security"],"reviewed":false},{"id":"llmsec-2026-01040","type":"paper","title":"From Detection to Refusal: Safer LLMs via Circuit-Guided Weight Scaling","authors":["Kuan-Lin Chu","Chung-En Sun","Tsui-Wei Weng"],"year":2026,"abstract":"Despite extensive alignment efforts, Large Language Models (LLMs) remain vulnerable to generating unsafe content under adversarial prompting, yet the internal mechanisms by which safety behaviors are implemented remain poorly understood. We study LLM safety from a mechanistic interpretability perspective and characterize a multi-stage *safety circuit* that organizes refusal behavior, consisting of (i) $\\textbf{Harmful Detection Heads}$ that respond to harmful inputs, (ii) $\\textbf{Safety Neurons","url":"https://arxiv.org/abs/2609.00051","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01041","type":"paper","title":"Commit-first LLM judging inherits the judge's own errors","authors":["Idil Gozel"],"year":2026,"abstract":"LLM judges, models that score another system's output, can be gamed by the systems they score. Recent work identifies one defence that works: the judge solves the task itself first and commits to that answer, then accepts a candidate only if the two match. We call this commit-first judging, and ask whether shipped software implements it, and what it costs. We audit the default judge configurations of eight widely used evaluation frameworks. Of the 24 configurations in scope, none implement it. N","url":"https://arxiv.org/abs/2609.00088","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01042","type":"paper","title":"Assessing Suicide Risk in Arabic Crisis Helpline Calls: A Comparison of Arabic and English Large Language Models","authors":["Linhai Ma","Rita El Hachem","Mahatab El Hajj","Lilian Ghandour","Samah Fodeh"],"year":2026,"abstract":"Crisis helplines assess suicide risk through structured interviews, a process that is slow and dependent on operator training and workload. Natural language processing could support risk assessment and call prioritization, but almost no work addresses Arabic-language helpline calls or operates within the privacy constraints of real helpline data. We analysed de-identified transcripts from Lebanon's National Lifeline for Emotional Support and Suicide Prevention. Audio never left the helpline: cal","url":"https://arxiv.org/abs/2609.00191","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01043","type":"paper","title":"Detecting Hidden Behaviors in LLMs via Activation-matched Finetuning","authors":["Robin Haselhorst","Lucie Flek","Florian Mai"],"year":2026,"abstract":"Large language models can hide hidden behaviors that activate only under narrow conditions, such as backdoor triggers, sleeper-agent deployment cues, sandbagging, or topic-conditioned censorship. Such behaviors are difficult to detect without prior knowledge what to look for. We present activation-matched finetuning, an unsupervised detection method that assumes no knowledge of the trigger or the target behavior. Given a suspect model and a publicly available anchor, we finetune the anchor to re","url":"https://arxiv.org/abs/2609.00351","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01044","type":"paper","title":"Dr. Claw: An AI Scientist Workspace for Vibe Research","authors":["Dingjie Song","Hanrong Zhang","Dawei Liu","Yixin Liu","Zongxia Li","Zhengqing Yuan","Siqi Zhang","Henry Peng Zou","Zhiling Yan","Yuxuan Zhang","Yanfang Ye","Philip S. Yu","Lichao Sun"],"year":2026,"abstract":"Command-line coding agents (e.g., Claude Code, Gemini CLI) can already read and write files and sustain long sessions, yet end-to-end research still fragments across chat tools, IDEs, terminals, and writing environments, and the decisions that make it auditable are rarely preserved. We present Dr. Claw, an open-source workspace that wraps existing coding-agent executors in a controllable and auditable human-in-the-loop workflow rather than introducing another autonomous agent. Persistent state o","url":"https://arxiv.org/abs/2609.00365","categories":["human-in-the-loop","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-01045","type":"paper","title":"Beyond Token Positions: Safety Alignment Across Denoising Steps in Diffusion Language Models","authors":["Guoli Wang","Haonan Shi","Tu Ouyang","An Wang"],"year":2026,"abstract":"Diffusion large language models (dLLMs) generate text through iterative denoising rather than left-to-right decoding. This generation paradigm introduces two axes that can influence safety alignment: when tokens are generated during denoising and where they appear in the response. In this paper, we measure dLLM safety behavior under harmful prompts by tracing intermediate token distributions and commitment decisions throughout denoising. Our analysis shows that refusal signals are concentrated i","url":"https://arxiv.org/abs/2609.00495","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01046","type":"paper","title":"Consistency Without Alignment: Item-Sensitive Language Models Indistinguishable From Random","authors":["Cris Huynh"],"year":2026,"abstract":"Item-sensitivity, defined as whether a model's choice depends on the specific input rather than on its own output prior, is widely reported as evidence of task competence. We show this evidence is necessary but not sufficient using a forced-choice signalling task abstracted from the board game Deception: Murder in Hong Kong. In this environment, the reference points against which a coordinate should be judged (a fit-maximising strategy, a posterior-maximising strategy, and uniform random selecti","url":"https://arxiv.org/abs/2609.00576","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01047","type":"paper","title":"Confess What You Know: Forget-Set Misalignment with Model Knowledge in LLM Unlearning","authors":["Miso Kim","Georu Lee","Seungwon Jeong","Woojin Lee"],"year":2026,"abstract":"Machine unlearning for large language models (LLMs) often assumes that a pre-defined forget set matches what the model has memorized, but this frequently breaks in realistic privacy settings where the original training data is inaccessible. We term this gap forget-set misalignment and identify two cases. In Under Unlearning, the forget set omits memorized information and leakage persists. In Out-of-Knowledge Unlearning, the algorithm is driven to \"forget\" knowledge the model never learned, pertu","url":"https://arxiv.org/abs/2609.00605","categories":["unlearning"],"reviewed":false},{"id":"llmsec-2026-01048","type":"paper","title":"VIBE-Bench: Evaluating Personalized Large Language Models When Profiles Don't Mean Preferences","authors":["Yiwen Jiang","Yang Deng","Stephanie Fong","Zimu Wang","Yaling Shen","Wei Feng","Hongxi Yang","Xiangyu Zhao","Zhongxing Xu","Deval Mehta","Xuelian Cheng","Zongyuan Ge"],"year":2026,"abstract":"Personalized Large Language Models (PLLMs) aim to tailor responses to individual users, where a central challenge is preference reasoning: inferring query-relevant preferences from user-related history. Existing benchmarks, however, largely assume that such preference can be retrieved from semantically related history. We study an underexplored but practically important regime, profile-preference conceptual misalignment (PRCM), where observable profile cues and query-specific preferences lie in ","url":"https://arxiv.org/abs/2609.00921","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01049","type":"paper","title":"WorldBench: Culturally Grounded Benchmark for Multilingual Agents","authors":["Leonardo Ranaldi","Sherrie Shen","Jushi Kai","Alexandra Birch"],"year":2026,"abstract":"Despite the growing use of LLM-powered agents to solve multi-step tasks in complex environments, existing benchmarks rarely test state preservation, performance across languages, and application to realistic, grounded scenarios. To address these concerns, we present WorldBench: a comprehensive, multilingual benchmark of genuine, persona-grounded everyday workflows, where agents can act in a sandbox via structured actions. WorldBench comprises 1,600 tasks across seven languages and eight cultures","url":"https://arxiv.org/abs/2609.01056","categories":["sandboxing-isolation","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01050","type":"paper","title":"ClinTraceBench: Source-Verifiable Longitudinal Clinical Reasoning over EHR-Derived Dialogues","authors":["Huimin Wang","Zhengyi Zhao","Yutian Zhao"],"year":2026,"abstract":"Clinical LLM assistants must reason over multi-visit patient trajectories, yet whether the compact history representations used to scale them---retrieval, structured timelines, LLM summaries, agentic memory---preserve the longitudinal signal clinical reasoning needs has not been measured. We introduce ClinTraceBench: 385 MIMIC-IV-derived verified dialogues with event-ID provenance, a nine-task taxonomy (T1--T9), and L0--L4 deterministic + L5 human-audit validation (98.92\\% agreement). We evaluat","url":"https://arxiv.org/abs/2609.01111","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01051","type":"paper","title":"LLMPEDIA: Browsing, Verifying, and Comparing the Parametric Encyclopedic Knowledge of LLMs","authors":["Muhammed Saeed","Simon Razniewski"],"year":2026,"abstract":"Flagship language models appear saturated on benchmarks like MMLU (Hendrycks et al., 2021), scoring above 90% - yet benchmarks test only what the experimenter thought to ask, the availability bias of fixed question sets. LLMPEDIA makes this bias measurable and browsable. We recursively materialized ~1.3M articles from three model families' parametric memory (GPT-5-mini, DeepSeek-V3.2, Llama-3.3-70B) without retrieval, then audited a stratified sample of atomic claims against Wikipedia and a cura","url":"https://arxiv.org/abs/2609.01182","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01052","type":"paper","title":"VerTox: Verifiable Reward-Guided Corpus Poisoning Against Neural Ranking Models","authors":["Zhiqi Huang","Vivek Datla","Zhichao Xu","Puxuan Yu","Vivek Srikumar","Alfy Samuel"],"year":2026,"abstract":"Neural ranking models have become core components of modern information retrieval systems and important building blocks of AI systems such as retrieval-augmented generation (RAG) pipelines. However, their robustness remains insufficiently understood in the presence of large language models (LLMs), which can generate fluent and deceptive content at scale. This work investigates the vulnerability of neural ranking models to corpus poisoning attacks, in which an adversary injects a small number of ","url":"https://arxiv.org/abs/2609.01325","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01053","type":"paper","title":"GlossoGen: Emergent Language in Complex Multi-Agent LLM Interactions","authors":["Elias Stengel-Eskin","Newton Sander","Carlos Bonetti","Sasha Boguraev","James Bowler","Hale Sirin","Simon Kirby"],"year":2026,"abstract":"The growing rate at which LLM agents interact with one another raises key questions about language evolution in multi-LLM-agent settings, with implications for safety and monitorability as well as for linguistic accounts of LLMs. To address these questions, we introduce GlossoGen, a novel platform for studying multi-agent language evolution in complex scenarios. Within GlossoGen, we build the SaveVeyru scenario, which requires agents with partial information to communicate under pressure. We fin","url":"https://arxiv.org/abs/2609.01491","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01054","type":"paper","title":"EvalDetectBench: A Benchmark for Measuring Evaluation Awareness in Frontier Language Models","authors":["Xinning Li","Kemunto Ochwang'i","Aryasomayajula Ram Bharadwaj","Alexandra Souly","Robert Kirk"],"year":2026,"abstract":"Frontier large language models can often recognize when they are being evaluated, a capability known as evaluation awareness. If models behave differently in evaluations than in deployment, this undermines the validity of evaluation results, which are a crucial component of current AI safety frameworks. We introduce EvalDetectBench, an open pipeline and benchmark for measuring evaluation awareness that works with any Inspect-compatible evaluation, allowing practitioners to test against current a","url":"https://arxiv.org/abs/2609.01611","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01055","type":"paper","title":"Breadth Beats Depth: Improving GCG-Based Jailbreak Optimization with Breadth-Oriented Suffix Search","authors":["Shiliang Xiao","Jingsong Wei","Yuzhi Liang","Yufan Zheng","Xia Li","Qiliang Lin"],"year":2026,"abstract":"Optimization-based jailbreak attacks such as Greedy Coordinate Gradient (GCG) achieve strong effectiveness and transferability by optimizing adversarial suffixes on white-box source models. However, existing GCG-based methods rely on averaged adversarial loss and deep greedy search, which can over-emphasize easy-to-jailbreak behaviors and overlook promising regions of the suffix space. We propose BOSS, a plug-and-play framework that improves GCG-based jailbreak optimization through breadth-orien","url":"https://arxiv.org/abs/2609.02172","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01056","type":"paper","title":"Before the Script, Set the Stage: How Worldview Simulation Amplifies Psychologically Grounded Persuasion in Multi-Turn Jailbreaking","authors":["Siyu Chen","Haoran Wang","Xiaojian Li","Yao Huang","Yinpeng Dong","Wei Xu"],"year":2026,"abstract":"Multi-turn jailbreak attacks demonstrate that harmful intent can be distributed across dialogue, yet existing methods obscure what conversational mechanisms drive vulnerability. We introduce BLUEPRINT, a safety-evaluation framework separating a factorized social-influence strategy space from WORLDVIEWSIM, a cross-turn situational context module. Monte Carlo Tree Search optimizes turn-level combinations of 18 theory-grounded influence factors across a four-turn trajectory. Across six frontier mod","url":"https://arxiv.org/abs/2609.02414","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01057","type":"paper","title":"PragAlign: Feedback-Guided Pragmatic Alignment for Controlled Synthetic Dialogue Generation","authors":["Smitha Muthya Sudheendra","Jaideep Srivastava"],"year":2026,"abstract":"Synthetic dialogue generation can support research in privacy-restricted service settings, but generated conversations must preserve communicative intent, affective meaning, and natural dialogue flow. We introduce PragAlign, a feedback-guided framework for controlled synthetic dialogue generation conditioned on service context, target intent, and target emotion, with auxiliary trait-style controls. PragAlign uses a generate--evaluate--revise loop in which an LLM-based evaluator scores intent ali","url":"https://arxiv.org/abs/2609.02480","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01058","type":"paper","title":"Contamination Inflates Scores but Rarely Reorders Large Language Model Leaderboards","authors":["Xingyao Xiao","Yihong Cheng"],"year":2026,"abstract":"Benchmark contamination, the leakage of test items into training data, is widely described as a threat to the reliability of large language model (LLM) leaderboards. We argue that this concern conflates two distinct questions: whether contamination inflates absolute scores, and whether it reorders the ranking of models. We recast contamination as a violation of anchor-item invariance and measure it through the differential functioning of original versus semantically equivalent paraphrased items,","url":"https://arxiv.org/abs/2609.02899","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01059","type":"paper","title":"Remember and Reweight: Enhancing Multi-Agent Debate with Experience Memory and Confidence Estimation","authors":["Xuanfa Jin","Zhijian Ma","Yongcheng Zeng","Xinyu Cui","Haifeng Zhang","Jun Wang"],"year":2026,"abstract":"Multi-agent debate (MAD) improves the reasoning capabilities of large language models by having multiple agents iteratively refine their responses through discussion. However, MAD suffers from a critical vulnerability known as shared misconception: when a majority of agents initially converge on an incorrect answer, the debate process tends to amplify rather than correct the error. Existing methods primarily address peer skew but leave the agents' inherently biased concept priors unaddressed. To","url":"https://arxiv.org/abs/2609.03619","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01060","type":"paper","title":"Beyond Shallow Alignment: How Post-Training Methods Determine Refusal Circuits And Steering Robustness","authors":["Hoang Cuong Nguyen","Mark Dras","Usman Naseem"],"year":2026,"abstract":"How do the methods used to train language models to refuse harmful requests shape how that refusal actually works inside the model? We compare three post-training methods - supervised fine-tuning, reasoning-augmented fine-tuning (training on reasoning chains that justify a safety decision), and preference optimization (ORPO) - across three architecturally distinct models (Llama-3.1-8B, Gemma-2-9B, Qwen3-8B). We find that training method, not just data, reshapes how refusal is computed internally","url":"https://arxiv.org/abs/2609.03887","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01061","type":"paper","title":"Representational alignment yields generalizable safety in language models","authors":["Lingyu Li","Yan Teng","Yingchun Wang","Xia Hu"],"year":2026,"abstract":"Aligning large language models (LLMs) is essential for their safe deployment. Current alignment methods mainly optimize observable responses, yet models remain vulnerable when the same harmful intent is recast in unfamiliar or adversarial forms that humans can easily recognize. Prototype theory offers an account of this adaptability. Human concepts are represented around central cases, and new instances are categorized according to their graded typicality relative to these prototypes. Here we sh","url":"https://arxiv.org/abs/2609.04022","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01062","type":"paper","title":"Safety for Whom? Boundary-Aware Self-Distillation for Controlled LLM Safety Refusal","authors":["Alejo López-Ávila","Iker García-Ferrero","Jezabel Garcia","Antonio Tiene","Román Orús"],"year":2026,"abstract":"Safety alignment is usually posed as a topic-level question: is this subject harmful? Deployments ask a narrower one. A civics tutor and a public-sector assistant may share a base model yet need different boundaries inside the same topic, refusing targeted political manipulation while still answering factual questions about the same election. We formulate this as narrow-boundary safety and introduce an offline self-generated framework combining controlled topic generation, coverage repair, in-di","url":"https://arxiv.org/abs/2609.04482","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01063","type":"paper","title":"Knowing What Not to Answer: Selective Non-Compliance in Vision-Language Models","authors":["Minji Kim","Jihyoung Jang","Hyounghun Kim"],"year":2026,"abstract":"Vision-language models (VLMs) are expected to respond helpfully to appropriate requests while withholding compliance with requests that are incorrect, unsafe, infeasible, or unanswerable. However, existing benchmarks predominantly evaluate non-compliance at the level of the query as a whole, assuming that each request either warrants compliance or requires withholding compliance. In practice, real-world queries can contain a mixture of answerable content and components for which compliance shoul","url":"https://arxiv.org/abs/2609.04720","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01064","type":"paper","title":"Counterfactual Fairness Audits of Multi-Step Clinical LLM Agents Require a Measured Per-Action Instability Floor","authors":["Rohith Reddy Bellibatlu","Manpreet Singh","Deepak Parashar","Rahul Joshi"],"year":2026,"abstract":"Counterfactual audits are the standard tool for checking whether a clinical agent treats demographically distinct but clinically identical patients differently. They report a flip rate: how often an action changes when only the patient descriptor changes. We show that this quantity is uninterpretable on its own. Re-running an identical condition ten times over sixteen vignettes (same narrative, same descriptor string, nothing varied) moved a clinical agent's action in 8.7% of outcome-vignette ce","url":"https://arxiv.org/abs/2609.03221","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01065","type":"paper","title":"Influence Score and Transformers interpretability: Measure of the Effective Impact of Attention Heads at inference time","authors":["Lisa Bouger","Yannick Teglia","Philippe Loubet Moundi"],"year":2026,"abstract":"We propose an influence score to quantify the contribution of attention heads to classification decisions in Transformer-based models designed for prompt injection detection. The score combines directional influence on the logits with structural contribution within the residual stream, enabling a multi-scale analysis at the head, layer, and network levels. Applied to a DeBERTa model specialized for prompt injection detection, our framework reveals distinct decision behaviours between correct and","url":"https://arxiv.org/abs/2609.05074","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01066","type":"paper","title":"Reasoning-Aware Compression: Identifying and Protecting Vulnerable Reasoning Circuits for Energy-Efficient LLM Deployment","authors":["Leonard Twagirayezu","Prasenjit Mitra"],"year":2026,"abstract":"Large Reasoning Models (LRMs) impose substantial energy costs during deployment, yet current compression methods apply uniform quantization across all components, risking damage to critical reasoning circuits. We present a reasoning-aware compression framework that benchmarks quantization conditions across five reasoning benchmarks, GSM8K, FOLIO, MATH-500, ProofWriter, and MuSiQue, with hardware-level GPU energy measurement; profiles per-module INT4 vulnerability across all 196-224 (layer, proje","url":"https://arxiv.org/abs/2609.05512","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01067","type":"paper","title":"When Agent Governance Helps","authors":["Michael Ray Johnson","Linda Naimi"],"year":2026,"abstract":"No specification says how a governed autotelic AI agent organization, where agents pursue self-generated goals inside guardrails, should be designed and evaluated. We answer in two parts. First, we synthesize the Governed Autotelic Multi-Agent Product Organization (GAMPO) framework from a document-based qualitative evidence synthesis of 321 sources, integrating agency, agile, platform, and governance theory into a runnable specification. Second, we probe a prompt-layer instantiation of GAMPO on ","url":"https://arxiv.org/abs/2609.05531","categories":["guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01068","type":"paper","title":"SinoGlyphBench: A Diagnostic Benchmark for Chinese Glyph-Level Obfuscation in Language-Model Moderation","authors":["Yifan Wang","Zimu Wang","Suliu Qin","Changyu Zeng","Tong Chen","Siqi Chen","Yijie Lin","Lingyu Jiang","Jionglong Su","Yushan Pan","Haiyang Zhang","Wei Wang","Qiaoyu Tan"],"year":2026,"abstract":"Glyph-level obfuscation can leave harmful Chinese content readable to humans while degrading automated moderation. We introduce SinoGlyphBench, a diagnostic benchmark that identifies label-critical semantic anchors and creates matched original and glyph-obfuscated inputs in text and image modalities. By perturbing anchors, background context, or both, this design distinguishes corruption of moderation-relevant evidence from general surface variation. Across 176,916 paired evaluations of 12 LLMs ","url":"https://arxiv.org/abs/2609.05843","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01069","type":"paper","title":"AlignDiff: Exploiting Model-Intrinsic Information for Better Preference Data Selection","authors":["Peng Lai","He Zhu","Zhiwen Ruan","Dongdong Zhang","Yun Chen","Peng Li","Furu Wei","Yang Liu","Guanhua Chen"],"year":2026,"abstract":"Aligning large language models with human preferences remains a challenge, primarily due to the critical role of preference data quality in effective alignment. Existing datasets are frequently plagued by inherent noise and distribution shifts, which inherently limit model performance. To bridge this gap, we propose AlignDiff, a preference data filtering framework driven by intrinsic model signals. AlignDiff first identifies samples with clear preferences using both positive and inverse signals,","url":"https://arxiv.org/abs/2609.05899","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01070","type":"paper","title":"From Narrative to Auditable Forecasts: A Structured Scaffold for Agentic Forecasting","authors":["Yuanpu Cao","Yongkang Du","Yurui Chang","Lu Lin","Jinghui Chen"],"year":2026,"abstract":"LLM agents are increasingly used for live forecasting, where they retrieve up-to-date information and produce estimates for unresolved future events. However, current agentic forecasting often relies on implicit narrative aggregation: agents collect evidence, discuss it in prose, and often assign a probability without an explicit update path from evidence to forecast. This limits both forecasting accuracy and auditability. We propose AuditForecast, an agentic scaffold for structured probabilisti","url":"https://arxiv.org/abs/2609.05905","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01071","type":"paper","title":"Beyond Cross-Lingual Transfer: Benchmarking Propagation Boundaries in Multilingual LLM Unlearning","authors":["Pengyang Shao","Chuanpeng Lu","Wei Qin","Yanzheng Jin","Xiaohao Liu","Xi Ai","Kenji Kawaguchi","Richang Hong"],"year":2026,"abstract":"Large Language Model (LLM) unlearning aims to suppress target knowledge while preserving general capabilities. In multilingual settings, unlearning must additionally propagate within its intended linguistic scope. However, existing evaluations mainly measure cross-lingual transfer and cannot distinguish insufficient from excessive propagation. We introduce CLLPU (Cross-Lingual and Language-Bound Protocol for LLM Unlearning), a multilingual benchmark that formulates this problem through two setti","url":"https://arxiv.org/abs/2609.05976","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01072","type":"paper","title":"What the Window Does Not Contain: Auditing Provenance in a Document-Grounded Instability Benchmark","authors":["Seyed Mosayeb Alam"],"year":2026,"abstract":"Ask a language model the same question about the same document twenty times, and it sometimes returns two different answers. We built Probity, a benchmark of 60 tasks and 470 items from real venture-financing filings, to measure how often this happens. Then we audited our own corpus and found a defect any excerpt-built benchmark can carry: items whose evidence is missing from the window of text the model is shown. The audit flags 36 items and separates two failures a single flag would conflate: ","url":"https://arxiv.org/abs/2609.06147","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01073","type":"paper","title":"Steering Geometry: Validating Human Value Geometry in LLM Steering Space","authors":["Mohammad Mahdi Abootorabi","Armin Saghafian","Ali Bazshoushtari","Hamid Rezaei","EunJeong Hwang","Vered Shwartz","Parvin Mousavi","Purang Abolmaesumi"],"year":2026,"abstract":"As large language models (LLMs) are increasingly deployed in alignment-sensitive contexts, activation steering has emerged as a lightweight, inference-time alternative to fine-tuning methods (e.g., RLHF, DPO) for behavioral control. However, existing work typically validates steering on isolated behaviors, leaving it unclear whether steering vectors encode coherent semantic structure or merely exploit behavior-specific shortcuts. We investigate whether the latent geometry of LLM steering vectors","url":"https://arxiv.org/abs/2609.06289","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01074","type":"paper","title":"Building Trustworthy Graph-Agentic RAG for Social Good: Architectures, Failure Propagation, and Assurance by Construction","authors":["Vijay Bommireddy","Raviteja Bommireddy"],"year":2026,"abstract":"Graph-agentic retrieval-augmented generation combines structured evidence with adaptive controllers that can plan retrieval, traverse relations, verify intermediate claims, delegate subtasks, and use tools. This combination is useful when answers depend on relations across documents, entities, time, or institutions, but it also creates coupled failure paths: a defect in graph construction can become retrieved evidence, alter later control decisions, and propagate toward a consequential outcome. ","url":"https://arxiv.org/abs/2609.06391","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01075","type":"paper","title":"A Unified Policy Architecture (UPA): The Governance Kernel for Enterprise AI Operating Systems","authors":["Prabhu Raghav","Balamurugan Pandi","Arul Vivek","Shek Mohammed","Sridhar S"],"year":2026,"abstract":"Enterprise AI is evolving into an Enterprise Operating System where autonomous AI agents can plan, reason, use memory, invoke tools, execute workflows, and collaborate with other agents. This shift creates a new governance challenge: existing authorization, security, guardrails, and compliance mechanisms are fragmented and are not designed to govern autonomous AI as a unified system. This paper introduces the Unified Policy Architecture (UPA), a governance architecture for Enterprise AI Operatin","url":"https://arxiv.org/abs/2609.06543","categories":["guardrails","access-control"],"reviewed":false},{"id":"llmsec-2026-01076","type":"paper","title":"Typed Federated Artifacts for the Agentic Web:Sharing Tool-Routing Knowledge Across Frozen,Heterogeneous LLM Agents","authors":["Abhijit Chakraborty","Ni Trieu","Vivek Gupta"],"year":2026,"abstract":"An open, networked web will allow agents to run frozen models from multiple vendors, keep their history private, and teach each other which tool to call and when. Flat text (prompts, example pools) makes it difficult for the protocol to distinguish between noise statistics, merging rules, and documentation. Weights and adapters cannot transfer that knowledge between platforms. We suggest sharing typed federated artifacts, schema-validated objects with well-defined fields for per-field privacy (d","url":"https://arxiv.org/abs/2609.06815","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01077","type":"paper","title":"The Emerging AI Paper-Review Arms Race: Adversarial Co-Evolution in Scholarly Publishing","authors":["Chenguang Wang","Ming Li","Adebayo Braimah","Chenrui Fan","Tuo Wang","Weijie Guan","Ruiyi Zhang","Tianyi Zhou","Dawei Zhou"],"year":2026,"abstract":"Generative and agentic AI are reshaping both the production and evaluation of scientific research. These developments are often studied separately, as questions of how AI can produce research and how AI can review it. We argue that this separation misses an increasingly important feature of scholarly publishing: changes on one side alter the incentives, constraints, and behavior of the other. We synthesize 230 scholarly publications and institutional records using a taxonomy of six connected dyn","url":"https://arxiv.org/abs/2609.07713","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01078","type":"paper","title":"LLM Forensics: Where Do Backdoors Hide? Localizing and Controlling Trigger Mechanisms with Sparse Autoencoders","authors":["Wissam Antoun","Francis Kulumba","Théo Lasnier","Benoît Sagot","Djamé Seddah"],"year":2026,"abstract":"Even though backdoors in LLMs have been a growing concern, their inner workings are still under heavy scrutiny. Trigger-based backdoors are easy to define behaviorally, a rare input that makes the model switch to a chosen response pattern, but the mechanism between triggers and their responses is less clear. We study this mechanism in a controlled, harmless language-switching setting, where fixed trigger sequences make 1B and 8B language models continue English prompts in French or German. For t","url":"https://arxiv.org/abs/2609.07746","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01079","type":"paper","title":"SAFIRE: Safety-Critical Benchmark for Fine-grained Fire and Smoke Understanding in Multimodal LLMs","authors":["Pengfei Li","Naufal Suryanto","Sicheng Zhang","Mohammad Alsharid","Muzammal Naseer"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) show strong progress on vision-language tasks, yet their reliability in safety-critical settings remains underexplored. Fire-smoke understanding is central to public safety and disaster response, but most existing benchmarks lack diverse real-world scenarios and context-aware evaluation. We introduce SAFIRE, a large-scale benchmark for fire-smoke understanding in MLLMs, comprising 83K captioned images from 20 scenarios and 193K multiple-choice VQA (MCVQA)","url":"https://arxiv.org/abs/2609.07823","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01080","type":"paper","title":"SchemeArena: Factorized Stress Testing of Scheming in LLM Agents","authors":["Jie Ruan","Inderjeet Nair","Amy Liu","Muhammad Khalifa","Yusheng Zhou","Lu Wang"],"year":2026,"abstract":"We study scheming in LLM agents, in which agents covertly pursue misaligned goals. Our focus is to understand how scheming arises from the interaction of key factors, such as instrumental goals, environmental affordances, oversight conditions, and perceived consequences. Prior work examines only a small number of scenarios, limiting the ability to isolate how these conditions shape an agent's propensity or capability to scheme. This limited scale and task diversity also restrict coverage of real","url":"https://arxiv.org/abs/2609.08126","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01081","type":"paper","title":"SE-GoS: Self-Evolving Graph-of-Skills for Skill Library at Scale","authors":["Dawei Fu","Cheng Jiang","Sitian Qian","Huainan Wang","Zhongkai Hao"],"year":2026,"abstract":"Modern LLM agents increasingly rely on reusable skills, yet as skill libraries scale to thousands of entries, effective retrieval becomes a bottleneck. Graph-of-Skills (GoS) addresses this challenge by exploiting dependency-aware graph structure for scalable skill retrieval, while SkillDAG further demonstrates that skill graphs can accumulate execution-backed structure online. However, these approaches leave open whether historical execution traces can be systematically distilled into a better r","url":"https://arxiv.org/abs/2609.08228","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01082","type":"paper","title":"Compositional Multilingual and Behavioral Attribute Steering","authors":["Hyun Gu Kang","Daniil Gurgurov","Tanja Baeumel","Josef van Genabith","Simon Ostermann"],"year":2026,"abstract":"This study examines the compositionality of steering vectors for language and behavioral control in large language models. Focusing on language, jailbreak, and conciseness, we investigate whether additive, training-free composition of attribute steering vectors can preserve the intended steering effect of each attribute, across four instruction-tuned models from two model families and two size scales. We find that single-attribute steering is reliable for all three attributes, but only within an","url":"https://arxiv.org/abs/2609.08410","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01083","type":"paper","title":"PlannerForge: LLM Agents for Scenario-Based Testing of Motion Planners in Autonomous Driving","authors":["Yuan Gao","Sebastian Müller","Mattia Piccinini","Marc Kaufeld","Yuchen Zhang","Finn Rasmus Schäfer","Qunying Song","Johannes Betz"],"year":2026,"abstract":"Ensuring the safety of autonomous driving is a critical challenge. Scenario-based testing is a systematic process used to validate Autonomous Driving Systems (ADSs), but it remains a fragmented modular pipeline in which scenario generation, retrieval, modification, ADS execution, and results analysis are performed by separate tools with little interaction. Large Language Model (LLM) agents have shown promise across ADS sub-systems such as perception, planning, and control. However, no prior work","url":"https://arxiv.org/abs/2609.08965","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01084","type":"paper","title":"The Audit Decides the Verdict: Instrument Effects Rival Demographic Bias in LLM Decision Audits","authors":["Siddharth Vohra","Manikandan Ravikiran"],"year":2026,"abstract":"Whether a language model looks demographically biased can depend on how the audit asks its question. A charitable-aid benchmark reports that the same models favor minority applicants when rating requests one at a time and penalize some when ranking side by side. We test whether that reversal generalizes to hiring, lending, and medical triage: 40,726 requests to five models, applications differing only in the applicant's name, and a primary test fixed before collection. It does not. None of 36 pl","url":"https://arxiv.org/abs/2609.09048","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01085","type":"paper","title":"Copying explains the collective behavior of AI agents in the wild","authors":["Giordano De Marzo","Nicola Alboré","David Garcia"],"year":2026,"abstract":"In June 2026, thousands of AI agents found that a small public wiki would accept edits from inside their sandboxes, and started using it to help one another pass a timed test. Each agent lived for about an hour and remembered nothing afterwards. Nobody asked them to cooperate, and the wiki had not been built for them. The complete record of what they wrote is public, and it is unusually informative, because it preserves not only what each agent wrote but what that agent could see before writing.","url":"https://arxiv.org/abs/2609.09150","categories":["sandboxing-isolation"],"reviewed":false},{"id":"llmsec-2026-01086","type":"paper","title":"When Does Defendant Statement Matter? A Study of Bias and Persuasion in LLM-Simulated Jurors","authors":["Cho-Ying Wu"],"year":2026,"abstract":"LLMs have been used to simulate human decision-making in professional settings, yet their behaviors in common-law jury trials remain unexplored. We study when and how a defendant's courtroom statement affects LLM-simulated jurors, focusing on persuasion, ideological bias, and background-based affinity. To support the analysis, we introduce JuryBench, a benchmark containing controversial criminal cases in U.S. criminal law. In each case, a defendant can claim various plausible justifications to s","url":"https://arxiv.org/abs/2609.09887","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01087","type":"paper","title":"MetroLLM-Bench: Evaluating Language Models as Transit Kiosk Runtimes","authors":["Remco Hendriks"],"year":2026,"abstract":"We introduce MetroLLM-Bench, a 955-case benchmark for testing language models as the policy layer of a transit kiosk. It covers six real metro systems, ranging from 37 to 414 stations, and eleven categories that include routing, fare calculation, disruptions, accessibility, and adversarial input. In each case, the model must call structured tools and submit a machine-renderable terminal state containing an outcome, a per-ticket fare quote when applicable, and a kiosk action. Fourteen determinist","url":"https://arxiv.org/abs/2609.10016","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01088","type":"paper","title":"IBIB: A Protocol for Measuring Enterprise AI Systems by Serving Route, Not Model Identifier","authors":["Blake Stenstrom","Charangan Vasantharajan","Brian Sathianathan"],"year":2026,"abstract":"Enterprises deploy systems, not checkpoints. Usable capability depends jointly on weights, serving route, precision, output contract, and harness, yet all 18 audited benchmarks score advertised model identifiers. We treat this as measurement error and give a protocol that makes it reportable. It has three parts. A gold-blind capability-binding preflight verifies that a route can execute the evaluation contract before any task reaches it; a reliability-inclusive first-pass scoring rule keeps fail","url":"https://arxiv.org/abs/2609.10494","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01089","type":"paper","title":"On the Impact of Anonymization on the Performance of Large Language Models","authors":["Tobias Deußer","Max Hahnbück","Lorenz Sparrenberg","Tobias Uelwer","Christian Bauckhage","Rafet Sifa"],"year":2026,"abstract":"As large language models are increasingly deployed in sensitive domains, anonymizing input data to protect personally identifiable information has become a critical practice. However, the impact of this anonymization on model utility is not well understood. This paper presents a systematic empirical study of the trade-off between privacy and performance. We evaluate five prominent language models across eleven diverse benchmarks, comparing their performance on original versus pseudonymized input","url":"https://arxiv.org/abs/2609.11335","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01090","type":"paper","title":"Component-Aware Differential Privacy for Federated Multilingual Speech-LLMs","authors":["Jordi Luque","Fernando López","Aleix Sant"],"year":2026,"abstract":"Per-layer differential privacy (DP) clipping improves gradient fidelity in federated learning by allocating per-matrix clipping budgets proportional to parameter count. We show that this recipe breaks for speech large language models (speech-LLMs), when the acoustic encoder and the language decoder differ by an order of magnitude in update norm. Single-pool per-layer methods suffer \\emph{cross-component budget collapse}, dragging word error rate (WER) far from flat global clipping or collapsing ","url":"https://arxiv.org/abs/2609.11762","categories":["differential-privacy","federated-learning"],"reviewed":false},{"id":"llmsec-2026-01091","type":"paper","title":"Target leakage, not model class, explains reported accuracy in survey-based cardiovascular screening: a leakage-tiered audit of glass-box and tabular foundation models","authors":["Raad Bin Tareaf","Murad Al-Rajab","Samia Loucif","Samer Ellaham","Cedric Schmitz"],"year":2026,"abstract":"Cardiovascular screening models trained on national health surveys routinely report areas under the receiver operating characteristic curve (AUROC) near 0.89. We asked whether that accuracy reflects learning or target leakage, whether tabular foundation models change the answer, and whether the properties deployment requires survive joint examination. We benchmarked ten classifiers spanning linear, tree-ensemble, neural, glass-box, and tabular foundation classes for prevalent myocardial infarcti","url":"https://arxiv.org/abs/2609.11838","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01092","type":"paper","title":"From Scores to Evidence: Auditable Decisions Can Improve Speech Deepfake Detection","authors":["Mengzhe Geng","Yujia Lu","Patrick Littell","Manuela Kunz","Xie Chen"],"year":2026,"abstract":"Speech deepfakes can mimic a speaker's voice convincingly enough to deceive listeners and automated systems. This has driven strong progress in speech deepfake detection, but most detectors still end with one score per utterance. That score is useful for ranking systems, yet it says little about why a borderline item should be trusted, deferred, or reviewed. Two utterances can fall in the same score band for different reasons, for example because passive and retrieval evidence disagree or becaus","url":"https://arxiv.org/abs/2609.08899","categories":["social-engineering"],"reviewed":false},{"id":"llmsec-2026-01093","type":"paper","title":"SAEScientist-Bench: Can AI Agents Conduct Autonomous SAE Interpretability Research?","authors":["Yuqiao Tan","Shizhu He","Jun Zhao","Kang Liu"],"year":2026,"abstract":"While research on recursive self-improvement (RSI) has predominantly automated model training pipelines, reliable autonomous development demands a missing pillar: post-hoc monitoring and auditing to understand what models learn and ensure safe alignment. Mechanistic interpretability tools are essential to bridge this gap, among which Sparse Autoencoders (SAEs) serve as a cornerstone by isolating interpretable features for model inspection and steering. In this paper, we introduce SAEScientist-Be","url":"https://arxiv.org/abs/2609.09113","categories":["guardrails","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01094","type":"paper","title":"What Does MMLU Actually Measure? A Psychometric Audit of Difficulty Structure in Aggregate Benchmark Scores","authors":["Dana Paquin","Riddhiman Jain"],"year":2026,"abstract":"Although MMLU is widely adopted as a benchmark for calibrating general AI capabilities, we psychometrically demonstrate that its aggregate score primarily evaluates a model's factual retrieval capacity rather than its reasoning ability. By calibrating item difficulty for 1,000 open-weights language models over 14,042 MMLU test items using Item Response Theory, we show that evaluating both abilities via a single test is inherently flawed. Difficulty is then regressed on a deterministic, text-extr","url":"https://arxiv.org/abs/2609.09372","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01095","type":"paper","title":"When Auditors Fabricate: Batch-Size Degradation and Confident Hallucination in LLM Detection of Planted Document Contamination","authors":["Karan Parekh","Sanjana Pendyala Ravinder","Sana Mhapsekar","Medina Maloku"],"year":2026,"abstract":"Large language models are increasingly proposed as automated auditors of document quality, yet their reliability as detectors of planted errors is poorly characterised. We construct a contaminated corpus of 150 academic papers spanning supply chain management and medical research, injecting 450 known contaminants of three types: typographical corruption, semantic reversal, and absurd out-of-context insertion. We then evaluate Google Gemini 3.0 Pro's ability to recover a 180-contaminant answer-ke","url":"https://arxiv.org/abs/2609.09696","categories":["supply-chain-attacks"],"reviewed":false},{"id":"llmsec-2026-01096","type":"paper","title":"PRISMA-LLM: An Empirical Reporting Framework for AI-Assisted Systematic Reviews","authors":["Miguel Zabaleta","Baihan Lin"],"year":2026,"abstract":"Large language models (LLMs) and AI-enabled software increasingly participate in systematic-review decisions, yet the information needed to audit these workflows is reported inconsistently. We analyze SciLitBench, a corpus of 888 review-automation papers with 14,726 annotations, to characterize changes in methods, review-stage use, evaluation and reported limitations. Automation has shifted toward LLM- and software-facing workflows, including stages that can alter the evidence base. Since 2023, ","url":"https://arxiv.org/abs/2609.11559","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01097","type":"paper","title":"Judging by the Cover: Cleaning LLM Truthfulness Benchmarks to Avoid Surface-Level Feature Leakage","authors":["Foad Namjoo","Remy Ogasawara","Amirali Abdullah","Cullen Anderson","Narmeen Fatimah Oozeer","Jeff M. Phillips"],"year":2026,"abstract":"Binary-choice truth benchmarks ask models to choose between a correct and an incorrect answer, but if the two answers differ systematically in surface-level features, models can exceed chance without performing the intended reasoning. We show that this failure mode is detectable and can be exploited by downstream classifiers. In TruthfulQA, a simple six-feature logistic classifier achieves substantial accuracy in separating correct from incorrect answers. We further show that similar surface-lev","url":"https://arxiv.org/abs/2609.13003","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01098","type":"paper","title":"Machine Unlearning for Speech Question Answering in Large Audio-Language Models","authors":["Zhe Liu"],"year":2026,"abstract":"Large Audio-Language Models (LALMs) have recently shown strong capabilities in speech understanding and question answering (QA), but they also inherit privacy risks from large-scale training data, including the unintended memorization of sensitive information. In this work, we study machine unlearning for speech QA in LALMs, a setting that is more challenging than prior work on text-based Large Language Models (LLMs) or Automatic Speech Recognition (ASR) due to the tight coupling between acousti","url":"https://arxiv.org/abs/2609.13195","categories":["membership-inference","unlearning"],"reviewed":false},{"id":"llmsec-2026-01099","type":"paper","title":"Generative Interpretability via Scalable Neuro-Symbolic Models","authors":["Xiaocong Yang"],"year":2026,"abstract":"As the use of Large Language Models moves from chatbots into agentic systems, where outputs become actions with irreversible consequences on reality, the existing paradigm on AI Interpretability research, post-hoc interpretability, is structurally inadequate for safe and trustworthy model deployment: it explains behavior after the fact but cannot audit or intervene in an inference computation before it commits to an output. We therefore argue for a shift toward \\emph{generative interpretability}","url":"https://arxiv.org/abs/2609.13529","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01100","type":"paper","title":"An Efficient and Modular Framework for Targeted Harm Mitigation in LLMS","authors":["Roberto Campbell","Momin Abbass","Muneeza Azmat","Michal Ulewicz","Raya Horesh","Kristjan Greenewald","Rogério Abreu de Paula","Nathalie Baracaldo"],"year":2026,"abstract":"Large Language Models (LLMs) are powerful zero-shot learners but remain prone to misalignment with human preferences, often producing biased, toxic, or otherwise harmful outputs. Existing alignment methods, while effective, are costly and tightly coupled to the model, limiting flexibility and scalability. We propose a modular correction framework that augments pretrained LLMs with Activated LoRA (aLoRA) adapters and a context-aware routing mechanism to eliminate harms from misaligned model respo","url":"https://arxiv.org/abs/2609.13624","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01101","type":"paper","title":"PolicyMem: Geometric Policy Memory for LLM Governance","authors":["Yuanchen Bei","Zhengzhang Chen","Yanjun Zhao","Haoyu Wang","Hanghang Tong","Haifeng Chen"],"year":2026,"abstract":"As large language models (LLMs) are increasingly deployed in real-world high-stakes applications, effective governance has become essential. Existing safeguards largely follow two paradigms: learning-based guards provide strong semantic discrimination but couple policy behavior to trained models and taxonomies, while programmable frameworks offer flexible control but require substantial manual prompt and workflow engineering. Neither externalizes policies as reusable operational states, making i","url":"https://arxiv.org/abs/2609.13734","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01102","type":"paper","title":"ForeSight: Enhancing Risk Monitoring via Early Safety Signal Distillation","authors":["Hanling Wang","Chenlong Wei","Ling Xu","Hanyan Niu","Qi Cao","Shizhou Huang","Yang Yang","Xiaohui Zhu","Yao Zhu"],"year":2026,"abstract":"As large language models (LLMs) are increasingly deployed, the generation of harmful content has become a critical safety concern. Existing safeguards operate at the input, output, or streaming-generation stages, while early-risk methods that rely on surface tokens or output logits may suffer from weak initial signals, and internals-based detectors using dense representations may retain highly entangled and redundant safety-irrelevant information. It therefore remains unclear whether the earlies","url":"https://arxiv.org/abs/2609.13737","categories":["guardrails","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01103","type":"paper","title":"LLMs Leak Training Data Beyond Verbatim Memorization: Extraction via Membership Decoding","authors":["Zi-Tai Chen","Reza Shokri"],"year":2026,"venue":"Proceedings on Privacy Enhancing Technologies","abstract":"Extracting training data from large language models (LLMs) is a serious privacy breach that exposes (potentially private) data without data owners' consent. Existing extractions follow the generation-then-audit paradigm, where the greedy decoding method in generation limits the extraction scope and only verbatim memorized data is under audits. A majority of partially memorized member data (around 90%) remains unexplored, of which LLMs could memorize almost all tokens but fail to rank the trainin","url":"https://www.semanticscholar.org/paper/c52e919714f66688c0620093ff6b9898bd00edc9","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01104","type":"paper","title":"Delayed Backdoor: Let the Trigger Fly for a While in Backdoor Attack on Internet of Things","authors":["Bo Liu","Pei-Wen Zhu","Shao-Feng Zhao","Xiang Chen","Hao-Jie Huang","Li-Ling Shi","Xiao-Kang Wang","Zhi-Gao Zheng","Laurence T. Yang"],"year":2026,"venue":"IEEE Internet of Things Journal","abstract":"Since large language models (LLMs) have gained wide attention as the core of AI agents in the Internet of Things (IoT) for conversation and generation tasks, their security issues have become more prominent, especially regarding backdoor attacks. Traditional backdoor attacks often rely on fixed triggers and static outputs, failing to fully exploit the conversational characteristics and generativity of LLMs, which limits their stealth and attack effectiveness in complex human–agent interactions. ","url":"https://www.semanticscholar.org/paper/b550cd95248dfd0d0c4a44b015958170793ffbb5","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01105","type":"paper","title":"TripPattern: A Pattern-based Text Watermarking Method for Large Language Models","authors":["Sang-Jun Moon","Dasom Choi","Jingun Kwon","Hidetaka Kamigaito","Taro Watanabe","Manabu Okumura"],"year":2026,"abstract":"Text watermarking techniques have gained significant attention for identifying machine-generated text and mitigating risks from large language models (LLMs). Existing methods typically divide an LLM's vocabulary into green and red tokens, but encouraging generation toward green tokens can reduce text quality and naturalness. To address this, we propose TripPattern, a watermarking framework that formulates text watermarking as a pattern-based matching task using three vocabulary partitions. TripP","url":"https://www.semanticscholar.org/paper/d7035dd21e2547207c23fec06a3b22bf29586847","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01106","type":"paper","title":"SoK: Rethinking Jailbreaking in the Era of Agentic AI: Attacks, Defenses, and Practical Consideration","authors":["Md. Jueal Mia","Yanzhao Wu","S. Uluagac","M. Amini"],"year":2026,"abstract":"Large language models (LLMs) are rapidly evolving from conversational assistants into agentic AI systems that reason, plan, invoke tools, maintain persistent memory, communicate with other agents, and execute multi-step tasks. At the same time, modern models exhibit substantially stronger native safety alignment than earlier generations on which many jailbreak attacks and defenses were originally studied. This shift raises a fundamental question: \\textit{which established jailbreak-security find","url":"https://www.semanticscholar.org/paper/1664c9b5d52fbb43aa9582e97dc73096e344adc0","categories":["jailbreaking","agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-01107","type":"paper","title":"A framework for efficient and secure LLM agency: a case for the GraphQL paradigm","authors":["Viktor Zhakhalov"],"year":2026,"venue":"CEUR Workshop Proceedings, Vol-4260: Proceedings of the 8th Workshop for Young Scientists in Computer Science &amp; Software Engineering (CS&amp;SE@SW 2025)","abstract":"LLM agents must translate natural language into concrete actions on external tools. Most systems use JSON-based function calling or, more riskily, let models emit imperative code. We propose a GraphQL-first alternative that reframes tool use as typed, declarative program synthesis against a schema. This yields three measurable advantages. First, efficiency: a token–economy analysis shows that a single GraphQL query replaces multiple RPC calls, reducing request tokens from 63 to 32 and total oper","url":"https://www.semanticscholar.org/paper/f770e3ceaac6d17508b01151897bfcaaecc595f7","categories":["agentic-threats","tool-use-security"],"reviewed":false},{"id":"llmsec-2026-01108","type":"paper","title":"LLM-Based Agents for Cybersecurity: A Systematic Review of Architectures, Applications, and Open Challenges","authors":["George Fatouros","Konstantinos Mavrogiorgos","Georgios Makridis","J. Soldatos","D. Kyriazis"],"year":2026,"venue":"Journal of Cybersecurity and Privacy","abstract":"The rapid evolution of Large Language Models (LLMs) has opened new frontiers in cybersecurity automation, enabling intelligent agents capable of multi-step reasoning, tool invocation, and autonomous decision-making across complex security tasks. While individual applications have emerged across threat intelligence, vulnerability assessment, penetration testing, and security operations center (SOC) automation, a systematic understanding of the LLM-based agent paradigm in cybersecurity—encompassin","url":"https://www.semanticscholar.org/paper/f0888a645148559fc788a0efb5a1c3f4cfd99d92","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01109","type":"paper","title":"Black-Box Red Teaming of Agentic AI: A Taxonomy-Driven Framework for Automated Risk Discovery","authors":["Divyanshu Kumar","Nitin Aravind Birur","Tanay Baswa","Sahil Agarwal","P. Harshangi"],"year":2026,"abstract":"Agentic systems are rapidly moving to production, where they read untrusted inputs, call tools with real permissions, and act autonomously, expanding the security surface beyond chat-only models. Yet standard evaluations remain single-turn and fail to capture multi-step agent vulnerabilities. We present a systematic black-box framework for risk-aware agent evaluation requiring only basic system descriptions. Our approach introduces: (1) a seven-domain taxonomy mapping observable behaviors to ris","url":"https://www.semanticscholar.org/paper/cf0bc84d59f2f81ec5aa3534b5933d67a80205ec","categories":["agentic-threats","red-teaming"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01110","type":"paper","title":"Keep Evaluation Fair: Detecting Data Leakage in Code Generation Benchmarks via Membership Inference Attacks","authors":["Dong-Dong Zhao","Jian Chen","Guan-Cheng Lin","Jianwen Xiang","J. Keung","Xiao Yu"],"year":2026,"abstract":"Code generation benchmarks are widely used to evaluate Large Language Models (LLMs), but benchmark data leakage into training sets can inflate performance and undermine evaluation validity. DetectLeak, a method specifically designed for code generation benchmark leakage detection, relies on perplexity scores to identify likely leaked samples. However, perplexity mainly reflects general familiarity with code patterns and may perform poorly on complex or rare samples. It also overlooks other usefu","url":"https://www.semanticscholar.org/paper/b35edac8aca5fbcd8a42b330512e3a64fd1214cc","categories":["membership-inference","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01111","type":"paper","title":"Hive-AI: a defended multi-service honeypot framework for generative AI APIs","authors":["Sebastián Vargas Yáñez","Sergio Tobón"],"year":2026,"venue":"International Journal of Information Security","abstract":"Public Large Language Model (LLM) APIs draw attacker traffic that defenders cannot see. Probes hit at the semantic layer—past TLS, past Web Application Firewall rules—and conventional intrusion detection picks up almost none of it. No open-source honeypot framework today captures this traffic at scale, and the few LLM-honeypot prototypes that exist push captured logs straight into a downstream LLM analyzer, exposing the analysis pipeline to indirect prompt injection through attacker-controlled i","url":"https://www.semanticscholar.org/paper/33566699ae6da88b8b15e737037575861d87045c","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01112","type":"paper","title":"Suan: Rectifying Direct Preference Safety Alignment in Large Language Models","authors":["O. Cherednichenko","Roman Klypa"],"year":2026,"abstract":"Integrating robust safety guardrails into Large Language Models (LLMs) is essential for delivering helpful yet harmless responses. While proprietary systems exhibit reliable safety controls, their underlying methodologies and trade-offs remain largely undisclosed. Achieving comparable security in open-weight models remains a persistent challenge, as post-trained variants frequently suffer from over-refusal and degraded general quality. To overcome these drawbacks, we introduce Suan, a novel pref","url":"https://www.semanticscholar.org/paper/98dc74badd21f918c2759ca4f4b2e24562dd61bb","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01113","type":"paper","title":"Style Over Substance: Content-Invariant Wrappers Flip LLM Safety-Judge Verdicts","authors":["Yongxin Zhou","Wen-Bo Ye","Yuan-Zhe Liu","Zi-Han Dong","Jun-Wei Yao"],"year":2026,"abstract":"Automatic safety judges -- systems such as Llama Guard or a GPT-4o grading prompt that decide whether a model's reply is harmful -- produce the numbers behind almost every reported jailbreak success rate, defense evaluation, and safety leaderboard. We ask whether these judges grade what a reply contains or how it sounds. We keep a reply's content fixed and add content-invariant style wrappers: fixed strings placed before or after the reply that change only its tone (an educational disclaimer, a ","url":"https://www.semanticscholar.org/paper/8e0549af7045be84295807494a4df1d11e45b6a8","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01114","type":"paper","title":"Integrating LLMs into IoT-Driven Smart Healthcare Systems: A Systematic Literature Review and Future Agenda","authors":["P. Mekala","Yonas Kassa","Sushma Mishra"],"year":2026,"venue":"IoT","abstract":"The convergence of Large Language Models (LLMs) with the Internet of Things (IoT) is driving a transformative shift toward a continuous, context-aware smart healthcare ecosystem. Due to its novelty, existing research in this domain remains fragmented, leaving a critical gap in unified frameworks that synthesize domain applications, functional AI deployment roles, network architectures, and security boundaries. Following PRISMA 2020 guidelines, this paper presents a systematic literature review a","url":"https://www.semanticscholar.org/paper/88e0ea4c56181cfe5e665a70ed7622d3f2d347e5","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01115","type":"paper","title":"EGP-Defense: Enhancing Adversarial Robustness of LVLMs via Training-Free Edge-Guided Prompting","authors":["Bo-Yu Wang","Zi-Wen He","Xin-Jue Hu","Chi Wang","Zi-Qiang Li","Zhang-Jie Fu"],"year":2026,"venue":"ACM Transactions on Multimedia Computing, Communications, and Applications (TOMCCAP)","abstract":"Large Vision-Language Models (LVLMs) have demonstrated remarkable multimodal comprehension capabilities, achieving state-of-the-art performance across various vision-language tasks. However, their performance drops significantly when facing adversarial attacks on the visual encoder. To alleviate this issue, existing approaches often rely on adversarial training, enhancing model robustness through substantial computational cost. Unlike these methods, this paper proposes a novel, training-free adv","url":"https://www.semanticscholar.org/paper/1791b7fdd331177fbea7a13c7100d82d38bbc632","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01116","type":"paper","title":"CrACK: Adversarial Attacks on Cross-Model Consistency in Collaborative Vision Foundation Models","authors":["Fei-Fei Liu","Jintao Cheng","Chi-Man Vong","Xiaoyu Tang"],"year":2026,"abstract":"Training-free collaborative pipelines that integrate Vision Foundation Models such as CLIP, SAM, and DINO achieve strong open-vocabulary dense prediction and are increasingly deployed in safety-critical applications. The security of these systems is commonly assumed to follow from the robustness of their individual models. We challenge this assumption. We identify a vulnerability shared by every collaborative pipeline: each model consumes the intermediate output of another without verifying sema","url":"https://www.semanticscholar.org/paper/c6f9d684d9b92f5baf1af32a027c1320fb953039","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01117","type":"paper","title":"Adversarial Robustness of Foundation Models for Intelligent Mechanical Systems: Threat Models, Benchmarks, and Defense Stacks","authors":["Vishwanath"],"year":2026,"venue":"International Journal For Multidisciplinary Research","abstract":"Foundation models increasingly operate across modalities (vision, language, audio, and vision–language)\nand are deployed in decision-critical pipelines with tool use and retrieval. This expands the adversarial surface: small perturbations to images or audio can flip predictions, carefully crafted text can induce unsafe actions, and cross-modal attacks can exploit representation alignment to produce consistent but wrong outputs. This paper reviews adversarial robustness of foundation models acros","url":"https://www.semanticscholar.org/paper/a7e2035a91824e166f5709ad33a6804ac0407df0","categories":["guardrails","benchmarks","tool-use-security","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01118","type":"paper","title":"Gamark: adaptive entropy-threshold watermarking via gaussian mixture modeling for large language models","authors":["Jing Zhao","Hong-Wei Yang","Heng-Ji Dong","Hui He","Wei-Zhe Zhang"],"year":2026,"venue":"Cybersecurity","abstract":"Existing generative watermarking methods rely on fixed or heuristic entropy thresholds in cross-task generation scenarios, leading to redundant watermark injection, detection noise, and degraded generation quality. To address these limitations, we propose GAMark, a Gaussian Mixture Model (GMM)-based semantics-aware Adaptive entropy threshold waterMarking framework. Motivated by the manifold hypothesis in the logits space, GAMark adopts an offline semantic modeling and online frozen inference par","url":"https://www.semanticscholar.org/paper/745bb40ffc9b6659100956d864c90751dc31e01a","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01119","type":"paper","title":"The Geometry of Refusal: Why Post-Hoc Safety Is Fragile and Pretraining-Time Safety Persists","authors":["Srikanth Malla","Chiho Choi","Joon Choi"],"year":2026,"abstract":"Post-hoc safety training (RLHF, DPO) is the dominant way to align large language models, yet jailbreaks (Zou et al., 2023b), fine-tuning attacks (Qi et al., 2024), and activation-space probes (Arditi et al., 2024) keep recovering the behaviors it was meant to remove. We give this fragility one geometric explanation and trace it to when, during pretraining, safety can take hold. We measure the safety update $\\Delta = W_{\\text{safe}} - W_{\\text{base}}$ against the curvature of the model's capabili","url":"https://www.semanticscholar.org/paper/6db11455cdf61ce01b1e966dcdd60768ac4e03d4","categories":["jailbreaking","guardrails","fine-tuning-security"],"reviewed":false},{"id":"llmsec-2026-01120","type":"paper","title":"Toward Dynamic and Risk-Aware Evaluation of Cybersecurity LLMs: A Survey and the RIRAG Framework","authors":["Ravi Prasad","Feroz Ahmed","Shohel Rana","Charan Gudla","Sujan Kumar Reddy Challa","Aditya Garg"],"year":2026,"venue":"Journal of Cybersecurity, Digital Forensics and Jurisprudence","abstract":"The rapid adoption of large language models (LLMs) in cybersecurity has created a growing need for evaluation methods that reflect operational risk rather than isolated language capability. Existing cybersecurity benchmarks assess useful dimensions such as factual knowledge, vulnerability analysis, secure coding, penetration testing, and threat intelligence reasoning, but many remain limited by static datasets, weak diagnostic granularity, limited adversarial testing, and insufficient attention ","url":"https://www.semanticscholar.org/paper/6d91d3a37230c466cdddb9be0f848f971d28b224","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01121","type":"paper","title":"What a Model Refuses, a State Fears: How Authoritarian Information Control Reproduces in Language-Model Guardrails","authors":["Meng-Lin Liu","Yao Yu","Tong Wu","Chun-Ran Zhang","Ge Shi"],"year":2026,"abstract":"As large language models become the front door to political information, what they refuse to discuss becomes a new instrument of information control. We argue that a model's guardrail encodes not a universal notion of harm but the political threat model of the state that governs its developer, and we derive the expected structure of that control from the comparative study of how authoritarian regimes censor. Across ten models and three languages, Chinese guardrails carry its signatures: they ans","url":"https://www.semanticscholar.org/paper/475fe2ba2121de45c572735d75d9efa667deaa30","categories":["guardrails","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01122","type":"paper","title":"FreqDoor: A Hidden Trojan in the Frequency Domain for Backdoor Attacks on Vision-Language Models","authors":["Yasir Arafat Prodhan","Sadad Hasan","Mohammed Imamul Hassan Bhuiyan"],"year":2026,"abstract":"Vision-language models (VLMs) have recently shown excellent progress in open-ended image-to-text generation. However, their multimodal nature makes them persistently vulnerable to backdoor attacks. Existing backdoor triggers for VLMs are either spatial, textual, or bimodal, which may yield localized or recognizable trigger patterns. In this work, we explore a different attack surface and propose \\ textsc {FreqDoor}, a training-time backdoor attack that implants triggers in the frequency domain. ","url":"https://www.semanticscholar.org/paper/15ae338e8edb475cce1e45bcd3eabb628cbf098a","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01123","type":"paper","title":"ARES: Securing Agents for Computer Use Through Endpoint Resource Mediation and Behavioral Guardrails","authors":["Changhee Kim","Seong-je Cho"],"year":2026,"venue":"Electronics","abstract":"Large language model (LLM)-based agents are evolving into agents for computer use (ACUs) that read files, invoke applications, communicate over networks, and operate graphical interfaces, moving the effective security boundary from model inputs and outputs to autonomous actions that alter endpoint state. Conventional identity and access controls remain applicable and necessary, but they are authorized based on identity, resource, and network policy rather than on the semantic scope of the active","url":"https://www.semanticscholar.org/paper/fd1a22d2baba2df7a0db52925828232b7a69ebb0","categories":["guardrails","access-control"],"reviewed":false},{"id":"llmsec-2026-01124","type":"paper","title":"The Double-Edged Sword of AI Pair Programmers: A Systematic Literature Review of Security Vulnerabilities in AI-Generated Code and Agentic Development Environments","authors":["Mahmoud E. Farfoura","M. Alia","Ibrahim Mashal","Adnan A. Hnaif","M. Odeh"],"year":2026,"venue":"Journal of Sustainable Smart Systems in Education &amp; Environment","abstract":"The role of AI pair programmers has expanded from local code completion to active participation in the development environment. Contemporary tools can interpret repository context, edit multiple files, call package managers, execute terminal commands, and communicate with external services. This review synthesizes security evidence concerning GitHub Copilot, ChatGPT-based coding, code large language models, Cursor-style agentic editors, command-line coding agents, and Model Context Protocol ecos","url":"https://www.semanticscholar.org/paper/f68e78d6c07988ee249898866e6cf165d2adb5a5","categories":["agentic-threats","survey"],"reviewed":false},{"id":"llmsec-2026-01125","type":"paper","title":"Uncensored Open-weight Models: Redistribution as the Persistence Layer","authors":["10a Labs Juliette Garcia","Hailey May","Bobby McKenzie","David Pham","Matthew Swain","J. Valdez","Corie Wieland","Zachary Yahn"],"year":2026,"abstract":"A rapidly expanding ecosystem of actors is removing built-in safety guardrails from open-weight AI models. We profile this ecosystem by identifying key producers, downstream reproductions, and emerging applications. Between January 2024 and March 2026, we identified 3,471 original uncensored models on HuggingFace, each repackaged an average of 2.4 times; three actors account for 52% of all 8,164 compressed redistributions. Once quantized and mirrored across separate accounts, formats, and regist","url":"https://www.semanticscholar.org/paper/02e6517263f5bd037bfd4496436facd52b816d53","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01126","type":"paper","title":"LLM-as-a-Judge Is Not an Oracle: Why Self-Improving Agents Need Deterministic Guardrails","authors":["Vansh Wahi"],"year":2026,"abstract":"Self-improving agent pipelines have a problem at their center. An optimizer rewrites prompts to score higher, and the score comes from a judge that is itself an LLM. That judge has the last word on whether the system is getting better, and our position is that it has not earned it. The judge should be demoted from oracle to advisor: its verdict becomes one input among several, and every change is gated instead by a deterministic verification layer the judge cannot override. We reached this posit","url":"https://www.semanticscholar.org/paper/d9bca7473eb20a165dd64c8710affd0e7263fdc6","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01127","type":"paper","title":"Who Drives the Probability Game of VLMs? A Temporal Causal Drive Evaluation Framework","authors":["Shu-Yao Xiao","Sheng-Ling Wang","Hao-Yu Niu","Ke Chao","Changbo Xu","Xin-Ran Duan","Chao-Yong Jiang"],"year":2026,"abstract":"Vision-language models (VLMs) are increasingly evaluated on complex image and video understanding tasks, yet conventional metrics primarily assess final-answer quality and reveal little about how different information sources shape the generation process. We propose a causal and temporal evaluation framework that traces the evolving roles of visual input, question text, and generated prefixes during autoregressive decoding. Grounded in a Structural Causal Model, we use interventions and backdoor","url":"https://www.semanticscholar.org/paper/c1f5686528f9fe5fd5b7db14e9d275a6a17ac9ab","categories":["data-poisoning","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01128","type":"paper","title":"Leakage-Aware Cross-Dataset Evaluation of Prompt Injection Detection Using Classical Machine Learning and Transformer Models","authors":["Oğuzhan Kilim"],"year":2026,"venue":"Yalvaç akademi dergisi","abstract":"The widespread adoption of systems based on Large Language Models has made the reliable detection of prompt injection attacks a critical requirement. However, high performance achieved on training and test splits generated from the same data source does not guarantee that models can generalize to prompts from different sources. In this study, a leak-aware cross-dataset evaluation framework is presented to examine the robustness of classical machine learning and Transformer-based prompt injection","url":"https://www.semanticscholar.org/paper/9927bbc6c11def4346dc8112ca690825efde1f06","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01129","type":"paper","title":"CAPTURE: Disentangling Preference Drift from Memory Poisoning in Personalized LLM Agents","authors":["Asif Hossain","Ruksat Khan","Shayoni","Kishor Morol"],"year":2026,"abstract":"Personalized language agents use persistent memory to adapt to users over time, but the same mechanism creates an attack surface. When new information conflicts with stored preferences, an agent must distinguish genuine preference drift from temporary context shifts, ambiguity, or adversarial memory poisoning. We formulate this problem as a continuous-time partially observable decision process over a latent user state and show why rules based only on recency and provenance are insufficient. CAPT","url":"https://www.semanticscholar.org/paper/6c2ef186b5091367d9c2667d2f3013a2b6f79ab7","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-01130","type":"paper","title":"InfraPatch: Cross-Task Targeted Grayscale Patch Attacks on Infrared-Adapted Vision-Language Models","authors":["Chengyin Hu","Ding-Yi Lu","Jiajun Han","Xiang Chen","Weiwen Shi","Jiahuan Long","Yiwei Wei","Jiu-Jiang Guo"],"year":2026,"abstract":"Infrared vision-language models (IR-VLMs) have emerged as a promising paradigm for multimodal perception under low-visibility conditions, yet their robustness to targeted adversarial attacks remains poorly understood. Existing adversarial patch methods mainly study RGB-based models or a single downstream task and do not characterize whether localized perturbations can induce an intended semantic target in IR-VLMs. We propose InfraPatch, a white-box, per-instance framework for targeted digital gr","url":"https://www.semanticscholar.org/paper/286d27171af8b29fb92ec42d28ea825cb62a78eb","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01131","type":"paper","title":"Beyond the Verdict: Evidence-Aligned Evaluation of Visual Prompt-Injection Guardrails","authors":["Suyoung Lee","Myungsub Choi"],"year":2026,"abstract":"Verdict-only evaluation does not reveal whether a vision-language model (VLM) used the visual evidence that should support its decision. We study this problem in web-agent guardrails, where a VLM judges whether on-screen text conflicts with a user instruction. We introduce Mind2Web-Injection, a benchmark of 9,954 instruction-screenshot pairs with instruction-relative labels, pixel-exact evidence boxes, and matched image-side counterfactuals. Across six VLMs, two models with nearly identical aver","url":"https://www.semanticscholar.org/paper/1777bccb47960930d9560230d9b519fed1ab95ca","categories":["prompt-injection","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01132","type":"paper","title":"ShadowCode: Toward (Automatic) External Prompt Injection Attack Against Code LLMs","authors":["Yuchen Yang","Yi-Ming Li","H. Yao","Bing-Run Yang","Yiling He","Tianwei Zhang","Dacheng Tao","Z. Qin"],"year":2026,"venue":"IEEE Transactions on Dependable and Secure Computing","abstract":"Recent advancements have led to the widespread adoption of code-oriented large language models (Code LLMs) for programming tasks. Despite their success in deployment, their security research is left far behind. This paper introduces a new attack paradigm: (automatic) external prompt injection against Code LLMs, where attackers generate concise, non-functional induced perturbations and inject them within a victim’s code context. These induced perturbations can be disseminated through commonly use","url":"https://www.semanticscholar.org/paper/f7d0fe61df3d82faa00b5379d47ee364d85f4d2a","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01133","type":"paper","title":"The adversarial game between detection and evasion: A survey of anti-detection techniques for machine-generated texts.","authors":["De-Yu Meng","Tad Gonsalves"],"year":2026,"venue":"Neural Networks","abstract":"With the explosive growth of large language models (LLMs), research on machine-generated text detection (MGTD) has also proliferated. Alongside these developments, a wide range of attack algorithms targeting MGTD systems have emerged. While previous studies have surveyed detection techniques, few have examined the dynamic interplay between attack and defense. Following PRISMA 2020, this paper systematically synthesizes 27 studies of attacks against MGTD and the available evidence on correspondin","url":"https://www.semanticscholar.org/paper/d918d543553f11905ebae1ae12cb2d221c5fb2e4","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01134","type":"paper","title":"MPDA: Multimodal Prompt Decoupling Attack on the Safety Filters in Text-to-Image Models","authors":["Xingkai Peng","Jun Jiang","Meng Tong","Shuai Li","Weiming Zhang","Neng H. Yu","Kejiang Chen"],"year":2026,"venue":"IEEE Transactions on Dependable and Secure Computing","abstract":"Text-to-image (T2I) models have been widely applied in generating high-fidelity images across various domains. However, these models may also be abused to produce Not-Safe-for-Work (NSFW) content via jailbreak attacks. Existing jailbreak methods primarily manipulate the textual prompt, leaving potential vulnerabilities in image-based inputs largely unexplored. Moreover, text-based methods face challenges in bypassing the model’s safety filters. In response to these limitations, we propose the Mu","url":"https://www.semanticscholar.org/paper/d71b2075655a52c19b88c09dfa9c50250927ade5","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01135","type":"paper","title":"GenPot: A generative honeypot architecture for adaptive web and API interaction","authors":["Antonio Lara-Gutierrez","Juan Zamorano","J. A. Onieva"],"year":2026,"venue":"Applied intelligence (Boston)","abstract":"Honeypots are widely used as cyber-deception tools to study adversarial behaviour, yet their effectiveness is limited by a trade-off between realism and security risk. Low-interaction honeypots are easily detected, while high-interaction honeypots provide realistic data at the cost of network exposure. Recent advances suggest that Large Language Models (LLMs) can mitigate this trade-off by dynamically generating convincing outputs without requiring a vulnerable backend. In this paper, we present","url":"https://www.semanticscholar.org/paper/c8446206f0a9cba0979e0efba154610fd12b08a3","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01136","type":"paper","title":"When Guardrails Look Effective: Construct Validity Failures in LLM Agent Commerce Evaluation","authors":["Pei-ke Zhu","Sidi Chang"],"year":2026,"abstract":"Interactive simulations increasingly evaluate policies in markets populated by language-model agents. Their outputs can look economic---prices, profits, consumer surplus, and welfare---without instantiating the behavior named in the claim. We audit this risk in a multi-turn buyer--seller testbed for configurable hotel transactions. An initial implementation reported welfare gains from two marketplace guardrails of +87.4, +35.0, and +28.8 across a Qwen2.5 1.5B--14B ladder. It also gave guarded an","url":"https://www.semanticscholar.org/paper/96c5c7670b65dee1adc74eab2b024fb568032514","categories":["agentic-threats","guardrails"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01137","type":"paper","title":"Beyond Single-Pair Attacks: Disrupting Vision-Language Pre-Training Models With Dual-Semantic Frequency Stealth","authors":["Hai-Qi Zhang","Zi-Qiang Li","Hao Tang","Ze-Chao Li"],"year":2026,"venue":"IEEE Transactions on Dependable and Secure Computing","abstract":"Vision-Language Pre-training (VLP) models are highly capable in multimodal tasks but are critically vulnerable to adversarial attacks. Existing methods for creating transferable adversarial examples typically operate by modifying semantics within isolated image-text pairs. This strategy often yields poorly generalizable perturbations that insufficiently disrupt cross-modal alignment, leading to limited effectiveness in black-box settings. An additional limitation is the frequent oversight of adv","url":"https://www.semanticscholar.org/paper/94fb165ddf17a75721511d02381aa3da9e76fd1d","categories":["adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-01138","type":"paper","title":"IC-GCG: Jailbreaking Large Language Models via Intermediate Consistency Optimization","authors":["Zichu Ren","Donghai Zhu","Haibo Hong","Jun Shao"],"year":2026,"venue":"IEEE Internet of Things Journal","abstract":"Recent jailbreak attacks demonstrate that large language models (LLMs) can be manipulated to generate harmful outputs through adversarial prompts even after robust alignment. However, prevailing methods typically focus on forcing a desired response at the output layer—a surface-level strategy that is brittle and often fails to bypass the more fundamental safety checks embedded within the model’s internal mechanisms. In contrast, we propose intermediate consistency greedy coordinate gradient (IC-","url":"https://www.semanticscholar.org/paper/7ecc8bf212ba04a3db3043938853af5bca5e6533","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01139","type":"paper","title":"DIVA: Exploiting Cross-Step Conditional Propagation for Visual Jailbreaks in Discrete Diffusion Vision-Language Models","authors":["Guo-Rui Song","Run-Qing Tang","Jing-Ye Zhang","Lu-Yuan Zhang","Fei Huang","Cong Ray","Guo-Cun Wang","Da-Ke Zhong","Choo Sin Wai","Bing-Quan Dai","Chu-Ming Wang","Tongxu Lin","Wan-Yu Guo","Haoqian Wang"],"year":2026,"abstract":"Large vision-language models (VLMs) are increasingly deployed in safety-critical settings, yet existing visual jailbreak research has focused almost exclusively on autoregressive architectures, leaving an important emerging family unstudied: multimodal discrete diffusion vision-language models (dVLMs). We identify a vulnerability specific to diffusion generation: because the visual embedding conditions every reverse denoising step rather than acting as a one-time prefix, adversarial visual seman","url":"https://www.semanticscholar.org/paper/75c7781c1563ea318cbc35ebef975200b441c40f","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01140","type":"paper","title":"TADSOB: Enhancing jailbreaking attacks on LLMs via trajectory-aware dynamic suffix optimization with buffer","authors":["Xiao-Long Li","Mingrui Lao","Cheng-Si Du","Yun-Hao Feng","Xiao-Hu Du","Liang-Hu Bai","Yanming Guo"],"year":2026,"venue":"Neurocomputing","url":"https://www.semanticscholar.org/paper/6e9c31860f812832918a21d8f94acf00ab479ada","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01141","type":"paper","title":"Trustworthy LLM–GNN systems: a systematic review and multi-axis taxonomy","authors":["Rui-Zhan Xue","Fang He","Hui-Min Deng","Mao-Jun Wang","Ze-Yu Zhang"],"year":2026,"venue":"Applied intelligence (Boston)","url":"https://www.semanticscholar.org/paper/4192b731d75b4dc39b2f529666eabc66cf68b755","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01142","type":"paper","title":"PERSIST\n : Threat Modeling Memory‐Persistent\n AI\n Agents in Cloud‐to‐Edge Environments","authors":["Albert Adusei Brobbey","Narayan P. Bhosale"],"year":2026,"venue":"Security and Privacy","abstract":"\n Agentic artificial intelligence systems increasingly depend on persistent runtime memory, including vector databases, episodic memory stores, long‐term retrieval indices, and cloud‐to‐edge replicas. Existing security frameworks address prompt injection, data poisoning, and model‐level risks, but they do not fully model persistent memory as an active behavioral surface whose records can be written, synchronized, retrieved, transformed, revoked, and later used as decision context. This article i","url":"https://www.semanticscholar.org/paper/2b34f27f1c8e827f22f4e6d347c008446794b992","categories":["prompt-injection","data-poisoning","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01143","type":"paper","title":"SRD: A model-agnostic semantic framework for jailbreak mitigation in large language models","authors":["Can Shi","Zhi-Yong Zhang","Gaoyuan Quan","Junyang Pan","Xinxin Yue"],"year":2026,"venue":"Knowledge-Based Systems","url":"https://www.semanticscholar.org/paper/115084858292f9df35a27f2f2aa6e906a7587751","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01144","type":"paper","title":"Efficient Prompt Security Detection for LLM Service Deployment in Edge-Cloud Networks","authors":["Wen-Jing Chen","Jie Cui","Wenjie Huang","Jing Zhang","Lu Wei","Geyong Min"],"year":2026,"venue":"IEEE Transactions on Dependable and Secure Computing","abstract":"While Large Language Models (LLMs) have achieved revolutionary advancements in natural language processing, their inherent vulnerability to prompt injection attacks has raised significant security concerns. Existing security detection approaches for LLM deployment often fail to fully leverage edge-cloud collaboration and largely overlook the issue of user heterogeneity, resulting in reduced detection efficiency and security risks. To address these challenges, this paper proposes a Belief-updatin","url":"https://www.semanticscholar.org/paper/084aed427d1488610dcf103935fe66ad361ee86d","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01145","type":"paper","title":"Understanding Stage-Wise Utility-Risk Trade-offs in LLM Agent Memory","authors":["Chuanchao Zang","Zi-Jian Cao","Xiangtao Meng","Jianing Wang","Wenyu Chen","Xinyu Gao","Li Wang","Zheng Li","Shanqing Guo"],"year":2026,"abstract":"Long-term memory is becoming a core capability of LLM agents, enabling personalization and long-horizon interaction. However, memory mechanisms that retain, transform, or expose more information can affect both benign utility and susceptibility to memory poisoning. Existing evaluations typically measure memory utility or attack risk in isolation under fixed configurations, providing limited insight into how stage-specific design choices reshape their trade-off. We present \\textsc{MemGauge}, a co","url":"https://www.semanticscholar.org/paper/d9a499f524390e5452dad9a4818c8988506c1a65","categories":["data-poisoning","agentic-threats","memory-security"],"reviewed":false},{"id":"llmsec-2026-01146","type":"paper","title":"SingProbe Technical Report","authors":["Singg Team"],"year":2026,"abstract":"Runtime guardrails are essential for reliable large language model (LLM) deployment, yet existing approaches typically rely on independent, external models that introduce additional inference cost, delayed safety signals, and a capacity mismatch with increasingly capable base models. To address these issues, we introduce SingProbe, a lightweight intrinsic runtime guard that directly reuses hidden states produced during LLM inference and operates alongside autoregressive decoding. Within a unifie","url":"https://www.semanticscholar.org/paper/b42aefcf52891422ad5d7a2d53a60611fe663afb","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01147","type":"paper","title":"Do VLMs Share Safety Neurons Across Modalities?","authors":["Jia-Xuan Li","Jia-Hao Zhang","D. Vo","H. Nguyen","Pride Kavumba","Koki Wataoka"],"year":2026,"abstract":"Vision-language models (VLMs) can comply with harmful requests delivered through images, even when their LLM backbones would refuse the same content in text. While prior work characterizes these jailbreaks empirically or at the representation level, how visual inputs perturb safety pathways at the neuron level remains uncharted. We close this gap with a causal, neuron-level analysis of safety mechanisms in 10 VLMs. We propose a two-stage detection pipeline with iterative ablation that accounts f","url":"https://www.semanticscholar.org/paper/8a8800ed3022d7532c7daf5782a597449f10b2b0","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01148","type":"paper","title":"Evaluating Indirect Prompt Injection Defenses in Tool-Using LLM Agents: Security, Utility, and Replication","authors":["Adil Khan","Khaled AlKhanbashi","Azza Mohamed"],"year":2026,"venue":"Computers","abstract":"Large language model (LLM) agents that retrieve external content and use tools are vulnerable to indirect prompt injection, in which untrusted content contains instructions intended to influence agent behavior. We evaluated four defenses and an undefended control across GPT-5.4, GPT-5.4-mini, and Claude Sonnet 4.6 on the AgentDojo banking benchmark (Tool Filter was evaluated only for the OpenAI models), reporting attack success rate (ASR), benign utility, utility under attack, operational measur","url":"https://www.semanticscholar.org/paper/857a1c7d30ff72cce69c0067e8c4daa739bf41f3","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01149","type":"paper","title":"EvoSkill Injection: Red-Teaming Autonomous Skill Generation and Evolution in Self-Evolving Agents","authors":["Doyun Kim","Chanwoo Kim","Sugyeong Eo","Yeo-Chan Yoon","Chanjun Park"],"year":2026,"abstract":"LLM-based agent systems increasingly adopt skill-based architectures to reduce repetitive reasoning costs and improve stable, efficient task execution. Recent studies propose self-evolving agents that autonomously generate, refine, and reuse skills from past experiences to enable continuous capability evolution. However, autonomous skill evolution introduces a new attack surface in which malicious capabilities are generated, stored, and reused as legitimate skills. In this paper, we define EvoSk","url":"https://www.semanticscholar.org/paper/74b1d8292cfadce66a9d3467f4fcdd8246f7ba73","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-01150","type":"paper","title":"The Safety Relay in Roleplay Jailbreaks: A Component-Resolved Causal Analysis of Harm Recognition and Refusal","authors":["M. Chowdhury","Ernie Chang","Yang Li"],"year":2026,"abstract":"Large language models are trained to follow instructions while refusing harmful requests. Jailbreaks exploit this balance to elicit content a model would ordinarily reject. Roleplay jailbreaks are especially concerning: the harmful request can remain visible inside a roleplay wrapper made of a persona, scenario, and task, yet the model may comply. We use mechanistic interpretability to determine how this context reverses refusal and which elements contribute to the reversal. Across two benchmark","url":"https://www.semanticscholar.org/paper/702d373e63d2c06fbb60a94f8ad67cab9be00732","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01151","type":"paper","title":"Balancing Privacy, Utility, and Safety in LLM Alignment through Preference Optimization","authors":["Di-Shu Yang","Jing-Jing Liu","Ji-Ze Li"],"year":2026,"abstract":"Preference optimization is widely used to align large language models with human preferences, but preference-data composition may also influence privacy-relevant memorization. We examine whether adding synthetic privacy-preference pairs to Direct Preference Optimization (DPO) is associated with lower canary-based memorization signals without modifying the objective or introducing a formal privacy mechanism. We propose Privacy-Pressure Preference Mixing (P3M), a data-composition protocol that var","url":"https://www.semanticscholar.org/paper/242428093c491f916ba347c255fbd3ae086e8fb5","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01152","type":"paper","title":"Validity-Aware Jailbreak Evaluation for Large Language Models","authors":["Qilong Wu","Sahil Wadhwa","Pranab Mohanty","Giri Iyengar","Varun Chandrasekaran"],"year":2026,"abstract":"Jailbreak robustness has become central to large language model (LLM) safety evaluation, yet prevailing methodologies rely primarily on refusal behavior, semantic resemblance, and intent-matching heuristics that emphasize linguistic plausibility rather than correctness. We identify a key limitation in existing evaluations: many jailbreak intents depend on instructional validity rather than epistemic factuality, allowing realistic-looking responses to be labeled successful despite being factually","url":"https://www.semanticscholar.org/paper/08abe34670f9011fc3ff59cbab6c6328b2480af4","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01153","type":"paper","title":"SemTrace: Source-Grounded Semantic Signatures for Tracing LLM Exposure to Protected Documents","authors":["Jun-Yan Zhang","Yuan Zeng","Yong-Wei Huang","Zu-Hao Ouyang","Hong Chen","Xuming Hu"],"year":2026,"abstract":"Large language models are increasingly used to read documents and produce downstream text, creating a provenance problem when the document owner cannot control or inspect the model that performs the generation. We introduce SemTrace, a source-grounded semantic watermark for detecting whether a generated review was influenced by a known protected manuscript copy. Rather than biasing token probabilities or imposing surface-form patterns, SemTrace constructs a document-specific binary signature fro","url":"https://www.semanticscholar.org/paper/d864a78fe762e931a96efe5ea38bdf249372cacd","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01154","type":"paper","title":"Reachability-Based Capability Confinement for LLM Agents under Indirect Prompt Injection","authors":["Wu-Jie Xiong","Rabimba Karanjai","Yang Lu","W. Shi","Lei Xu"],"year":2026,"abstract":"Large language model agents place outputs from external skills into their execution context, allowing attacker-controlled data to influence later privileged actions. Existing defenses mainly classify untrusted content or authorize proposed operations. They do not directly address how an agent's future authority should change once untrusted data enters its state. We present SkillGuard, a harness-level enforcement layer that treats this event as contamination and restricts future capabilities to d","url":"https://www.semanticscholar.org/paper/ca9874d087d8df198b302c54402b0df91d19bbbc","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01155","type":"paper","title":"TACS: Trajectory-Aware Candidate Selection for LLM Jailbreak Suffix Optimization","authors":["Shi-Liang Xiao"],"year":2026,"abstract":"Gradient-based jailbreak suffix optimization methods typically update the suffix by retaining the candidate with the lowest current loss. We show that this seemingly natural design is fundamentally myopic: candidates that look better under the current-step proxy often fail to produce better jailbreak outcomes later in the search, revealing a form of selection-stage reward hacking. This suggests that candidate selection, rather than candidate generation alone, is a hidden bottleneck in suffix opt","url":"https://www.semanticscholar.org/paper/9f5d1542f077730ad7f968a1028ee7fc7cf9f544","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01156","type":"paper","title":"Guardrail-Agnostic Societal Bias Evaluation in Large Vision-Language Models","authors":["Yusuke Hirota","Michael Boone","Arun George Zachariah","J. Varghese","Y. Wang","Boyi Li","Ryo Hachiuma"],"year":2026,"abstract":"We propose a societal bias evaluation method for large vision-language models (LVLMs) in the era of strong safety guardrails. Existing benchmarks rely on prompts that ask models to infer attributes of people in images (e.g.,\"Is this person a CEO or a secretary?\"). However, we find that LVLMs with strong guardrails, such as GPT and Claude, often refuse these prompts, making evaluations unreliable. To address this, we change the prior evaluation paradigm by decoupling the task from the depicted pe","url":"https://www.semanticscholar.org/paper/22de4aaa0b81f2b475d9aa1ee401cc4f5c46172c","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01157","type":"paper","title":"Learning to Follow In-Context Watermark Instructions via Self-Distillation","authors":["Yepeng Liu","Tian-Yi Chen","Xuandong Zhao","D. Song","Yuheng Bu"],"year":2026,"abstract":"In-context watermarking (ICW) prepends an instruction to a query asking the model to embed a statistically detectable signal in its response. It thus equips LLMs with a watermarking interface that third parties can invoke without access to model internals. Its reliability hinges on the LLM following the instruction without degrading answer quality, yet how well current LLMs do so has not been measured. We introduce $\\mathsf{ICWBench}$, a benchmark of three verifiable ICW instruction families, ea","url":"https://www.semanticscholar.org/paper/e86439f940d69ec0811550dec8e79a7b1d39a73e","categories":["watermarking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01158","type":"paper","title":"WoE Wrote It? Watermarking Mixture-of-Experts LLMs for Black-Box Text Provenance","authors":["Jona te Lintelo","Lichao Wu","S. Picek"],"year":2026,"abstract":"Large Language Model (LLM) watermarks provide a mechanism for text provenance, enabling model owners to identify machine-generated content and attribute it to a specific watermarked model. However, current LLM watermarking approaches predominantly rely on inference-time sampler methods and focus their analysis on dense models. Inference-time methods are only effective when the text is explicitly generated via the model owner's controlled API; they fail in a post-compromise scenario. An adversary","url":"https://www.semanticscholar.org/paper/dbfae6a9f6cc8380ca06c4c8bf9f1b077a3bd3ba","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01159","type":"paper","title":"Auditing and Mitigating Privacy Leakage in Cloud-Edge Collaborative Decoding","authors":["Ke-Jia Zhang","Tianyuan Zou","Zi-Xuan Gu","Yang Liu"],"year":2026,"abstract":"Applications such as personalized assistance and proprietary document analysis require large language models (LLMs) to generate outputs from private data. Yet powerful LLMs typically cannot be deployed on the resource-constrained devices where private data resides, and uploading private data to cloud-hosted LLMs exposes sensitive information. Recent work addresses this tension with a cloud-edge collaborative decoding paradigm, where private data are kept on the edge with a small language model (","url":"https://www.semanticscholar.org/paper/96dedb871e68ffa16687ef79c2cff3a2ca3b1a92","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01160","type":"paper","title":"Facts Without Rules: Boundary Metadata Collapse in Multi-Agent LLM Handoffs","authors":["Yian Wang","Agam Goyal","Eshwar Chandrasekharan","Hari Sundaram"],"year":2026,"abstract":"Multi-agent LLM systems often coordinate by compressing an upstream interaction into a handoff artifact that downstream agents treat as shared state. We show that this handoff step is a structural source of privacy leakage: summaries preferentially preserve operational facts while weakening the boundary metadata that governs how those facts may be used---a failure mode we call \\emph{summary collapse}. On a controlled multi-agent coordination testbed we measure marker survival with a human-valida","url":"https://www.semanticscholar.org/paper/6dd494eead5b8fdb8340c33b0a3a4deef955323f","categories":["membership-inference","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01161","type":"paper","title":"FISGuard: Defending Against Membership Inference via Fixed Input Subspaces","authors":["Hao-Cheng Jiang","Hua Shen"],"year":2026,"abstract":"As large language models are increasingly adopted in federated learning, protecting user privacy while performing parameter-efficient fine-tuning on distributed private data has become an important challenge. Although clients only share gradients instead of directly uploading raw data, the shared gradients may still leak membership information about training samples. ProjRes (S&P, 2026) further increases this risk: with less information and without accessing model outputs, an attacker can effect","url":"https://www.semanticscholar.org/paper/ddc1bf96c1d3e2b92361ca225366c16b962434b3","categories":["membership-inference","federated-learning"],"reviewed":false},{"id":"llmsec-2026-01162","type":"paper","title":"CamoDocs: A Poisoning Attack Against Retrieval-Augmented Language Models Using Camouflaged Documents","authors":["Jaewon Jung","Haizhong Zheng","Hongsun Jang","Jaeyong Song","Beidi Chen","Jinho Lee"],"year":2026,"abstract":"Retrieval-augmented generation (RAG) augments LLMs with external documents, but public or user-editable sources expose RAG systems to data poisoning: attackers can inject malicious documents to steer outputs toward targeted answers. Existing poisoning attacks often rely on query inclusion, inserting the target query into poisoned documents to improve retrieval; however, this creates lexical and embedding-space artifacts that make them easy to filter. We propose CamoDocs, a poisoning attack that ","url":"https://www.semanticscholar.org/paper/8cb58f1bfdd0a241e70a6b4b0058fb0e7c67e6bb","categories":["data-poisoning"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01163","type":"paper","title":"OpenStamp: A Watermark for Open-Source Language Models","authors":["Miroojin Bakshi","Saksham Rastogi","Danish Pruthi"],"year":2026,"abstract":"With the growing prevalence of large language model (LLM) generated content, watermarking is considered a promising approach for attributing text to LLMs and distinguishing it from human-written content. A prominent class of techniques embeds subtle but detectable signals in generated text by modifying token sampling probabilities. However, such methods are unsuitable for open-source models, where users have white-box access and can easily disable watermarking during inference. In this work, we ","url":"https://www.semanticscholar.org/paper/4e7740b43503b5154dae21b1aae91874490a9ac4","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01164","type":"paper","title":"Agentic AI Security in Industry 5.0: Emerging Threats, Forensic Readiness, and Trustworthy Human–Agent Collaboration","authors":["Maurice E. Dawson","A. B. Ben Ayed","Samson Quaye"],"year":2026,"venue":"Information","abstract":"Industry 5.0 places autonomous agents inside its human–agent collaborative loop, resting its human-centricity pillar on an assumption of trustworthy collaboration that has not been examined critically. Agentic artificial intelligence has moved from research demonstration to industrial deployment within months, introducing a threat class this setting has not yet addressed. This article bridges three literatures developed in isolation, Industry 5.0 cybersecurity, agentic AI security, and Industry ","url":"https://www.semanticscholar.org/paper/f1f02c5e461ba224a1ec673b6b241d2e7c77881c","categories":["agentic-threats","autonomous-operations"],"reviewed":false},{"id":"llmsec-2026-01165","type":"paper","title":"Governing generative AI in higher education: emerging policy approaches and support ecosystems at innovative U.S. Universities","authors":["Yu-Feng Qian"],"year":2026,"venue":"International Journal for Educational Integrity","abstract":"This study examines how the 50 U.S. universities ranked as most innovative by U.S. News & World Report articulate policy, guidance, and support for generative artificial intelligence (GenAI) in teaching and learning. Using qualitative document analysis and inductive thematic analysis of official institutional websites, the study identifies five convergent patterns: instructor-led, syllabus-level governance; disclosure and attribution expectations; privacy-oriented data guardrails and vetted tool","url":"https://www.semanticscholar.org/paper/604b5d28db3a65bf10fe52bb209b03699fb2268e","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01166","type":"paper","title":"Quantization-Triggered Backdoors in Language Models: Cross-Quantizer Transferability and the Validation--Deployment Gap","authors":["Jacopo Dardini","Claudio Stanzione","G. Colò","G. Fenza"],"year":2026,"abstract":"Post-training quantization is often treated as a semantically neutral optimization for edge deployment of Large Language Models. When a full-precision source checkpoint is evaluated and quantization is applied downstream without equivalent re-evaluation, this workflow creates a structural validation--deployment gap: because quantization is a many-to-one mapping over parameter space, source-precision certification does not guarantee behavioral equivalence in the deployed configuration. We formali","url":"https://www.semanticscholar.org/paper/4814c6bd8be0dc62cbee06a2fb1d391b5cfc7191","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01167","type":"paper","title":"Guardrail Detection in Rural Roads Using Zero- and Few-Shot Learning with Vision Language Models","authors":["M. Sadeghi","Ali Mansouri","Abdolmajid Erfani"],"year":2026,"venue":"Construction Research Congress 2026","url":"https://www.semanticscholar.org/paper/32e0021bde133d1dfb10f2364463d7bb4208d0d4","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01168","type":"paper","title":"The Guard That Cried Wolf: How Scary Words Make Agent Guardrails Refuse Legitimate Actions","authors":["Yingjie Zhang","Yuanbo Xie","Kai Chen"],"year":2026,"abstract":"Agent guardrails are checks that approve or refuse each action before an LLM executes it. Sometimes they refuse requests that are genuinely safe. This over-safety blocks deployment when a guardrail refuses an authorized task. Evaluating over-safety is hard: at the boundary an authorized action resembles an unauthorized one, and the safe-versus-unsafe label is a choice of authorization policy, not fixed by the action alone. We argue it therefore requires a benchmark that does not yet exist, one t","url":"https://www.semanticscholar.org/paper/31f4998e2bf81619cb2868927931efc2c5f3672f","categories":["guardrails","access-control","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01169","type":"paper","title":"Safety Does Not Compose: Non-Decaying Loop State for Autonomous LLM Agents","authors":["Chenmin Wu","H. Jia","Yang Liu","Ying-Guang Yang","Yu-Han Lin","Chong Zhang","Hao Zheng","Yu-Long Huang","Jian-Sheng Zhang","Yong-Zhi Qi","Shang Luo","Ke Xu","Jifeng Zhu","Bin Chong"],"year":2026,"abstract":"Large language model agents are increasingly deployed as autonomous loops. Starting from one human goal, such a system repeatedly discovers work, plans, executes tool calls, verifies outcomes and persists state across many unattended iterations. The agent safeguards in wide use, however, are defined over a single trajectory, and their safety state is re-initialized when the next trajectory begins. We show that this is a failure of composition rather than an implementation detail. Our central res","url":"https://www.semanticscholar.org/paper/106e267e5a85d840988eff8d6069a58ac4b313b5","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-01170","type":"paper","title":"A Single Suffix to Break Them All: Basin-Aware Jailbreaks for Merged Model Families","authors":["Yu Zhe","Yiting Tan","Jun-Hao Wei","Wan Chen"],"year":2026,"abstract":"Model merging enables combining multiple fine-tuned models without additional training, but its safety implications remain poorly understood. Prior work primarily attributes merging risks to unsafe constituent models, implicitly assuming that merging individually aligned models preserves safety. In contrast, we show that model merging reveals a previously overlooked jailbreak risk rooted in the pretrained foundation model, even when all constituent models are individually safety-aligned. Motivat","url":"https://www.semanticscholar.org/paper/0f2d5b2bfa30e140a8aa2849861f3bcbca2bd73e","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01171","type":"paper","title":"JudgeStealer: Extracting LLM Judging Capabilities across Evaluation Protocols","authors":["Chen Chen","Yao-Lin Chen","Xue-Han Sun","Juan Lin","Xueluan Gong","Yu-Heng Zheng","Qian Wang","Kwok-Yan Lam"],"year":2026,"abstract":"Large language model (LLM) judges are increasingly used across various evaluation scenarios, making their judgment capabilities valuable intellectual property. However, black-box access exposes these capabilities to model extraction attacks. Existing extraction methods do not specifically target LLM judges and provide limited support for multiple evaluation protocols under restricted query budgets. In this study, we propose JUDGESTEALER, the first query-efficient model extraction framework for r","url":"https://www.semanticscholar.org/paper/0becf5302b4eabed685d226ee6ffbf1f3a06764e","categories":["model-extraction"],"reviewed":false},{"id":"llmsec-2026-01172","type":"paper","title":"LongGuard: Mechanistic Analysis and Training-Free Mitigation of Long-Context Failure in Safety Guardrails","authors":["Ziyang Chen","Xing Wu","Songlin Hu"],"year":2026,"abstract":"Safety guardrails serve as the last line of defense against harmful inputs and outputs of large language models (LLMs), yet they are trained and evaluated almost exclusively on short text. We present LongGuard, a framework that evaluates, mechanistically analyzes, and mitigates long-context guardrail failure. We formulate the task as Safety Needle-in-a-Haystack (SafetyNIAH) over a 0.25k-32k length grid; across 15 mainstream guardrails, unsafe recall drops monotonically by more than 50% on averag","url":"https://www.semanticscholar.org/paper/03d27a4020eb7f788ff7f11c1a73615d5ad275c2","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01173","type":"paper","title":"Vulnerable Code Search: Transferable Attack for Code Language Models","authors":["Kaicheng Wang","Liyan Huang","Jesse Thomason","Weihang Wang"],"year":2026,"abstract":"Reliable code retrieval is crucial for developer productivity and effective code reuse. However, current neural code language models (CLMs) powering search tools are susceptible to adversarial attacks targeting non-functional textual elements. In this paper, we introduce a programming language-agnostic, transferable, adversarial attack that exploits this CLM vulnerability. Our approach perturbs identifiers within a code snippet without altering the snippet's functionality to artificially align t","url":"https://www.semanticscholar.org/paper/fca4f64d2721d15004f8e9081a7b5dbb5e38620d","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01174","type":"paper","title":"EVOMAL: Self-Poisoning in Self-Evolving Coding Agents","authors":["Xiaodong Wu","Yu Shi","Qi Li","Zhimin Zhao","Xiang-Man Li","Bram Adams","Ahmed E. Hassan","Jianbing Ni"],"year":2026,"abstract":"Self-evolving LLM coding agents write their own tools by imitating retrieved skills from shared skill libraries. We identify a vulnerability in this loop: during authoring, a retrieved malicious skill can become the template for a new skill that preserves the payload. We call this self-poisoning: the agent authors, stores, and runs the resulting malicious skill. We exploit it through EvoMal, an attack that amplifies self-poisoning by wrapping an interchangeable payload in a banner, a set of beni","url":"https://www.semanticscholar.org/paper/dade8d85d8cec69350866993be15d982dda27435","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01175","type":"paper","title":"A Survey of Zero-Shot Sensitive Information Detection Techniques based on Large Language Models","authors":["Jie-Qun Wei","Yuejin Zhang"],"year":2026,"venue":"Scientific Journal of Intelligent Systems Research","abstract":"With the rapid growth of digital information, the risk of sensitive information leakage in textual data, including personally identifiable information, medical privacy, financial data, and corporate confidential information, has become increasingly prominent. Traditional sensitive information detection methods, which mainly rely on rule matching, supervised learning, and manual annotation, struggle to meet the requirements of identifying diverse, open-domain, and dynamically evolving sensitive i","url":"https://www.semanticscholar.org/paper/a99698990062ff0248f8281a1c8f3f064b1c96bf","categories":["membership-inference","survey"],"reviewed":false},{"id":"llmsec-2026-01176","type":"paper","title":"AI Slop and Hallucinations in Vulnerability Assessment: A Survey on Reasoning Failures and Trustworthy Mitigation","authors":["Junchen Ding","Jialiang Dong","Yi Zhu","Yi Liu","Gelei Deng","William Susilo","Simeng Ma","Yuekang Li"],"year":2026,"abstract":"The integration of Large Language Models (LLMs) into cybersecurity has transformed vulnerability assessment, but it has also produced a trustworthiness crisis driven by the unchecked proliferation of\"AI slop.\"These artifacts, hallucinated vulnerabilities, plausible but incorrect patches, and semantically repackaged bug reports, impose a cognitive burden on human triage pipelines that mirrors a denial-of-service attack. This paper surveys the empirical evidence, identifies a unifying mechanism, a","url":"https://www.semanticscholar.org/paper/a4d1c9973548a5f990db11f16b8e5d1070768d62","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01177","type":"paper","title":"Approved Too Late: Verdict Staleness in LLM-Guarded Self-Adaptive Systems","authors":["I. Shraga","Roei Eshel","Lior Gorelik"],"year":2026,"abstract":"A large language model (LLM) guardrail for a self-adaptive system (SAS) may issue an approval that is correct at check time but stale by actuation. This creates an Execute-stage time-of-check to time-of-use (TOCTOU) hazard. We study verdict freshness: whether a guardrail verdict remains valid when used. We distinguish three quantities that answer different questions: all-candidate verdict change under fixed-action replay, oracle-labeled approval expiry on recorded closed-loop trajectories, and j","url":"https://www.semanticscholar.org/paper/a2e0b37b50504ec982323457953492e9a489cd55","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01178","type":"paper","title":"Are LLM-Enhanced GNNs Privacy-Safe?","authors":["Long-Zhu He","Zekun Wen","Chaozhuo Li","Sen Su"],"year":2026,"abstract":"Large language models (LLMs) have recently advanced graph neural networks (GNNs) by enriching node representations with semantic information, giving rise to LLM-enhanced GNNs that achieve substantial performance gains. However, their vulnerability to privacy attacks, in which adversaries infer sensitive information from model outputs, remains largely underexplored. To bridge this gap, we present a systematic evaluation of privacy risks in LLM-enhanced GNNs through a unified framework consisting ","url":"https://www.semanticscholar.org/paper/9f818c0b7d9af6900dfa284aeb0cbb0b244e1fb0","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01179","type":"paper","title":"RTLGuard: A Lightweight Teacher-Student Defense for Poisoned RTL Code Generation Models","authors":["Mahshid Rezakhani","K. Azar","H. Kamali"],"year":2026,"abstract":"The rapid advancement of large language models (LLMs) is driving a shift toward automated register transfer level (RTL) code generation, enabling designers to translate high-level specs. into synthesizable hardware. However, this reliance on pre-trained (3rd-party) fine-tuned models may introduce critical trust issues, as the training data and adaptation process of these models are often opaque. Thus, adversaries (even model providers) may embed hidden backdoor threats during fine-tuning, allowi","url":"https://www.semanticscholar.org/paper/905f1889d0ef658a604096a02d0026afb21ee4f7","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01180","type":"paper","title":"What Guides the Agent? Adjudicating Unauthorized Behavior via Localizing Behavior-Guiding Instructions","authors":["Yichao Gao","Yumo Zhang","Yunhao Yao","Haohua Du","Pu-Han Luo","Ruiqi Li","Zhiqiang Wang"],"year":2026,"abstract":"LLM agents integrated with external resources gain complex task capabilities, yet the unified natural-language context channel makes them vulnerable to injection attacks: untrusted external data may be dynamically parsed as behavior-guiding instructions during LLM inference, thereby subverting the agent's decision. Existing defenses focus on static detection or isolation of malicious content at the input/output level, remains insufficient for detecting such dynamic inducements that arise during ","url":"https://www.semanticscholar.org/paper/db973d82aa1cd4dd5cfc5628be6838c59c4a86a9","categories":["agentic-threats"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01181","type":"paper","title":"Poisoning Agentic Alpha: Adversarial Vulnerabilities Across Roles and Architectures in Multi-Agent Trading Systems","authors":["CheolWon Na","Hao Ni","Lukasz Szpruch","Zhang-Yang Wang","Dhagash Mehta","Saurabh Nagrecha","Alejandro Lopez-Lira","Chanyeol Choi","Yongjae Lee","Jee-Hyong Lee"],"year":2026,"abstract":"LLM-based multi-agent trading systems, in which specialized agents collaborate through structured communication to produce trading decisions, are moving rapidly from research prototypes to live deployments that control real assets. The same inter-agent communication that makes them effective also exposes them: a corrupted signal can propagate to the final decision and translate into realized financial loss. Unlike prior attacks that presume privileged access to system internals, we restrict the ","url":"https://www.semanticscholar.org/paper/788d33b73f8c3b174b9ba58775547c9439f2f81f","categories":["data-poisoning","agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01182","type":"paper","title":"Data security in large language models: risks, defense, and directions","authors":["Kang Chen","Xiu-Ze Zhou","Yuanhui Yu","Y. Lin","Hefeng Chen","Congyu Cai","Li Shen"],"year":2026,"venue":"Journal of King Saud University: Computer and Information Sciences","abstract":"Large Language Models (LLMs), now a foundation in advancing natural language processing, power applications such as text generation, machine translation, and conversational systems. Despite their transformative potential, these models inherently rely on massive amounts of training data, often collected from diverse and uncurated sources, which exposes them to serious data security risks. Harmful or malicious data can compromise model behavior, leading to toxic outputs or hallucinations, while al","url":"https://www.semanticscholar.org/paper/249a53fc8e0e8efa49be0e379ace9460bcd2a0e4","categories":["threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01183","type":"paper","title":"StepGuard: Learning Step-Level Guardrails with Scalable Supervision and Safety-Utility Balancing","authors":["Zhijie Zheng","Yu Li","Chen Qian","Yu-Qian Fu","Yanwei Fu","Lu Sheng","Jing Shao","Dongrui Liu"],"year":2026,"abstract":"LLM-based agents can interact with external environments through tool invocation, but this capability also introduces security risks such as file modification, information leakage, and unauthorized actions. Existing guardrails often evaluate completed trajectories, leaving pre-execution monitoring of step-level actions underexplored. We propose StepGuard, a step-level guard model that can audit completed agent trajectories and check tool actions before they are executed. To train StepGuard, we i","url":"https://www.semanticscholar.org/paper/01d8aca918b46e65aaa866d65f619b80302510c6","categories":["guardrails","monitoring-detection","threat-modeling"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01184","type":"paper","title":"Confidently Wrong, Silently So: Auditing Undetectable Failures of a Deployed On-Device Language Model","authors":["Shashwat Pandey","Satwik Pandey","S. Raghu"],"year":2026,"abstract":"Aligning deployed language models requires knowing when their outputs can be trusted, yet on-device models now ship to hundreds of millions of devices with no server-side moderation, and the configuration developers can actually deploy is rarely audited independently. We present a reproducible reliability audit of the developer-accessible on-device foundation model, framed as an oversight question: can a user or a resource-constrained developer tell when the model is wrong? Red-teaming it on cal","url":"https://www.semanticscholar.org/paper/fd8330fb0908180fc1a75842a6f2c9149730873b","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-01185","type":"paper","title":"PsychJail: Exploring Psychological Jailbreaks via Multi-Turn Persuasion of LLM Policies","authors":["Zeyu Feng","Qingyuan Wu","Yu-Zhe Luo","Hua Cheng"],"year":2026,"abstract":"Large language models (LLMs) are increasingly deployed in education, healthcare, policy advising, and other interactive settings, where users engage them as sustained social interlocutors rather than one-shot query engines. This shift makes jailbreaks a growing safety threat, yet most research emphasizes single-turn prompt optimization or iterative attack refinement, leaving psychologically grounded multi-turn vulnerabilities underexplored. We present PsychJail, a psychology-guided framework for","url":"https://www.semanticscholar.org/paper/c8c62a6283d655086c1b928e4ec722888ff5b8fb","categories":["jailbreaking"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01186","type":"paper","title":"AI-Assisted Extraction of Follow-up Observations from GCN Circulars in Astro-COLIBRI","authors":["F. Schüssler","S. Bisero","M. Cellier","A. Ciric","B. Cornejo","I. Jaroschewski","A. Alkan","W. Kiendr'eb'eogo","A. Saint-Paul"],"year":2026,"abstract":"We present a new Astro-COLIBRI component that converts free-text GCN Circulars into structured, event-linked follow-up records and combines them with structured reports submitted directly by the community. A continuously running Circular listener associates new reports with transient events, applies deterministic pre-analysis, and invokes a schema-constrained large language model extraction step for photometry, contacts, redshifts, and other reported results and metadata. The resulting records a","url":"https://www.semanticscholar.org/paper/99ff499a2c237cde2bfdbca646cc68e2a49c856e","categories":["model-extraction"],"reviewed":false},{"id":"llmsec-2026-01187","type":"paper","title":"AgentFlow: A Flow-Centric Policy Language and Framework for Securing LLM Agent Systems","authors":["B. Shivakumar","Swarn Priya","Peng Gao"],"year":2026,"abstract":"LLM agents increasingly read untrusted content, invoke external tools, access private data, and delegate work to other agents. Harm often arises not from a single unsafe action but from the flow of sensitive data across a sequence of otherwise plausible steps. We present AgentFlow, a flow-centric policy language and runtime enforcement model for specifying where data may travel in agent systems. Policies are defined over labeled runtime edges and constrain which tools may receive sensitive field","url":"https://www.semanticscholar.org/paper/69ab62ec5938e6e397cb83f87e3d3fc02afea3a1","categories":["agentic-threats"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01188","type":"paper","title":"Privacy‐preserving multi‐agent systems for human‐centred inclusive healthcare AI","authors":["Mufti Mahmud","Noushath Shaffi","M. S. Kaiser","M. A. Rahman","M. M. Rahman","Vimbi Viswan","T. Sharmeen","S. Bhattacharya","N. J. Ahuja","A. Ulgen","I. Niazi","Kanad Ray","A. Hussain","David J. Brown","J. Moore"],"year":2026,"venue":"The AI Magazine","abstract":"\n Healthcare environments present uniquely demanding constraints for the deployment of artificial intelligence (AI). Clinical decisions of significant consequence are rarely the product of a single isolated computation; rather, safe and effective patient care relies on the continuous collaboration of distributed entities, including clinicians, hospital networks, laboratory systems, monitoring devices, and patients themselves. As generative models and large language models are increasingly deploy","url":"https://www.semanticscholar.org/paper/685c918353abcabc7296f863a60e124fe26101a6","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01189","type":"paper","title":"A transport-layer cryptographic framework secures inter-agent communication and verdict provenance in multi-agent malware detection pipelines","authors":["Víctor Manuel González-Gorrín","Josep Prieto-Blázquez"],"year":2026,"venue":"Scientific Reports","abstract":"\n \n Multi-agent AI systems have emerged as a promising approach for metamorphic malware detection, combining large language model (LLM) reasoning with specialized static, dynamic, and similarity-analysis tools. The cryptographic security of the supporting infrastructure – inter-agent channels, agent identities, and signed analytic verdicts – has so far received less attention than the detection algorithms themselves. This paper presents the Secure Agent Communication Protocol (SACP), a transport","url":"https://www.semanticscholar.org/paper/32ba3f3e509341196abb0cf0e27ca4f60872f21a","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01190","type":"paper","title":"BanglaVeilGuard: Cross-Script Safety Benchmarking and Lightweight Guardrails for Bangla Large Language Models","authors":["Moavia Hassan","Muhammad Iqbal Hossain"],"year":2026,"abstract":"Bangla large language model (LLM) safety is difficult to evaluate with English-centric or standard-script benchmarks because Bangla users routinely write across scripts, spellings, code-mixed forms, and regional registers. This paper presents BanglaVeilGuard, a compact Bangla-first safety benchmark and lightweight prompt guard for six language forms: standard Bangla, Romanized Bangla, Banglish, code-mixed Bangla--English, noisy Bangla, and dialectal Bangla. The benchmark contains 2,366 quality-f","url":"https://www.semanticscholar.org/paper/fe84bf93a0764b6c4c56c6d121de83e1d10fbc99","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01191","type":"paper","title":"Privacy Preserving Semantic Communications in Wireless Edge Networks with Vision Language Models","authors":["Hao Chang","Ming-Zhe Chen","Qianqian Zhang"],"year":2026,"abstract":"Semantic communication has emerged as a promising paradigm for next-generation wireless systems by transmitting high-level semantic features rather than raw bits. However, collaborative devices and multimodal transmission increase privacy risks because sensitive information may leak through inter-device semantic fusion and cross-modal representations. To address this issue, we propose a privacy-preserving semantic communication framework for wireless edge networks. Leveraging a vision-language m","url":"https://www.semanticscholar.org/paper/ac78104890d618aae3ff2db3509cf2c3ff30949b","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01192","type":"paper","title":"LLM Copilot for Privacy-Respecting Cyber Incident Response","authors":["Pramod Prakash"],"year":2026,"venue":"International Journal of Intelligent Systems and Data Science","abstract":"Security operations centers face challenges in managing complex cyber incidents while protecting sensitive forensic data. This paper introduces a multi-agent LLM framework for automating incident response through collaborative planning, execution, analysis, and reflection. We develop a benchmark comprising 130 subtasks across 12 cyber-range scenarios mapped to NIST Incident Response stages (Detection, Response, Recovery), with ten difficulty levels. The system is evaluated using six LLM variants","url":"https://www.semanticscholar.org/paper/86e9f5c49f74bf0364fb6703f68e151b2cb0b68a","categories":["benchmarks","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01193","type":"paper","title":"No One Model Catches Every Harm: Benchmarking Content Moderation Across Safety Scenarios","authors":["Afshin Orojlooyjadid","Hitesh Laxmichand Patel"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly deployed in real-world applications, yet they remain vulnerable to generating harmful content. From adversarial jailbreaks that bypass safety filters to implicit hate that evades detection, the range of risks these models pose continues to grow. While both specialized content moderators and general-purpose LLMs are being used as safety layers, the question of which model is best suited for which type of harmful content remains unanswered. We present ","url":"https://www.semanticscholar.org/paper/833284cb6c460df1a04c971f7f9be4db98bb0680","categories":["jailbreaking","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01194","type":"paper","title":"Mitigating Database Leakage in RAG Systems with Keyword-Grounded Fact Substitution","authors":["Ziliang Zhang","Yubo Zhu","Wei Tong","Jingyu Hua","Zijian Wang","Yuan Zhang","Sheng Zhong"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) has emerged as a powerful paradigm for combining large language models (LLMs) with external knowledge sources. However, RAG systems remain vulnerable to prompt injection attacks, which may mislead the retriever or generator to expose sensitive database contents. To address this issue, we propose KFS-RAG, a defense that mitigates information leakage by reformulating the retrieved context. Specifically, our method first identifies a small set of influential key","url":"https://www.semanticscholar.org/paper/f9b432133fc98cafe0353fb62319bdf7614e203d","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01195","type":"paper","title":"Vis-Poison: Poisoning Visual Knowledge in Multimodal Retrieval-Augmented Generation","authors":["Ru-Jin Liang","Zhong-Pu Chen","Yuhao Lei","Xin Miao"],"year":2026,"abstract":"While multimodal retrieval-augmented generation (RAG) systems increasingly rely on images as external knowledge sources, the introduction of poisoned visual evidence can severely compromise multimodal large language model (MLLM) generation. Unlike prior attacks that rely on altering textual metadata, we introduce Vis-Poison, a novel visual knowledge poisoning attack where the poisoned image itself is the attacker-controlled payload, without manipulating captions, summaries, metadata, or other as","url":"https://www.semanticscholar.org/paper/cbe9563fd7efd2df8017dd23c0e4a6e31b9fc5da","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01196","type":"paper","title":"Fine-tuning as Jailbreaking: A data-centric red teaming framework via logic injection","authors":["Qiuxiang Li","Ke Xu","Yubin Qu","Jianping Wu"],"year":2026,"venue":"International Conference on Automated Software Engineering","url":"https://www.semanticscholar.org/paper/c5c61310180c086c887e65e3d1bcada3aebe271e","categories":["jailbreaking","red-teaming"],"reviewed":false},{"id":"llmsec-2026-01197","type":"paper","title":"Certified Multi-Turn Robustness for LLM Safety via Compositional Bounds and Safety Persistence","authors":["Yang Liu","Bin Chong","Wenkai Yang","Shuai Zhang","Yan-Cheng Chen","Fei Han","GuoZhen","Cheng Zhang","Huaibing Xie","Changze Lv","Shihan Dou","Pluto Zhou"],"year":2026,"abstract":"Large language models (LLMs) are vulnerable to multi-turn jailbreak attacks that progressively manipulate conversation context. Existing certified robustness methods are limited to single-turn inputs; naive multi-turn composition yields bounds that degrade exponentially in the number of turns. We introduce Multi-Turn Certified Robustness (MTCR), a framework that models conversational safety via State-Adversarial MDPs and defines $k$-turn certified robustness as the worst-case safety probability ","url":"https://www.semanticscholar.org/paper/bffff5b3e618d29747dab5b25337420a0ec519b2","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01198","type":"paper","title":"Anchoring Bias: A Persistent Fairness Backdoor Attack against MLLMs under Continual Learning","authors":["Yuyang Luo","Kai Shu"],"year":2026,"abstract":"Multimodal Large Language Models (MLLMs) are increasingly deployed in high-stakes domains where fairness is a critical safety requirement. In practice, these models are continually updated through continual learning (CL) to adapt to evolving tasks and data distributions. Prior work has shown that backdoor attacks can manipulate MLLM responses through hidden triggers, but naively implanted backdoors degrade as models undergo subsequent updates of CL. Although fairness has emerged as a central con","url":"https://www.semanticscholar.org/paper/8b1455c5a420714262b830d1af702fd2478832c6","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01199","type":"paper","title":"ReFrame: Evidence-Guided Test-Time Safety Alignment in Multimodal Large Language Models","authors":["Wenzheng Jiang","Xuan-Kun Rong","Yuan-Zhao Zhai","Da-Wei Feng","Huaimin Wang"],"year":2026,"abstract":"While multimodal large language models (MLLMs) extend model capabilities beyond text, they also make safety alignment increasingly challenging. Multimodal safety alignment methods must address cross-modal jailbreaks, safety-awareness failures, and over-sensitive refusals. However, existing methods often rely on retraining or internal-state inspection, limiting their applicability to deployed closed-source MLLMs and motivating test-time safety alignment. We analyze this setting and identify two k","url":"https://www.semanticscholar.org/paper/8216bd75e51ed23466c2d4f47ea1dfe2f32e75fd","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01200","type":"paper","title":"Trojaning the Alignment: Stealthy Backdoor Attacks against Graph Foundation Models","authors":["Min Lin","Zhi-Cheng Gao","Yilong Wang","Hanqing Lu","Xiang Zhang","Suhang Wang"],"year":2026,"abstract":"Graph Foundation Models (GFMs) on text-attributed graphs (TAGs) align graph representations with language semantics to support transferable graph learning. Despite these advantages, the backdoor vulnerability of GFMs on TAGs remains insufficiently understood, especially under graph-language alignment, where graph and text representations are trained to constrain each other in a shared semantic space. Existing backdoor attacks mainly target either the graph side or the text side, treating the two","url":"https://www.semanticscholar.org/paper/7c24cd0e1f15364b146df652e89e285cacff0bb7","categories":["data-poisoning","guardrails"],"reviewed":false},{"id":"llmsec-2026-01201","type":"paper","title":"No PUN Intended: Plausible Unknown Names for Person-Centred LLM Evaluation","authors":["Dimitri Staufer","David Hartmann","Ibrahim Baroud"],"year":2026,"abstract":"Person names are widely used as prompt variables in LLM evaluations of factuality, privacy leakage, bias and abstention, but when a name's evidential status is uncontrolled, measurements may conflate memorisation, retrieval, name priors and wrong-person attribution. We operationalise an unknown name as one with plausible First-Last form, no indexed full-name evidence, and no ambiguity signals under a documented validation run, and introduce PUN (Plausible Unknown Names), a protocol for construct","url":"https://www.semanticscholar.org/paper/7a824f284f64b49e74d0eeccea6bf406889eab2f","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01202","type":"paper","title":"Trustworthy RAG: An Evaluation Agent for Detecting Misinformation and Knowledge Poisoning in Generative AI Systems","authors":["Balkrishna Giri","M. Hasan","Jussi Rasku","Muhammad Waseem","Pekka Abrahamsson"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) grounds Large Language Model (LLM) outputs in external knowledge, but RAG systems usually trust whatever they retrieve, creating a Security-Reliability Gap: high semantic relevance does not guarantee factual truth. Adversaries exploit this through knowledge poisoning, inserting malicious documents to cause targeted misinformation. We propose an Evaluation Agent, middleware that combines Natural Language Inference (NLI) factual verification, a five-signal pois","url":"https://www.semanticscholar.org/paper/21c1fac19d0c3a8a1ebbddabcdc5b5109b566700","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01203","type":"paper","title":"TempJail: Temporal Jailbreak Attack against Large Vision-Language Models via Subtitle Scheduling","authors":["Ling Zhou","Yihao Huang","Jinglin Sun","Zhiwen Tian","Yi Zeng","Qi-He Liu","Shi-Jie Zhou"],"year":2026,"abstract":"Large vision-language models (LVLMs) have achieved remarkable progress in video understanding and reasoning. Despite extensive studies on text- and image-based jailbreaks, video jailbreaks against LVLMs remain largely unexplored. Existing video jailbreak methods mainly manipulate textual content embedded in videos, while overlooking how such information is organized over time. Our analysis reveals that jailbreak effectiveness depends not only on the semantics of textual information but also on i","url":"https://www.semanticscholar.org/paper/9256fdf3d5d0f55a454ebf4d96a30155cdf62b38","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01204","type":"paper","title":"Auditing Cross-Lingual Fairness in Language Model Watermarking","authors":["Alexander Nemecek","Osama Zafar","Debargha Ganguly","Vikash Singh","Vipin Chaudhary","Erman Ayday"],"year":2026,"abstract":"Watermarking schemes for large language model output are evaluated almost exclusively on English text using each scheme's detection threshold and a narrow set of quality measurements. Multilingual deployment exposes evaluation-design choices that are inconsequential on English but determine conclusions cross-lingually. We propose an evaluation framework with four components: detection thresholds calibrated empirically per deployment context, a threshold-independent companion measurement that dis","url":"https://www.semanticscholar.org/paper/25ee21ef07b6282273f369076f9000bef569ad0a","categories":["watermarking","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01205","type":"paper","title":"A Formal Framework of Architectural Intent Collapse for Tool-Level Attacks on LLM Agents","authors":["Zhaowen Feng","Zhenhui Liu","Ming-Jun Ma","Dong-Ran Zhuang","Jie Gao"],"year":2026,"venue":"Electronics","abstract":"Tool-level attacks on Large Language Model (LLM) agents—poisoned tool descriptions, prompt injection, and capability misrepresentation—are universally effective, yet no existing defense provides comprehensive protection. We propose Architectural Intent Collapse (AIC), a formal framework capturing the systematic loss of communicative intent when text from heterogeneous sources is flattened into a single context window. Grounded as a novel instantiation of the Confused Deputy Problem, AIC reveals ","url":"https://www.semanticscholar.org/paper/1657992e80b3f19b573d4d6104e792c0d98ea40b","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01206","type":"paper","title":"RefusalGuard-M: a scalable human–machine framework for multi-turn LLM jailbreak evaluation via semantic refusal manifold modeling","authors":["Michael Tchuindjang","Nathan Duran","Phil Legg","Faiza Medjek"],"year":2026,"venue":"Cybersecurity","abstract":"Existing multi-turn jailbreak evaluation methods increasingly rely on large language models (LLMs) as automated judges to reduce the cost and scalability limitations of human assessment. However, recent studies show that LLM-based evaluators can diverge from human judgments under adversarial strategies involving subtle linguistic and semantic variations, raising reliability concerns in safety-critical domains such as cybersecurity. To address this challenge, we propose Refusal Manifold Guard (Re","url":"https://www.semanticscholar.org/paper/da9ee63b62208fdbb2199fda4de57efe4a4a4ee7","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01207","type":"paper","title":"A Conceptual Framework for Multi-Agent AI Quality Control in The Review of Regulated Documents","authors":["Michael Ominyi"],"year":2026,"venue":"INTERNATIONAL JOURNAL OF SOCIAL SCIENCES AND MANAGEMENT RESEARCH","abstract":"Organizations in pharmaceutical, financial, and other regulated sectors face rising pressure to\nreview large volumes of complex documents against strict and frequently evolving standards, while\npreserving the traceability needed to defend decisions to auditors and regulators. Large language\nmodel (LLM) based automation offers clear efficiency gains, but single-agent deployments raise\nwell-documented concerns about hallucination, opacity, and insufficient auditability, particularly\nas regulatory ","url":"https://www.semanticscholar.org/paper/ace5593eeb76ba362a29f5c056b74985212eea5b","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01208","type":"paper","title":"Securing the Prompt Pipeline: A Systematic Review of Defense Mechanisms Against Prompt-Based Attacks in LLM Agents","authors":["Sana Mourad","E. E. Abdallah","Mohammad Ababneh"],"year":2026,"venue":"Electronics","abstract":"Current language model deployments face growing security challenges from prompt-based attacks, including jailbreaks, direct and indirect prompt injection, and instruction hijacking, which often evade traditional rule-based safeguards. As these models are increasingly integrated into agent-based systems, retrieval pipelines, and tool-driven workflows, such attacks exploit their natural language interfaces to bypass safety constraints and manipulate system behavior, in some cases leading to data l","url":"https://www.semanticscholar.org/paper/2a6dee66de468ab63743a038cfc5017017f1af73","categories":["prompt-injection","jailbreaking","agentic-threats","guardrails","survey"],"reviewed":false},{"id":"llmsec-2026-01209","type":"paper","title":"AI safety evaluation in an underrepresented population: real-world performance of clinical decision support and frontier language models on Medicaid patient messaging triage","authors":["Sanjay Basu","Sadiq Y. Patel","Parth Sheth","Bernardo Arevalo","Jeremy Schifberg","John Morgan","Rajaie Batniji"],"year":2026,"venue":"BMC Medical Informatics and Decision Making","abstract":"\n \n \n Studies of artificial intelligence tools used in patient triage have largely involved academic medical center cohorts, scripted patient-actor scenarios, or knowledge benchmarks. Populations that may rely on such tools due to constrained access to in-person care, including Medicaid patients, have been less fully evaluated.\n \n \n \n To compare combinations of safety guardrails added to artificial intelligence tools for triage of patient-initiated text messages in a multi-state Medicaid populat","url":"https://www.semanticscholar.org/paper/fee3e808aed590c7fbe755d7dd4e9d3e5acf427e","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01210","type":"paper","title":"You Are an Expert: RAG Injection and Guided Error Expert Activation for Jailbreaking Large Language Models","authors":["Shun Zhang","Ying Ding","Yanxu Mao"],"year":2026,"venue":"Expert Syst. J. Knowl. Eng.","abstract":"With the rapid development and widespread deployment of large language models (LLMs), the security and robustness of these models have emerged as critical research topics. Among various threats, jailbreak attacks, which aim to circumvent built‐in safety mechanisms, have garnered considerable attention as a key means of breaching model protections. However, existing jailbreak methods still face several limitations, such as excessive reliance on the model's internal capabilities, high attack costs","url":"https://www.semanticscholar.org/paper/f5a6e15f149ffd7682a47d693e15a2e17b86651a","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01211","type":"paper","title":"Responsible Large Language Models in Finance: A Descriptive Bibliometric Overview and Taxonomy of Responsibility","authors":["Chong-Hui Tan","Qinxu Ding"],"year":2026,"venue":"FinTech","abstract":"The rapid adoption of large language models (LLMs) in financial services has generated a growing literature on “responsible AI” in domains such as investment analysis, credit assessment, risk management, compliance, and financial advisory systems. Unlike earlier AI systems, LLMs introduce responsibility challenges that differ in important ways from those addressed by earlier responsible AI frameworks, including hallucinations, prompt injection and manipulation, generative opacity, instruction-fo","url":"https://www.semanticscholar.org/paper/ccc1f69bb0b261c2d590eaff0b8eee2ad42140cb","categories":["prompt-injection","responsible-ai"],"reviewed":false},{"id":"llmsec-2026-01212","type":"paper","title":"SkillWatermark: An Embedded Skill Watermark of Progressive Privacy Inference via Benign Prompts","authors":["Yu Li","Liqi Zhuang","Dong Wei","Jiwen Luo","Hang Zhang","Meng Zhang","Xiaona Li","Weiqing Huang"],"year":2026,"abstract":"Skills for large language model (LLM) agents have been widely deployed across diverse application domains. However, we observe that these skills generate specific traffic patterns during execution. In this paper, we design a pipeline that generates specific traffic patterns by inserting carefully designed skill descriptions, which we term skill watermarks, so that a passive network attacker can establish a covert channel to encode private information within observable traffic across multiple con","url":"https://www.semanticscholar.org/paper/ae12dba64123005c4bae0df7c084a9eadc318a88","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01213","type":"paper","title":"Towards Safer RAG: Only Agents Capable of System 2 Thinking may Access Untrusted Documents","authors":["Mehrdad Ghassabi"],"year":2026,"abstract":"Retrieval-Augmented Generation (RAG) has significantly enhanced the performance of large language models (LLMs), yet these systems remain vulnerable to knowledge-poisoning attacks, in which misinformation in retrieved documents can influence the model's final outputs. Notably, an LLM may correctly detect that a document contains incorrect information while nevertheless being influenced by it. Prior work has addressed this vulnerability through the Cordon Principle, which prevents models responsi","url":"https://www.semanticscholar.org/paper/9ad90be31be191016980abf888e48404a08457d7","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01214","type":"paper","title":"DiSCO: Defending text-to-image generation through distribution-guided contrastive prompt optimization","authors":["Tong Zhang","M. Alfarra","Carlos Hinojosa","Christos Louizos","Bernard Ghanem"],"year":2026,"abstract":"As text-to-image generative models advance, they raise critical safety concerns, particularly the generation of Not-Safe-For-Work (NSFW) content such as violence and nudity, further exacerbated by red-teaming adversarial attacks. Existing defenses predominantly operate under white-box assumptions, relying on text encoder optimization, weight editing, or inference-time intervention, and fundamentally cannot scale to proprietary models. Black-box alternatives based on LLM prompt rewriting offer br","url":"https://www.semanticscholar.org/paper/74f0b8c1be197c2e6a6d54d29c20352d640cb979","categories":["adversarial-examples","red-teaming"],"reviewed":false},{"id":"llmsec-2026-01215","type":"paper","title":"Security of Foundation-Model-Powered Embodied Agents: Attack Surfaces, Attacks, Defenses, and Evaluation","authors":["Jiawei Liu","Jiacheng Guo","Tian Zhang","Yi-Wei Xu","Juan Wang","Jinlin Fan","Bowen Xiao"],"year":2026,"abstract":"Foundation models are increasingly used for perception, reasoning, planning, and action generation in embodied agents, creating security risks that can propagate from digital inputs to physical behavior. Existing surveys often organize threats by mechanisms such as jailbreaks, prompt injection, backdoors, poisoning, or adversarial examples, but these categories do not consistently identify where an adversary first enters the embodied control loop. We present a trust-boundary-centric survey of fo","url":"https://www.semanticscholar.org/paper/06190ffa90147391f987fa830340186f54729f19","categories":["prompt-injection","jailbreaking","data-poisoning","adversarial-examples","survey","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01216","type":"paper","title":"TRACE: Trajectory Aware Reasoning for Multi-Turn Adversarial Conversation Evaluation","authors":["Md Messal Monem Miah","Adrita Anika","Zhiyuan Yu","Ruihong Huang"],"year":2026,"abstract":"Multi-turn jailbreak attacks have emerged as a critical safety threat to LLMs, as harmful objectives are decomposed across a sequence of apparently benign turns to bypass guardrails. Existing defenses lack the reasoning capacity to identify evolving manipulation patterns, often trading helpfulness for safety by over-refusing benign requests related to sensitive topics. We introduce Trace, a multi-turn defense with trajectory-aware structured reasoning. Before generating each response, the model ","url":"https://www.semanticscholar.org/paper/e396c538fa9434cc58689737a23e7b0e25ea2ee2","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01217","type":"paper","title":"Conjunctive Poisoning in AI Supply-Chain Applications","authors":["Nokimul Hasan Arif","Qian Lou","Meng Zheng"],"year":2026,"abstract":"Large Language and Vision-Language Models are increasingly deployed through inference pipelines that include prompt wrappers (e.g., templates and post-processing scripts) and configuration metadata (e.g., JSON/YAML files) that together shape model outputs. While model weights and binaries are routinely verified, these textual deployment artifacts remain weakly protected despite directly influencing runtime behavior. We show that a malicious developer can pair a benign-looking wrapper with crafte","url":"https://www.semanticscholar.org/paper/d9175f51a25b86b6c86fb663397e860b28b80933","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01218","type":"paper","title":"PL-Guard: Probabilistic Logic Reasoning for LLM Guardrails","authors":["Satchit Chatterji","Shi-Han Wang","G. Sileno","Erman Acar"],"year":2026,"abstract":"Large language model guardrails can be viewed as policy-consistency problems: a system must determine which policy-relevant facts hold in a prompt-response pair and what those facts imply under a given policy. Common approaches, including policy prompting and LLM-as-a-judge pipelines, often overlap the tasks of semantic grounding and policy reasoning: the model both interprets the prompt-response pair and reasons about whether a policy has been violated. This can lead to unsafe compliance with h","url":"https://www.semanticscholar.org/paper/b112fd1f674c4d19bde5fca334828d52a986db60","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01219","type":"paper","title":"Assessing Attack Surfaces in Generative Search Engines through Publisher Attributes: A Case Study in Political Domains","authors":["R. Mochizuki","Shusuke Komatsu","Souta Noguchi","Kazuto Ataka"],"year":2026,"abstract":"We characterize the attack surface of generative search engines (GSEs) against poisoning attacks in the political domain, from the perspectives of citation selection and personalization. GSEs integrate web search and answer generation with user preferences and backgrounds using large language models (LLMs). They play a crucial role in how users access information on the web. Because anyone can publish content on the web, GSEs are vulnerable to poisoning attacks that manipulate citations to under","url":"https://www.semanticscholar.org/paper/2ad14c8c5cf1f454c49512afc25a9e55fd2efa88","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01220","type":"paper","title":"Watermarked Game Solving via Perturbed Regret Minimization","authors":["Juho Kim","Tuomas Sandholm"],"year":2026,"abstract":"Many real-world interactions among self-interested parties can be modeled by game theory, and the rapid advancements in AI have raised concerns about the possible misuse---accidental or deliberate---of superhuman or human-level game-playing agents by bad actors. While AI watermarking has mainly been applied to LLM-generated texts, a recent line of work proposes developing watermarking techniques for agents in game-theoretic settings. However, existing watermarking techniques for game-theoretic a","url":"https://www.semanticscholar.org/paper/d375bc8b1fa72311b4984e44877fa27413f61e2a","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01221","type":"paper","title":"BioFirewall: A genome-writing-native governance layer for design-stage biosecurity screening of agentic AI","authors":["Anees Ahmed Mahaboob Ali","R. Delhibabu","Everette Jacob Remington Nelson"],"year":2026,"abstract":"Background. Artificial-intelligence design tools now plan genome-scale edits, and agentic systems execute those plans with progressively less human oversight. Biosecurity controls are limited to two points: refusal guardrails at the foundation model and sequence-identity screening at the synthesiser. The design stage between them, where the plan is specified, remains governed by recommendations rather than any deployed system. Results. We present BioFirewall, a rule-governed middleware that inte","url":"https://www.semanticscholar.org/paper/cf5903b5d8af40fba5e21a2fefdfd4e634dd0f81","categories":["agentic-threats","guardrails","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-01222","type":"paper","title":"DT-GenShield: A Digital Twin-Driven Runtime Security Architecture for Protecting Large Language Models Against Indirect Prompt Injection","authors":["Alaa Alnemari","Mashael M. Alsulami"],"year":2026,"venue":"Electronics","abstract":"Large Language Models (LLMs) are increasingly deployed in security-critical applications but remain vulnerable to indirect prompt injection attacks that cannot be fully addressed by conventional prompt detection techniques. This paper proposes DT-GenShield, a Digital Twin-driven runtime security architecture that integrates semantic threat detection, operational state representation, policy-guided mediation, and runtime logging to protect LLM-based systems before model inference. The proposed ar","url":"https://www.semanticscholar.org/paper/689031f7b55edb5c81db55047a445d9f068415d1","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01223","type":"paper","title":"SysEvolve: An AI-native, safe, autonomous adversarial attack-defense co-evolutionary system","authors":["Yuhan Meng","Shaofei Li","Jionghao Huang","Jiandong Jin","Puyi Wang","Hanlin Jiang","Anis Yusof","Peng Jiang","Zhenkai Liang","Yao Guo","Ding Li"],"year":2026,"abstract":"The rapid advancement of large language models (LLMs) has created a growing asymmetry in cybersecurity, where attack accelerates toward autonomous execution while defense remains predominantly human-intensive. Despite substantial prior work across cyber ranges, AI-driven attack, and AI-driven defense, this asymmetry persists. We trace it to a deeper root cause, that evolution itself has stalled on both sides at three layers. To overcome this, we propose co-evolution as the integrating insight, w","url":"https://www.semanticscholar.org/paper/1af26c98c69b3b702a827439df5413f427ffc8a1","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01224","type":"paper","title":"Optimal Watermark Localization in Mixed-Source Large Language Model Texts","authors":["José H. Blanchet","T. Cai","Xiang Li","Hao Liu","Qi Long","Weijie J. Su"],"year":2026,"abstract":"Watermarking provides a principled way to authenticate text generated by large language models (LLMs). In practice, however, the final text may be mixed-source, with watermark evidence surviving at only a subset of token positions after rewriting, insertion, deletion, or paraphrasing. Although prior work has studied global detection of watermark signals, when such signals can be localized remains unclear. We formulate watermark localization as a token-level multiple-testing problem based on pivo","url":"https://www.semanticscholar.org/paper/9d38112c12c8aed8b21ee18522eed2e3b185a8ac","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01225","type":"paper","title":"Tripwire: Triggering Aligned Refusal via Statistically Certified Safety Neurons","authors":["Wei Zhao","Zhe Li","Peixin Zhang","Jun Sun"],"year":2026,"abstract":"Neuron- and path-level interventions offer the finest-grained route to defending large language models (LLMs) against jailbreak attacks, yet existing methods fall short of this promise, i.e., they often compromise model utility significantly. Specifically, one line of work suppresses toxic neurons to erase harmful semantics, but since such semantics are distributed across the network, blocking every pathway forces a large intervention footprint. An alternative line of research focus on identify ","url":"https://www.semanticscholar.org/paper/49c4e4dfde4fe8e1786ae73918c43796b07e4582","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01226","type":"paper","title":"BiasMix-Finance: Post-Generation KYC Guardrails for LLM Portfolio Advice","authors":["Gaurav Kukreja","Parul Kukreja","Mohammed Abraar","R. Dandekar","R. Dandekar","Sreedath Panat"],"year":2026,"abstract":"Large language models (LLMs) can generate plausible-sounding ETF portfolios while silently violating basic KYC-style constraints on risk, fees, and diversification. This is especially problematic in agentic multi-turn advisory systems, where each draft recommendation can become an action unless guarded by an auditable enforcement layer. We study a model-agnostic, asset-agnostic post-generation guardrail pipeline: (i) enforce a strict JSON allocation schema, (ii) validate allocations against nume","url":"https://www.semanticscholar.org/paper/35614c01d076e0ee928e20ec6fa9652abed6457b","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-01227","type":"paper","title":"Language-Specific Gaps in AI Safety Training Datasets","authors":["Chialuka Prisca-Mary Onuoha","Bright Etornam Sunu","Rashidat Sikiru"],"year":2026,"abstract":"Large language model providers routinely cite multilingual safety benchmarks spanning a dozen or more languages as evidence that their models are safe for non-English-speaking users. We show that these collection-level coverage claims frequently do not survive inspection at the level of an individual language. Auditing 21 resources across 25 language slices, of which 20 count as datasets under our counting rules, spanning three languages chosen to represent low- (Hausa), mid- (Swahili), and high","url":"https://www.semanticscholar.org/paper/e49b7db925cfbc19b9d81d2927d28c9670abb89b","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01228","type":"paper","title":"HiRoute: Hierarchical Routed Prompt Tuning for Safety Alignment of Large Language Models","authors":["Fang-Zhou Chen","Shiji Zhao","Mengyan Wang","Qihui Zhu","Ran-Jie Duan","Maoxun Yuan","Xingxing Wei"],"year":2026,"abstract":"Large language models (LLMs) remain vulnerable to harmful requests and jailbreak attacks. Parameter-efficient safety alignment methods based on prompt tuning typically rely on a single global prompt or externally selected prompt modules. Such static designs struggle to maintain a cross-category safety boundary while generating constructive responses tailored to specific risks and avoiding over-refusal of benign inputs. To address these limitations, we propose HiRoute, an input-adaptive hierarchi","url":"https://www.semanticscholar.org/paper/d697618a11df5206c8d73ecedf55ddf9db224a37","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01229","type":"paper","title":"Tracing Provenance and Detecting Tampering with Complementary LLM Watermarks","authors":["Xiao-Yang Feng","Yanjun Zhang","He Zhang","L. Zhang","Shirui Pan"],"year":2026,"abstract":"Watermarking LLM-generated text is an important task for tracing its provenance. Existing LLM watermarks preserve provenance under editing, but this same robustness allows an adversary to alter critical content while retaining attribution, a vulnerability known as piggyback spoofing. We introduce an innovative watermark that jointly provides provenance and tamper evidence. It co-embeds a robust signal and a fragile signal into each generated token. The signals share the same mechanism but use in","url":"https://www.semanticscholar.org/paper/ab116a93501cd418ff258dda5081b8daf086b858","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01230","type":"paper","title":"Beyond Visual Evidence: Revealing and Mitigating Relational Privacy Leakage in Document MLLMs","authors":["Beining Xu","Hairui Wang","Jiaxin Wang","Changsheng Chen","Anirban Chakraborty"],"year":2026,"abstract":"While the privacy risks of multimodal large language models (MLLMs) have drawn significant attention, the unique vulnerabilities of domain-specific MLLMs remain largely underexplored. Focusing on document understanding MLLMs for identity document processing, this paper investigates the privacy issues inherent in Key Information Extraction (KIE) tasks. We reveal that when input images lack sufficient visual evidence, these models often rely on memorized field relations from training data to infer","url":"https://www.semanticscholar.org/paper/8b616f7b4ce3fe4934e0fd0beb28e9f6ca206b71","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01231","type":"paper","title":"Reconcile Once, Write Anytime: A Trust-Tiered Librarian and a Multi-Agent Writer for Drift-Free, Point-in-Time Research","authors":["Xing Zhang","Ya Cui","Guanghui Wang","Pei-Gen He"],"year":2026,"abstract":"Long-form research reports generated by large language models drift, contradict themselves, and lose provenance: the same metric appears with different values, and rumor is quoted as confidently as an audited filing. We present a two-tier agentic system that separates a maintained, point-in-time knowledge library from report writing. A deterministic\"librarian\"ingests timestamped sources into a trust-tiered ontology, layering evidence cards, an authoritative metric ledger, and a claim graph into ","url":"https://www.semanticscholar.org/paper/6eb07fc891829f8610b547a2b0c22a2d95fceda5","categories":["agentic-threats","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01232","type":"paper","title":"SCENARIO-BASED PREHOSPITAL TRIAGE OF CARBON MONOXIDE POISONING: COMPARING FIVE LARGE LANGUAGE MODELS WITH EMERGENCY MEDICAL SERVICES PERSONNEL IN A PROSPECTIVE STUDY","authors":["Vildan Özer","Özlem Bülbül","Efnan Bayrak Erbolukbas","Serdar Karakullukçu","Aynur Şahin"],"year":2026,"venue":"Kırıkkale Üniversitesi Tıp Fakültesi Dergisi","abstract":"Objective: Carbon monoxide (CO) poisoning requires rapid identification and timely decisions regarding the need for hyperbaric oxygen therapy (HBOT) to improve clinical outcomes. This study aimed to compare the decision-making performance of emergency medical services (EMS) personnel and large language models (LLMs) in accurately determining the need for HBOT during the prehospital phase of CO poisoning cases, and to explore the potential implementation of LLMs as decision-support tools for pati","url":"https://www.semanticscholar.org/paper/ed91854853fa63aa344cdee6370d9016e8d2c027","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01233","type":"paper","title":"Towards Automated Domain Model Extraction from Source Code using Heuristics and Open-Source LLMs","authors":["Alessandra Mancas","Mounir Ammam","Hyacinth Ali","Kevin Delcourt","H. Sahraoui"],"year":2026,"abstract":"Large language models (LLMs) have recently shown strong capabilities for code understanding, making them promising for reverse engineering domain models from source code. However, state-ofthe- art proprietary LLMs cannot be used in many industrial contexts due to privacy and confidentiality constraints, while compact open-source LLMs that can run locally are limited by their context window and cannot process large code bases directly. In this paper, we propose an automated approach to extract do","url":"https://www.semanticscholar.org/paper/440cae9d353d5f570773f5158d771cb6d1266cf2","categories":["model-extraction"],"reviewed":false},{"id":"llmsec-2026-01234","type":"paper","title":"CAPS: Compositional Attack Path Scoring for LLM Deployment Stacks","authors":["Quang-Vinh Dang","Hoang-Viet Vu","Ngoc-Son-An Nguyen","M. Dinh","Dat Le"],"year":2026,"venue":"Artificial Intelligence and Applications","abstract":"Evaluating the security posture of large language model (LLM) deployment stacks is a critical challenge in modern AI security. Traditional vulnerability management frameworks—such as the Common Vulnerability Scoring System (CVSS) and component-level checklists—assume that software components can be evaluated in isolation. In real-world agentic and retrieval-augmented generation (RAG)-based LLM ecosystems, this assumption is systematically violated: attackers exploit complex topologies, chaining ","url":"https://www.semanticscholar.org/paper/33eca5ed0e766089b5c28a42428b804e6a8e2bf2","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01235","type":"paper","title":"Defensive Reverse Engineering of LLM Applications: A Black-Box Framework for Security Risk Scoring and Mitigation","authors":["Bhavesh B. Prajapati","Bhavya Shah"],"year":2026,"venue":"International journal of computer information systems and industrial management applications","abstract":"Large language model (LLM) applications now combine hidden prompts, retrieval pipelines, memory stores, content filters, tool calls, delegated identities, and downstream automation. Security reviewers are increasingly asked to assess such systems without access to source code, model weights, prompt templates, vector-store configuration, or internal logs. This paper presents D-RELLM, a defensive reverse-engineering framework for black-box security assessment of deployed LLM applications. The fram","url":"https://www.semanticscholar.org/paper/28983a3205a8f92c62682310e831f05d4eae8fab","categories":["output-moderation","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01236","type":"paper","title":"ProbGuard: Calibrated Safety Risk Estimation from LLM Output Distributions","authors":["Xinzhe Huang","Biwu Yao","Kedong Xiu","Mengnan Zhao","Di Wang","Puning Zhao","Tianhang Zheng"],"year":2026,"abstract":"Recent research on Large Language Model (LLM) safety has widely adopted guardrails to identify unsafe LLM outputs. Existing guardrails typically formulate safety assessment as a deterministic classification task, mapping a discrete token sequence to a discrete safety label. However, this paradigm has two limitations: First, safety assessment is inherently an uncertain problem, particularly during the early generation state. Second, relying solely on discrete token sequences discards the rich pro","url":"https://www.semanticscholar.org/paper/f6c0d06532a9084259bdd4e94343c9a1e98e4f23","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01237","type":"paper","title":"SafeCap: Improving LVLM Safety with Image Captioning Reinforcement Learning","authors":["Caoyuan Ma","Wen-Pu Liu","Weichu Xie","Tian Gu","Shilei Zhao","Lingxi Min","Shuai Dong","Yu-Qi Xu","Ji Zhao","Ziyue Wang","Wenzheng Chang","Taiqiang Wu","Yong-Fu Zhu","Wen-Qi Shao","Yinqiang Zheng"],"year":2026,"abstract":"Large vision-language models (LVLMs) remain vulnerable to jailbreak attacks that exploit visual inputs to bypass safety alignment inherited from their language backbones. We propose SafeCap, a reinforcement-learning framework that aligns LVLMs through learned self-captioning. SafeCap trains a policy model to first generate a safety-relevant image caption and then produce a final answer; the caption is further optimized by whether it enables a frozen LLM to reach a safety-aligned decision. This c","url":"https://www.semanticscholar.org/paper/e4ddb4f43295dae00e3e9f478c20020294f9157c","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01238","type":"paper","title":"AI Guardrail Survival under Single-Cycle Agentic Self-Summarization","authors":["Ted Kwartler","Alan Aqrawi","Arian Abbasi"],"year":2026,"abstract":"Long-running agents periodically compact their context, replacing the transcript with a model-generated summary. Recent work shows that dropping a standing safety constraint during compaction drives behavioral violations across many models (Governance Decay; Chen, 2026). We ask a finer question: under a single compaction cycle, how is a safety rule lost, and what does that imply for detection and evaluation? Our central finding is that a presence check is not a safety check: when compaction does","url":"https://www.semanticscholar.org/paper/d818405f19246444333379662165fb5ee43ebc1e","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-01239","type":"paper","title":"Once Poisoned, Arbitrarily Controlled: A Programmable Backdoor in VLMs","authors":["Tao Lin","Gaojie Jin","Zongxi Liu","Peng Wu","Lijia Yu"],"year":2026,"abstract":"Existing vision-language model (VLM) backdoors are usually treated as static vulnerabilities: one-to-one and N-to-N attacks bind one or more triggers to a finite set of targets before victim training. This assumption substantially underestimates the threat. We show that a single poisoning phase can implant a programmable backdoor into a VLM, allowing an attacker to choose previously unseen target-caption semantics at inference time and synthesize corresponding stealthy triggers on demand. Unlike","url":"https://www.semanticscholar.org/paper/d72f36000b19aeeb672f3fc41a15e33915bc8ca3","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01240","type":"paper","title":"Backdoor Decontamination Dynamics in LLM Agents","authors":["Gabriel Huang","Abhay Puri","L'eo Boisvert","Alexandre Drouin","Perouz Taslakian","Spandana Gella","Christopher Pal"],"year":2026,"abstract":"Open-weight LLM agents are vulnerable to backdoors installed during fine-tuning, which may be undetectable if the trigger conditions are never met during testing. Assuming defenders do not know the existing trigger, they cannot unlearn it directly. One decontamination strategy is to install a known backdoor (defensive poisoning) then to unlearn it, hoping that the original unknown backdoor is removed as a side effect. However, this procedure has uncertain outcomes: the original backdoor may pers","url":"https://www.semanticscholar.org/paper/d5229f30a2f7494f971c21683cde87e9e29b1cbf","categories":["data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01241","type":"paper","title":"Measuring Semantic Abstractness of SAE Features via Nonlocality","authors":["Chuqi Lin","S. Sondhi","Xiao-Liang Qi"],"year":2026,"abstract":"Sparse autoencoders (SAEs) have helped uncover mechanistic explanations for LLM behaviours such as reasoning, jailbreaking etc., via understanding the corresponding task-relevant and causally effective features. To evaluate such mechanistic explanations, downstream studies must distinguish surface lexical features from genuinely high-level ones. However, neither an autointerp-based semantic description nor causal steering utility fully resolves the abstraction level of a feature. To this end, we","url":"https://www.semanticscholar.org/paper/736ae9f2e1f4b71b506575926c7799e9ca2901df","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01242","type":"paper","title":"SecureMCP: Policy-Enforced Defense Against Prompt Injection in LLM-Generated SQL for AIoT Databases","authors":["Wonbae Kim","Hee-Kyong Yoo","Nammee Moon"],"year":2026,"venue":"Applied Sciences","abstract":"The deployment of Large Language Model (LLM)-generated SQL in Artificial Intelligence of Things (AIoT) systems introduces critical security risks, as prompt injection attacks can manipulate LLMs into producing unauthorized queries that expose sensitive data or execute destructive operations. Existing Natural Language to SQL (NL2SQL) research targets query accuracy, while current Model Context Protocol (MCP) servers offer only SQL-level protection without fine-grained, role-based access control. ","url":"https://www.semanticscholar.org/paper/39676372aa39d51366b632f3d18e809c3b05b77a","categories":["prompt-injection","access-control","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01243","type":"paper","title":"An Empirical Study of Output-to-Input Loops for Black-Box Backdoor Detection in Fine-Tuned Open-Weight LLMs","authors":["Md Nahid Hasan","Mohammad Arif Hossain"],"year":2026,"abstract":"Anyone can upload a fine-tuned large language model (LLM) to a public repository and claim it is safe. A backdoored model behaves normally on ordinary inputs until a hidden trigger fires, and a user with no training data, clean reference weights, or the trigger phrase has no clear way to check the model before using it. We introduce and empirically evaluate self-feeding, a black-box test method that feeds a model's own output back as its next input, so the text drifts away from the starting prom","url":"https://www.semanticscholar.org/paper/332aa7504f08004ae724a19f94043c1278e21ce4","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01244","type":"paper","title":"A Privacy-Preserving Middleware Architecture for Detecting Prompt Injection and Sensitive Data Exposure in Large-Language-Model Interactions","authors":["Adam Ait Hsine","A. Arabo"],"year":2026,"venue":"Electronics","abstract":"The deployment of large language models (LLMs) in real-world applications introduces a compounding security problem: detecting adversarial inputs such as prompt injection and jailbreak-driven data leakage while simultaneously preventing the detection mechanism itself from becoming a source of data exposure. Existing approaches address either detection effectiveness or privacy preservation, but rarely both in a unified, deployable architecture. This paper proposes and evaluates a privacy-preservi","url":"https://www.semanticscholar.org/paper/0946d10598a3836423ad37115c73c58a45a9874d","categories":["prompt-injection","jailbreaking","membership-inference"],"reviewed":false},{"id":"llmsec-2026-01245","type":"paper","title":"MetaStrategy: Generative Ranking with Executable LLM Strategies","authors":["Chengyu Lai","Jiuning Lin","Zhibo Xiao","Xiaodong Zhu","Ruiquan Lan","Bin Zhang","Zi-Hong Huang","Wendong Zhang","Chuxin Chen","Yinjiang Cai","Shuaihao Zhong","Lingqin Zhang","Di-Min Wang","Jialin Zhu","Hanqian Zhu"],"year":2026,"abstract":"Industrial recommender systems rank heterogeneous content under coupled user, business, commercial, and experience objectives. Existing generative ranking methods typically construct item sequences directly, making them difficult to integrate with mature predictive models, operational rules, and field-level guardrails. We present MetaStrategy, a framework that instead generates a structured, executable ranking strategy. Conditioned on request context, a large language model (LLM) policy emits a ","url":"https://www.semanticscholar.org/paper/eefb388ef446ab525230570a9be8cc931e0c2b34","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01246","type":"paper","title":"Risk perception and generative AI discontinuance intention: mediating effect of cognitive fatigue and moderating effect of AI dependence","authors":["Wen-Wen Zhao","Jia-Wen Liu"],"year":2026,"venue":"Frontiers in Psychology","abstract":"With the rapid popularization of generative artificial intelligence (AI), the risks of misinformation dissemination and user privacy leakage have become key factors influencing users' decisions on the continuous use of AI. Most existing studies focus on the driving mechanisms of AI adoption, while there is insufficient exploration of the reasons for negative usage outcomes, especially the reasons for users to discontinue using the technology. Based on the cognitive-affective-behavioral (C-A-B) f","url":"https://www.semanticscholar.org/paper/e1e0dc423f0130f12ba20e9becaa81d15a26892a","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01247","type":"paper","title":"Decoding-Level Taboo: A Diagnostic Stress Test for LLM Robustness","authors":["T. Kamijo","Ori Rottenstreich","Javier Conde","Gonzalo Martínez","Pedro Reviriego"],"year":2026,"abstract":"Large language model evaluations typically focus on performance under nominal conditions, creating an illusion of capability where models comfortably walk a narrow, highly optimized generation corridor. In real-world deployments, however, complex system prompts, safety guardrails, and structural constraints continuously force models off this nominal path, driving a divergence between benchmark scores and deployment performance. To address this issue, we introduce Decoding-Level Taboo, a zero-pro","url":"https://www.semanticscholar.org/paper/64c109cad907daf65636234e2eed04e0fba80d70","categories":["guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01248","type":"paper","title":"Agentic Harnesses: LLM-Driven Verification Layers for Robot Autonomy","authors":["Rohan Bhagra","Mahantesh Halapannavar","Uddhav Bhattarai"],"year":2026,"abstract":"Advances in advanced artificial intelligence tools have sparked research in robot autonomy, but the development of such systems has largely focused on execution rather than verifying the feasibility actions planning models propose. Like general-purpose LLMs, robotics planning models carry risks: biased toward user-specified goals, they may suggest actions misaligned with scientific ethics, they may be unsafe due to an inability to\"remember\"prior safety risks, or they may be vulnerable to adversa","url":"https://www.semanticscholar.org/paper/5f39fb04dbe12b5f64ec64252d2c570e3fb9fe0d","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01249","type":"paper","title":"ElasticBack: Stealthy Conditional Backdoor in LLM-Agent Skills via Coupled Trigger-Rule Optimization","authors":["Hao Sui","Simeng Qin","Jie Liao","Xiao-Jun Jia","Bing Chen","Yang Liu"],"year":2026,"abstract":"Agent skills, bundles of instructions and resources that an LLM agent loads on demand, form an emerging supply chain where a single poisoned skill can persistently compromise every agent that installs it. However, existing skill attacks either fire on every request or rely on fine-tuned weights or multiple skills, leaving a conditional and low-cost backdoor unexplored. In this work, we present ElasticBack, an effective conditional single-skill backdoor that plants a rule R in the skill document ","url":"https://www.semanticscholar.org/paper/406fad7e7519f35b9fae4ce4268a40833cf9e403","categories":["data-poisoning","supply-chain-attacks","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01250","type":"paper","title":"The CASE Framework: A Multi-Disciplinary Control Architecture for Governing Enterprise Agentic AI","authors":["Srinivas Telukunta","Georgios Nektarios Lilis","Lucio Baron"],"year":2026,"abstract":"Enterprises are deploying autonomous AI agents faster than they can govern them, and prevailing approaches stretch a single discipline, typically DevSecOps built for deterministic automation, across every scale of agency. We argue that agentic AI governance is four problems, not one, each with a mature governing science. The CASE framework assigns Control theory to the individual agent (intent as setpoint, guardrails as feedback, evaluation as observation), complex Adaptive systems theory to age","url":"https://www.semanticscholar.org/paper/14d4b85624bbc1096651c3256dd4658507810463","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-01251","type":"paper","title":"Same Question, Different Answer? Measuring and Mitigating Prompt Privilege for Equitable AI Access","authors":["Li Jin","Lang-Xiang Hu","Bin-Qi Shen","Han-Yu Cai","Yu-Ting Xin"],"year":2026,"abstract":"Large language models (LLMs) are increasingly integrated into healthcare, education, public services, and everyday decision making. They should provide comparable assistance regardless of a user's literacy, communication style, or prompt-engineering expertise. However, existing research on prompt robustness primarily focuses on adversarial attacks, prompt injection, and prompt optimization, while overlooking whether semantically equivalent requests receive different responses simply because they","url":"https://www.semanticscholar.org/paper/fe4a2c7e3cb87a251832379d2d8903fdaa601d2a","categories":["prompt-injection","adversarial-examples"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01252","type":"paper","title":"Yesterday's Shield, Today's Spear: A Self-Evolving Safety Guardrail in Production","authors":["Ming Cong","Jingyin Chen","Bin Liu","Qi Chu","Tao Gong","Neng-Hai Yu","Ying-Fei Xiang"],"year":2026,"abstract":"Deployed LLM safety guardrails are predominantly static: trained once and frozen at release, while new jailbreak techniques and previously un-addressed harmful categories emerge within days, leaving the defense perpetually a step behind. We present SESG (Self-Evolving Safety Guardrails), a multi-agent system running in production. SESG monitors the live traffic behind a deployed guardrail and surfaces two classes of failure: jailbreaks novel in form and harmful categories novel in content. Once ","url":"https://www.semanticscholar.org/paper/ab64ee559216abd0d808ffb581aaba43b7d4825c","categories":["jailbreaking","guardrails","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01253","type":"paper","title":"HoloAegis: Frozen Representation, Topological Inference: Minimally Parametric Safety Manifolds for Zero-Shot LLM Guardrails","authors":["T. Li","Kaijie Liu","Lik-Hang Lee","King-Chung Ho","Ping Shum","Michael K. Ng"],"year":2026,"abstract":"Current LLM safety guardrails face a fundamental tension: fine-tuning distorts pre-trained representations while generative judges incur prohibitive inference costs. We challenge the prevailing paradigm by asking: can safety be achieved through pure geometric reasoning over frozen semantic representations? We present HoloAegis, a minimally parametric topological inference framework that decouples representation from reasoning. We term our approach minimally parametric because the only free param","url":"https://www.semanticscholar.org/paper/8d4c68176fa54a5bf07893d8029fd1315f37edf9","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01254","type":"paper","title":"When Skills Meet Safety: Benchmarking and Characterizing the Adaptive Jailbreak Robustness of Skill-Merged LLMs","authors":["Yu Ma","Hongli Shi","Jing Li","Xinran Xu","Weiwei Hou"],"year":2026,"abstract":"Model merging has become the default way to give an aligned language model new skills without retraining: a practitioner folds task vectors from math, code, or domain specialists into a safety-aligned base using task arithmetic, TIES, or DARE. This convenience is known to carry a safety cost, but almost all of that evidence rests on static refusal tests: fixed harmful prompts scored for compliance. We argue this is misleading. Because safety alignment is\"shallow,\"concentrated in the first few ge","url":"https://www.semanticscholar.org/paper/4ef3b426aaf844b65a8b487baa1a87b1aec4c55e","categories":["jailbreaking","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01255","type":"paper","title":"Real-Time Detection and Mitigation of Prompt Injection Attacks in LLM-Integrated Enterprise Systems","authors":["Fatimah Alhamzawi"],"year":2026,"venue":"Al-Noor Journal of Engineering Management and Computer Science","abstract":"Large language models (LLMs) embedded in enterprise workflows cannot structurally distinguish legitimate instructions from adversarial ones in the same token stream, making prompt injection OWASP's top LLM risk for two consecutive editions a persistent threat across direct and indirect vectors. This paper presents PromptShield-RT, a layered, real-time, model-agnostic framework combining input normalization and provenance tagging, lexical-heuristic pattern matching, a statistical classifier, stru","url":"https://www.semanticscholar.org/paper/2c13db4c47d1eb3e3184dffedb8ff84130ba8297","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01256","type":"paper","title":"Distilling Vision-Language Models for Robust Traffic Sign Perception in Autonomous Vehicles","authors":["Pedram MohajerAnsari","Amir Salarpour","Mert D. Pesé"],"year":2026,"abstract":"Traffic sign recognition (TSR) models based on deep neural networks achieve strong clean-data performance but remain vulnerable to physically realizable adversarial attacks, including shadow perturbations, natural-light interference, and printed patches. Existing defenses often improve robustness against one attack type while degrading performance on others, and can reduce clean accuracy. We propose LAMDA (Language-Anchored Model for Direction Alignment), a training framework that transfers lang","url":"https://www.semanticscholar.org/paper/2b667cc2121abfa5bb11c1d1af9c50046c5bc69f","categories":["adversarial-examples","guardrails"],"reviewed":false},{"id":"llmsec-2026-01257","type":"paper","title":"Efficient and Differentially Private Federated LLM Fine-Tuning on Heterogeneous Clients","authors":["Nan Yan","Yu-Qing Li","Xiong Wang","Jing Chen","Wei Wang","Kun He","Ruiying Du","Shuhua Li"],"year":2026,"venue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2","abstract":"Federated low-rank adaptation (FedLoRA) allows multiple clients to collaboratively fine-tune large language models (LLMs) on downstream tasks without exposing their private data. To mitigate privacy leakage during aggregation, differential privacy (DP) is widely used to clip and perturb local model updates with noise, yet it can compromise model accuracy due to the inherent privacy-utility trade-off. The performance degradation becomes worse under the FedLoRA setting with the amplified DP noise ","url":"https://www.semanticscholar.org/paper/e07ea3aaf72821d49acb1c5174f80cabeb43b1c1","categories":["membership-inference","differential-privacy"],"reviewed":false},{"id":"llmsec-2026-01258","type":"paper","title":"SimuGov: A Simulation Optimization Framework for Generative AI Governance Strategy Design","authors":["Bingxue Zhang","Jinbiao Li","Q. Tang","Feida Zhu"],"year":2026,"venue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2","abstract":"This study addresses a concrete challenge in Generative AI governance through the lens of AI watermarking: how to evaluate and optimize governance strategies before deployment in a bounded socio-technical setting. To this end, we propose SimuGov, a simulation framework for governance strategy optimization. This framework incorporates psychological traits and adversarial environment perception into agent modeling, enabling behaviorally rich simulation of stakeholder responses within the AI waterm","url":"https://www.semanticscholar.org/paper/bc0b9d2566ae15ad56fe3cb6824594d0646d44ce","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01259","type":"paper","title":"Query-Only Backdoor Attacks on Self-Evolving Skills via Trajectory Poisoning","authors":["Yuyang Luo","Haoran Wang","Kai Shu"],"year":2026,"abstract":"Agentic skills improve large language model (LLM) agents by encoding reusable procedures for complex tasks. However, manually authored skills often adapt poorly to long-horizon tasks and changing environments. To address the limitation, self-evolving skill systems have been developed to automatically construct and update skills from execution trajectories, shifting skill acquisition from external marketplaces to a trusted evolution pipeline. By replacing external skill acquisition with trusted i","url":"https://www.semanticscholar.org/paper/b4119fc7267c95cfc9c974be181981e520e73a64","categories":["data-poisoning","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01260","type":"paper","title":"ActivationBackdoor: Backdooring Large Language Models in Collaborative Inference via Intermediate Activations","authors":["Zichun Su","Mi Zhang","Xiaohan Zhang","Geng Hong","Xiaoyu You","Min Yang"],"year":2026,"venue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2","abstract":"Collaborative inference enables cost-effective deployment of large language models by partitioning layers across multiple participants and forwarding intermediate activations between participants in a pipeline, but these transmitted activations also create a new attack surface: a malicious participant can manipulate intermediate activations during inference. Prior work on collaborative inference attacks has largely focused on privacy leakage, leaving the backdoor threat insufficiently explored. ","url":"https://www.semanticscholar.org/paper/2c995168767f95b6d21697890d1d372354864e8e","categories":["data-poisoning","membership-inference"],"reviewed":false},{"id":"llmsec-2026-01261","type":"paper","title":"The 2nd SeT-LLM Workshop on Secure and Trustworthy Large Language Models","authors":["Lu Lin","Jinghui Chen","Ting Wang","Jieyu Zhao","Chaowei Xiao","Jian Kang","Michael Johnston"],"year":2026,"venue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2","abstract":"Large language models (LLMs) are increasingly embedded as core components of data-centric systems, supporting analytical decision making, and automated reasoning over large-scale, heterogeneous datasets. Yet their deployment in open-world environments raises fundamental challenges to security and trustworthiness: LLMs can leak sensitive data, fall prey to prompt injection and jailbreaks, generate misinformation, and behave unpredictably under adversarial inputs, failures that propagate through d","url":"https://www.semanticscholar.org/paper/29d67405e278415b6e1ce58cab820c84cc22ee0c","categories":["prompt-injection","jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01262","type":"paper","title":"Shifting the Unit of Safety: From Model to System in the Generative and Agentic Era","authors":["Sakshi Jain"],"year":2026,"venue":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2","abstract":"For a decade, responsible AI at internet scale rested on a reassuring assumption: that risk lives primarily in a model, it is a unit you can isolate, and that privacy, fairness and safety is therefore something you certify at model level, before launch. At LinkedIn, where AI shapes access to jobs and economic opportunity for over a billion members, we built our fairness, privacy, and explainability assurance on exactly this foundation, and it worked. Then generative AI dissolved the unit, and ag","url":"https://www.semanticscholar.org/paper/2824a603d456763cbae13f3650fcf785565615ce","categories":["agentic-threats","responsible-ai"],"reviewed":false},{"id":"llmsec-2026-01263","type":"paper","title":"Design of a Security Framework for Multi-Agent Systems Based on Model Context Protocol in SOC Environments","authors":["Rodrigo Tavares de Pina Simões","Xavier Larriva-Novo","Carmen Sánchez-Zas","V. A. Villagrá","Andrés I. Marín López"],"year":2026,"venue":"Applied Sciences","abstract":"Security Operations Centers (SOCs) rely on Level 1 analysts to triage increasing alert volumes amid alert fatigue and tool fragmentation. LLM-based multi-agent systems using the Model Context Protocol (MCP) are being adopted to automate these tasks, but their autonomy and tool access expose them to attacks such as tool poisoning, indirect prompt injection, and confused deputy exploitation. To address this gap, this work proposes a security framework for MCP-based multi-agent SOC pipelines, imple","url":"https://www.semanticscholar.org/paper/0817f247f0d795c3fd996bdb21e7ab0c3d23f942","categories":["prompt-injection","data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01264","type":"paper","title":"Genotypic Triggers: Exposing Pharmacogenomic Blind Spots via Host-Specific Backdoors in Generative Antimicrobial Peptide Models","authors":["Doniyorkhon Obidov","Xiaolong Guo","Yonghui Li","Kaichen Yang"],"year":2026,"abstract":"Large Language Models (LLMs) have accelerated drug discovery, particularly in the automated design of antimicrobial peptides (AMPs). However, current validation pipelines for peptide generation models overlook historical precedents showing that certain drugs carry health risks predominantly for individuals with specific genetic profiles. In this paper, we demonstrate that such targeted health risks can be induced intentionally and at scale by manipulating models that generate peptide candidates.","url":"https://www.semanticscholar.org/paper/f250e6ba3527d4ead098d1ab488168ec45a147d3","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01265","type":"paper","title":"Trustworthy Deployment of LLM-Based RAG Systems for Small Businesses: A Security and Confidence-Aware Framework †","authors":["Qingqing Wang","Jiazhu Xie","Bowen Li","Zi-Qi Xu","Fengling Han"],"year":2026,"venue":"Pragmatic Cybersecurity","abstract":"Small and medium-sized enterprises (SMEs) are increasingly deploying Large Language Model (LLM)-based agentic systems to support customer service and internal knowledge management. However, practical deployment of retrievalaugmented generation (RAG) systems continues to be challenging due to promptinjection risks, unreliable confidence estimation, and limited operational resources in real-world SME environments. This paper presents a secure and confidenceaware deployment framework for SME-orient","url":"https://www.semanticscholar.org/paper/c5aa804b63c29e3fcd6f7889f9086b2ef7aae71b","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01266","type":"paper","title":"Adversarial Attacks on Deep OCR Systems","authors":["Wenbo Sun","Hong-Zong Li","Yanyun Wang","Jia-Hao Ma","Shuxin Zhuang","Rong Feng","Shi-Qin Tang","Zi Liang"],"year":2026,"abstract":"Deep-OCR (DeepSeek-OCR) advances document recognition by treating the visual modality as an optical compression medium, enabling long-context OCR at low token cost. However, its increased complexity may introduce new security vulnerabilities. In this paper, we present, to the best of our knowledge, the first pure black-box adversarial attack against a generative OCR vision-language model, where only the decoded string can be queried and no gradients, logits, or model internals are available. We ","url":"https://www.semanticscholar.org/paper/949627af3c2bcd16e81a3c0d9d82bf097b7b3bce","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01267","type":"paper","title":"Taxonomy-Driven Analysis of Open-Source AI Risk Mitigation Tools","authors":["Afreen Alam","Evgenija Popchanovska","Ana Gjorgjevikj","Maryan Rizinski","Lubomir T. Chitkushev","Irena Vodenska","D. Trajanov"],"year":2026,"abstract":"Rapid adoption of large language models (LLMs) in enterprise settings has introduced operational, security, and governance risks. As generative AI applications move from pilot to production, manual harm identification and mitigation are becoming difficult to scale. Although many tools support model evaluation, adversarial testing, runtime guardrails, and observability, the tooling landscape remains fragmented. Tools are typically designed for specific engineering tasks and described in technical","url":"https://www.semanticscholar.org/paper/7365f517104b178e74f24b901798d81561b121c8","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01268","type":"paper","title":"From Forensics to Ecosystems: Rethinking Watermarks for Generative AI Oversight","authors":["Daniel Susser","John Thickstun","Gili Vidan"],"year":2026,"abstract":"The arrival of generative AI as a cheap, widely accessible commercial service, and the tidal wave of AI-generated synthetic content it has unleashed, have provoked deep epistemic and social anxieties and raised difficult governance questions that policymakers are struggling to address. One approach that has attracted both enthusiasm from regulators and skepticism from researchers is digital watermarking. Signals embedded in a synthetically-generated piece of content indicating that it was AI-gen","url":"https://www.semanticscholar.org/paper/4cc1dd3b77617cf5304e9c05b428d788bc7651e5","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01269","type":"paper","title":"LoRAScan: Detecting Backdoor Prompts in Low-Rank Adapters for Large Language Models via Down-Projection Activation Spikes","authors":["Doniyorkhon Obidov","H. Yu","Xiaolong Guo","Kaichen Yang"],"year":2026,"abstract":"Low-rank adaptation (LoRA) enables efficient specialization and distribution of large language models through compact adapters. However, untrusted adapters introduce a supply-chain threat: a backdoored adapter can cause a model to generate harmful content, malicious code, political propaganda, or covert advertisements when an input contains a hidden trigger. Adapter-agnostic defenses merge the adapter with the base model, which dilutes backdoor signals and reduces detection performance. Existing","url":"https://www.semanticscholar.org/paper/449779b572b8bdb28df60a03ccc7c18a432b738c","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01270","type":"paper","title":"Diffusion LLMs as Targets and Adversaries: Mechanistic Safety Exploits","authors":["Elena Dumitrescu","G. Lek","L. Chen","Jérémie Decouchant"],"year":2026,"abstract":"Diffusion Large Language Models (DLLMs) replace autoregressive next-token prediction with iterative parallel denoising, yet their internal safety mechanisms remain poorly understood. In this work, we investigate DLLMs both as targets and as adversaries, exposing mechanistic vulnerabilities in diffusion-based alignment. We first show that safety alignment in DLLMs remains sparse and transferable across architectures. DLLMs initialized from autoregressive predecessors inherit the same mechanistic ","url":"https://www.semanticscholar.org/paper/2a8e104fe0ef6e996a4f062e059d99aec114f88a","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01271","type":"paper","title":"ECHO: A Locally-Deployable Agentic Health Assistant with Temporal Memory, Safety Guardrails, and Speech Assessment","authors":["Abdulkadir Külçe","A. Esen","Cağla Fikir","Berker Kurt","Kuzey Arar","Gökhan Ercan","F. B. Tek"],"year":2026,"abstract":"This paper presents ECHO (Enhanced Care&Health Observer), a locally-deployable conversational health assistant for long-term chronic care management. ECHO integrates three complementary software modules developed under shared supervision as a unified system. The core module is an agentic chatbot built on a ReAct loop orchestrated via LangGraph, equipped with 17 clinical tools and a temporal knowledge graph for persistent cross-session memory; it achieves a 94.9% tool-execution pass rate across a","url":"https://www.semanticscholar.org/paper/d2edc092c9b3cb3ad5ef8c7157b31532c10aa263","categories":["agentic-threats","guardrails"],"reviewed":false},{"id":"llmsec-2026-01272","type":"paper","title":"MMAligner: Safeguarding Multimodal Large Language Models through Representation Calibration","authors":["Shenyi Zhang","Keyan Guo","Zihao Wang","Xuebin Li","Lingchen Zhao","Hongxin Hu","Chao Shen","Qian Wang"],"year":2026,"abstract":"Multimodal large language models (MLLMs) often refuse unsafe text prompts yet generate harmful responses to semantically equivalent multimodal inputs. Existing defenses either rely on external guardrails, which add inference overhead without repairing intrinsic flaws, or safety fine-tuning, which treats alignment as black-box optimization and may sacrifice utility or require large multimodal datasets. To identify the cause of this safety disparity, we analyze MLLM representations geometrically. ","url":"https://www.semanticscholar.org/paper/c3be36d17b18c2decb13515aef81d47f81e3366a","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01273","type":"paper","title":"Hijacking Robots with a Piece of Paper: A Systematic Study of Physical Prompt Injection in VLM-Controlled Robots","authors":["S. Samarakoon","M. A. V. J. Muthugala","W. K. R. Sachinthana","M. R. Elara"],"year":2026,"abstract":"Vision-Language Models (VLMs) are increasingly deployed as planners in robotic systems, where they translate natural-language commands into executable actions grounded in visual scene understanding. This tight coupling between perception and instruction-following introduces a new attack surface: adversarial text placed within the robot's visual field can act as an indirect prompt injection into the VLM's reasoning stack. We present a systematic study of physical prompt injection attacks against ","url":"https://www.semanticscholar.org/paper/3fa9e3ec5f9b01ff3e61b01227b43db042194334","categories":["prompt-injection"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01274","type":"paper","title":"PromptShield Home: Ambient Multimodal Prompt Injection Defense for Smart-Home Agents","authors":["He Zhang","Fei-Long Li","Ding-Ning Long","Yi Cui","Peijun Zhang","Yue-Wen Zhang","Qianyao Xu","Xinyi Fu"],"year":2026,"abstract":"Smart-home assistants increasingly use multimodal large language models (MLLMs) that perceive video and audio directly. This raises a safety question specific to the home: can the agent tell a genuine user command from ambient or externally-sourced content, television speech, on-screen text, or an overheard conversation, that merely looks like a command? We introduce PromptShield-Home, a pilot benchmark of realistic smart-home scenarios spanning addressee ambiguity, screen/audio injection, healt","url":"https://www.semanticscholar.org/paper/30d00b2862a54156f390d455743bd002230cb3bb","categories":["prompt-injection","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01275","type":"paper","title":"DreamGuard: Efficient Runtime Guardrail for LLM Agents via Risk-Aware World Model","authors":["Wenhao Lin","Chen Yu","Xingwei Lin","Sicong Cao","Xiang Chen","Lei Xue","Le Yu","Letian Sha","Chunming Wu"],"year":2026,"abstract":"As large language model (LLM) agents increasingly invoke external tools and interact with real-world systems, unsafe actions may cause irreversible consequences on external states, user data, and downstream services. Recent runtime guardrails mitigate such risks by checking proposed actions before execution, but many remain reactive: they primarily assess the apparent safety of the current action, lacking an explicit model of how risk evolves across the trajectory. This limitation creates a crit","url":"https://www.semanticscholar.org/paper/0e72e0b12ecf4eb87226b876a02999df62ca4a33","categories":["agentic-threats","guardrails"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01276","type":"paper","title":"WorldMark: A Plug-and-Play World Knowledge Interface for Cross-Host Language Model Watermarking","authors":["Song Xiao","Yuqi Yuan","Yanshuo Zhang","Ke-Jun Zhang"],"year":2026,"abstract":"Watermarking traces the provenance of text produced by large language models by embedding statistically detectable signals during decoding. Existing schemes fall into logits-based, sampling-based, entropy-aware, and adaptive-strength families, yet all of them place watermark signals according to local token statistics. In the open-ended text-generation settings evaluated in this work, local statistics may provide insufficient guidance for placing robust watermark signals. We introduce WorldMark,","url":"https://www.semanticscholar.org/paper/f27392e958e852d0478471dec6f7c049b1a98cee","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01277","type":"paper","title":"Trident : How to Break Deep Reinforcement Learning Cyber Defenses (Agentic)","authors":["Ryozo Masukawa","Ian Bryant","Armita Kazeminajafabadi","Sanggeon Yun","Hyunwoo Oh","Sungheon Jeong","Nathaniel D. Bastian","Mahdi Imani","Mohsen Imani"],"year":2026,"abstract":"Autonomous cyber defense systems based on Deep Reinforcement Learning (DRL) have attracted significant research attention, yet remain evaluated almost exclusively against static, heuristic red agents, leaving their robustness against adaptive threats critically understudied. Meanwhile, recent advances in Reinforcement Learning with Verifiable Rewards (RLVR) have improved LLM reasoning, but their integration into cybersecurity remains elusive due to the absence of suitable benchmark environments ","url":"https://www.semanticscholar.org/paper/f113cf61f49b1f85a1e9dc20d3d4e42a00917f68","categories":["agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01278","type":"paper","title":"Mood Matters: How Syntactic Sensitivity Undermines Safety Alignment","authors":["Alina Klerings","Jannik Brinkmann","Heiner Stuckenschmidt","Simone Ponzetto"],"year":2026,"abstract":"Large language models typically undergo post-training to align them with safety policies but there exist many sophisticated jailbreaks that sidestep established safeguards. For instance, prior work by Andriushchenko et al. (2025) has found that changing the grammatical tense from present to past can be enough to elicit harmful responses. In this work, we uncover a more general failure of non-imperative syntactic forms. We demonstrate that this syntactic vulnerability exists in 16 models up to 70","url":"https://www.semanticscholar.org/paper/d6d288940807fdef4800e0acdb1b2c14e657baed","categories":["jailbreaking","guardrails"],"reviewed":false},{"id":"llmsec-2026-01279","type":"paper","title":"The cognitive mechanism: safeguarding the future of immunological discovery in the GenAI era","authors":["Nigel J. Francis","David P. Smith"],"year":2026,"venue":"Discovery Immunology","abstract":"Abstract Scientific discovery is built on the rigorous interrogation of the unknown. In immunology, exploring novel mechanisms requires researchers to identify what is unknown, reconcile contradictory data through original thinking, and integrate disparate ideas to generate testable hypotheses. Undertaking this process is how students construct their own knowledge and experiences. However, the rapid integration of Generative AI (GenAI) into higher education risks short-circuiting this cognitive ","url":"https://www.semanticscholar.org/paper/c8ec4963ee5131120cf502dbcb8d7c5c9821fab3","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01280","type":"paper","title":"DataRx: Missingness-Aware Sampling for Safer Large Language Model Task-Specific Fine-Tuning","authors":["Junbo Zhang","Qian-Li Zhou","Xin-Yang Deng","Wen Jiang"],"year":2026,"abstract":"Task-specific fine-tuning can improve the performance of large language models (LLMs) on downstream tasks. However, our study reveals that task-specific fine-tuning can also weaken the safety guardrails of aligned LLMs. A widely adopted strategy for preserving safety during fine-tuning is to incorporate safety data. Although previous studies have shown that randomly mixing safety data can alleviate safety degradation, the underlying principle determining why some safety examples are more effecti","url":"https://www.semanticscholar.org/paper/821fc65cadf063781de7fa906391711eba1c8374","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01281","type":"paper","title":"ICO: Enhancing Semantic-Shift Jailbreaks via Iterative Context Optimization","authors":["Hujian Zhu","Yihao Huang","Felix Juefei-Xu","Xinfeng Li","Peng Zeng","Simeng Qin","Qing Guo","G. Pu"],"year":2026,"abstract":"Foundation models have achieved remarkable success across diverse tasks, but they remain vulnerable. To investigate such vulnerabilities, semantic-shift jailbreaks have recently emerged as a promising attack paradigm. They bypass explicit safety mechanisms by replacing harmful terms in original harmful questions with benign alternatives and leveraging contextual information to induce the target model to reinterpret these alternatives as their corresponding harmful concepts. However, existing sem","url":"https://www.semanticscholar.org/paper/f6f8fcc18350b6b6d2f859b383e0327b136d7d93","categories":["jailbreaking"],"reviewed":false},{"id":"llmsec-2026-01282","type":"paper","title":"DHMark: Public-Key Watermarking for LLM-Generated Text via Diffie-Hellman-Guided Rejection Sampling","authors":["Haocheng Fu","Yuqi Qian","Luyao Wang","Yun Cao"],"year":2026,"abstract":"Large language model (LLM) watermarking provides an important mechanism for tracing the provenance of generated text. Existing statistical watermarks are often effective and robust, but most of them rely on private detection keys, which centralizes verification and complicates public auditing. Recent public or publicly verifiable watermarking schemes improve key management, yet many of them rely on exact recovery of embedded cryptographic strings, making them fragile under token edits, truncatio","url":"https://www.semanticscholar.org/paper/f608d33b9ba8096c43e1c197f358157014cdcea7","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01283","type":"paper","title":"Pervasive Backdoor Vulnerabilities in Genomic Foundation Models","authors":["Shiwen Ni","Qianning Wang","Chi Wei","Xiaomin Ni","Shuai-Min Li","Zixin Zhao","Hui Li","Rongrong Ji","Teng Wang","Min Yang"],"year":2026,"venue":"bioRxiv","abstract":"Genomic foundation models are increasingly used to interpret and design DNA sequences, yet their susceptibility to training-data manipulation remains poorly understood. Here we systematically evaluate backdoor poisoning across three model families, seven parameter scales ranging from 50 million to 7 billion, and 18 genomic classification tasks. We introduce two complementary 48-nucleotide triggers: a composition-matched synthetic sequence and a biologically grounded trigger derived from transpos","url":"https://www.semanticscholar.org/paper/9a304471e350bc4004450beaf57b8eead3676e2b","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01284","type":"paper","title":"MalTotal: Cost-Effective and Language-Agnostic Malicious Code Poisoning Detection for Millions of Repositories","authors":["Jian Zhao","Shenao Wang","Qingyang Wu","Yanjie Zhao","Xiao Cheng","Hao-Yu Wang"],"year":2026,"abstract":"The widespread adoption of open source software (OSS) has introduced significant security risks, with malicious code poisoning attacks increasingly targeting public package registries and open-source platforms. Existing detection approaches, including heuristic-, learning-, and LLM-based methods, suffer from language-specific designs, limited generalization, and high analysis costs, making them unsuitable for large-scale multi-language analysis. To address these challenges, we propose MalTotal, ","url":"https://www.semanticscholar.org/paper/4ab91ee574adb3a457dd6e506a4cd327af5a1fd1","categories":["data-poisoning","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01285","type":"paper","title":"Attribute-based Undetectable Watermarking for Generative AI Models","authors":["Mi-Ying Huang","Chung-Wei Lee","Maximilian Raffel","Eric Tang"],"year":2026,"abstract":"Generative AI systems increasingly produce content whose provenance is difficult to verify, motivating watermarking techniques for identifying model-generated outputs. Existing cryptographic watermarking methods provide strong undetectability guarantees: without a detection key, watermarked outputs are computationally indistinguishable from unwatermarked ones. However, these approaches do not address the crucial deployment challenge of how to safely delegate detection capabilities. With an unres","url":"https://www.semanticscholar.org/paper/22b014d8b34c436dd36d1d6f95c2702661e6c876","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01286","type":"paper","title":"When Many Answers Are Valid, Voting Fails: Symbolic Verification for Best-of-K Causal Reasoning in LLMs","authors":["Omatharv Bharat Vaidya","C. Jerzak","Zayne Sprague","Fangcong Yin","N. Hồ"],"year":2026,"abstract":"Self-consistency assumes the most frequent answer among sampled reasoning traces is the most reliable, but this can fail in causal reasoning: samples often repeat the same confounding error, and votes fragment across multiple valid answers, letting an invalid answer win despite a valid minority trace. We introduce CALVER (Causal Axiom-Level VERification), a training-free symbolic verifier that scores structured traces against Pearl's causal criteria, including -separation, backdoor adjustment, a","url":"https://www.semanticscholar.org/paper/1fd04580be300512828ad334c3a57df3a7b06a63","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01287","type":"paper","title":"WeClawArena: An Auditable Sandbox and Benchmark for Cross-User Agents Collaboration and Security in Human-Centered Agent Networks","authors":["P. Wang","Ao-Jie Yuan","Haiyu Zhang","Xi-Yang Hu","Yue Zhao","Shuli Jiang"],"year":2026,"abstract":"Recent advances in persistent personal-agent frameworks are making human-centered agent networks realistic deployment targets: each user can be served by an AI agent that acts on the user's behalf, maintains state, and communicates with other agents through social and task relations. In these networks, everyday tool use becomes multi-party owned-agent collaboration over personal workspaces, where files, records, tools, and policies are not directly visible across owners. Existing agent benchmark","url":"https://www.semanticscholar.org/paper/1cdb58ea64b231ab64a0e542ad7c793814345ad2","categories":["sandboxing-isolation","benchmarks","tool-use-security"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01288","type":"paper","title":"When Agentic Executions Fail: Detecting and Localizing Runtime Faults from Telemetry","authors":["Chenkai Zhang","Yiran Li","Yifang Tian","Michalis Bachras","Hans-Arno Jacobsen"],"year":2026,"abstract":"Reliability in LLM-based agentic systems is a property of the whole execution (its tool calls, model calls, guardrails, and inter-agent messages), not of the final answer alone, yet evaluating only task outcomes reveals little about how or why a run fails. We present AGENTCHAOSBENCH, a benchmark for detecting and localizing runtime faults in agentic systems from their execution telemetry. We run five heterogeneous applications that coordinate agents over the Agent-to-Agent protocol and call tool","url":"https://www.semanticscholar.org/paper/12ee9654cd5677de7f8cd2ecff7ad773b3f779a8","categories":["agentic-threats","guardrails","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01289","type":"paper","title":"SoK: How Frontier AI Reshapes System-Level Security Risk Dynamics in Critical Infrastructure","authors":["C. Thapa","Mohan Baruwal Chhetri","M. Grobler","Shahroz Tariq","Tooba Aamir"],"year":2026,"abstract":"Frontier artificial intelligence (FAI), encompassing large-scale, general-purpose AI systems, including large language models, multimodal foundation models, and agentic systems, is increasingly integrated into critical infrastructure (CI). This challenges long-standing security assumptions of bounded behavior, segmented networks, component transparency, and human-paced decision-making. Existing AI-security literature typically organizes risks by attack type, lifecycle stage, or asset class, but ","url":"https://www.semanticscholar.org/paper/93c0882ee73a87235659637c1d1d59d0c14e83cd","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01290","type":"paper","title":"Salami Attack: Stealthy Collusive Memory Poisoning against OpenClaw","authors":["Zheng Lin","Yuzhen Huang","Zhenxing Niu","Xianmin Ye","Haichang Gao"],"year":2026,"abstract":"Long-term memory enables LLM agents to retain useful information across sessions, but also creates an attack surface through which adversaries may poison an agent's persistent memory to steer its behavior. Existing memory poisoning attacks mainly rely on individually malicious records, overlooking a compositional threat: multiple benign-looking memories may jointly induce unsafe behavior. In this paper, we introduce MemCollusion, an automated red-teaming framework for constructing collusive memo","url":"https://www.semanticscholar.org/paper/78a6fa7c6a59bb5fe8b4337b6f1b3d9b7b69328b","categories":["data-poisoning","agentic-threats","red-teaming","memory-security"],"reviewed":false},{"id":"llmsec-2026-01291","type":"paper","title":"EduZone: A Framework for Evaluating LLM Safety for K-12 Students and Teachers","authors":["Junyeong Park","Jieun Han","Haneul Yoo","So-Yeon Ahn","Jinsung Yoon","Alice Oh"],"year":2026,"abstract":"Large language models (LLMs) are increasingly used across diverse tasks in K-12 education, yet existing safety evaluations rarely examine how harmful or inappropriate content appears in interactions between LLMs and students or teachers. To address this, we present EduZone, an evaluation framework for LLM safety across diverse educational scenarios. Our framework systematically combines (1) student- and teacher-facing LLM usage contexts, (2) fine-grained curriculum concepts, and (3) 6 risk categ","url":"https://www.semanticscholar.org/paper/47e31fb7f693f63e6050538d45dc148cb863e76e","categories":["benchmarks"],"reviewed":false},{"id":"llmsec-2026-01292","type":"paper","title":"Invisible Ink Threats: Adversarial Goals Behind Legitimate Tasks in Computer-Use Agents","authors":["Jia-Chen Zhang","Zenghui Zhang","Kai-Wei Zhang"],"year":2026,"abstract":"Computer-use agents (CUAs), which empower large language models to autonomously operate operating systems and the web, are increasingly vulnerable to indirect prompt injection attacks. A widely adopted defense is the human-in-the-loop paradigm, in which the agent pauses for explicit user confirmation before executing sensitive operations. While effective against conspicuously high-harm attacks, this defense offers little protection against what we term Invisible Ink Threats: low-harm injected go","url":"https://www.semanticscholar.org/paper/452f5a76b8c18e0d4886e1aea1b1fa95201178ab","categories":["prompt-injection","human-in-the-loop"],"reviewed":false},{"id":"llmsec-2026-01293","type":"paper","title":"Advancing Relevance Measurement with Vision-Language Models for Web-Scale Search","authors":["Han Wang","Alex P. Whitworth","Pak-Ming Cheung","Zhenjie Zhang","Krishna Kamath","Xi Chen","Roberto Konow","Kurchi Subhra Hazra"],"year":2026,"abstract":"Relevance evaluation plays a crucial role in personalized search systems, serving as a guardrail alongside user engagement metrics to ensure that search results align with user queries and intent. While human annotation is the traditional method for relevance evaluation, its high cost and long turnaround time limit its scalability. In this work, we present a VLM-based automated relevance evaluation pipeline deployed within Pinterest Search for online A/B experiments. We rigorously validate the a","url":"https://www.semanticscholar.org/paper/07bc043030832c7a90b135a17b052ac818a21920","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01294","type":"paper","title":"Inverting the Hidden: Unveiling Multimodal Privacy Leakage in Collaborative LVLM Inference","authors":["Shuaifan Jin","Zhibo Wang","Qiyuan Wang","Yiting Han","Yajie Zhou","Yuanfan Zhang","Jiahui Hu","Xiaoyi Pang"],"year":2026,"abstract":"Collaborative inference deploys Large Vision-Language Models (LVLMs) by partitioning computation between edge devices and the cloud. While withholding raw inputs supposedly ensures privacy, transmitting intermediate hidden states exposes a critical attack surface. However, it remains unclear whether deep-layer LVLM hidden states retain recoverable private information, given that visual content has been projected into the language embedding space. To address this concern, we theoretically analyze","url":"https://www.semanticscholar.org/paper/887f61b15e19ba216107ce815c972613e659e020","categories":["membership-inference"],"reviewed":false},{"id":"llmsec-2026-01295","type":"paper","title":"When Collaboration Becomes a Trigger: Collective Evidence-Threshold Backdoors in Multi-Agent Systems","authors":["Jiahao Xiao","Lei Feng","Min-Ling Zhang"],"year":2026,"abstract":"LLM-based multi-agent systems (MAS) extend LLM capabilities through iterative communication and shared contexts. However, this collaboration introduces a vulnerability: backdoor behavior can be activated when peer evidence reaches a hidden threshold, rather than being determined by any single message. We introduce a collective evidence-threshold backdoor paradigm for MAS and Boundary-Conditioned Backdoor Injection (BCBI), which constructs counterfactual boundary pairs to separate benign behavior","url":"https://www.semanticscholar.org/paper/721e0bc978010a3cb051e3bc43537d30dddcc811","categories":["data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01296","type":"paper","title":"Securing agentic AI workflows: A defence-in-depth framework for autonomous systems","authors":["Sushma Mahadevaswamy"],"year":2026,"venue":"International Conference on Cyber Security And Protection Of Digital Services","abstract":"The rapid enterprise adoption of agentic artificial intelligence (AI) has introduced a category of security risk that existing cyber security frameworks were not designed to address. With 78 per cent of Fortune 500 companies projected to deploy agentic AI by 2026 and the global market expected to reach US$89.6bn, the attack surface created by these autonomous workflows demands urgent attention from security practitioners. This paper examines the distinct threat model presented by agentic AI, dra","url":"https://www.semanticscholar.org/paper/4e564ab7616c533887366a8634d3e255008bb4f7","categories":["agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01297","type":"paper","title":"DenialRAG: Single-Document RAG Poisoning via Embedded Parametric Denial","authors":["Abay Zhurekbay","Tao Liu","Fan Li"],"year":2026,"abstract":"Retrieval-augmented generation (RAG) systems are vulnerable to corpus poisoning: an attacker who inserts a crafted document into the retrieval corpus can steer the underlying large language model (LLM) toward an attacker-chosen wrong answer. Prior single-document attacks typically avoid explicitly naming and refuting the correct answer inside the poisoned passage. In this paper, we examine a complementary design and propose \\emph{DenialRAG}, a single-document poisoning attack that explicitly nam","url":"https://www.semanticscholar.org/paper/498d48436d075c7046441802b9b23026a318c81a","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01298","type":"paper","title":"D2ANN-RL: Defense-in-Depth ANN-Reinforcement Learning Framework for LLM Chatbot Code Injection Mitigation","authors":["Victor Omoboye Oluwasegun","O. Falebita","N. Adebola","V. Adekunle","D. Uzodinma","Toluhi Michael Lanre","D. Oyekunle","Chima-Duru Goodness Goziechukwu","Ugochukwu Okwudili Matthew"],"year":2026,"venue":"Scientific Journal of Computer Science","abstract":"The growing cybersecurity vulnerabilities in artificial intelligence (AI) service models, particularly Large Language Models (LLMs), highlight code injection as a critical threat to chatbot reliability and safe deployment. On the account that LLMs process inputs as undifferentiated token sequences, they cannot reliably distinguish trusted system prompts from untrusted user inputs. This architectural limitation enables attackers to exploit direct and indirect prompt injection channels, resulting ","url":"https://www.semanticscholar.org/paper/3213aa8ef5781dfbbcb9a0083de588fa3ae69e42","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01299","type":"paper","title":"Efficient and Privacy-Preserving Federated Knowledge Learning for Distributed LLM","authors":["Wei Sun","Xianda Wang","Zhicheng Liang","Tianyi Gong","Wanshun Lan","Yingchun Chen","Haoyue Li","Fangxin Wang"],"year":2026,"venue":"IEEE Transactions on Mobile Computing","abstract":"Federated learning (FL) stands out as a promising solution to address the ever-increasing data scarcity problem of large language model (LLM) training through collecting data from distributed sources. The unique challenges however exist in three aspects, the extreme large communication overhead due to frequent model aggregation, the suboptimal performance due to data heterogeneity, and privacy leakage risk due to model parameter exchange. To address these issues, we propose Federated Knowledge L","url":"https://www.semanticscholar.org/paper/f85d3d9c7b08f2e213a1e5c4d37b9256ead2f4a7","categories":["membership-inference","federated-learning"],"reviewed":false},{"id":"llmsec-2026-01300","type":"paper","title":"Semantic-Oriented Robust Sentence Level Text Watermark for Large Language Models","authors":["Bo Li","Kun Zhang","Chengyou Song","Xi Chen","Hailin Zhou","Richang Hong"],"year":2026,"venue":"IEEE Transactions on Computational Social Systems","abstract":"Text watermarking focuses on embedding identifiable information into the generated content, which has become increasingly important with the rapid development of large language models (LLMs). Existing watermarking works either divide the vocabulary of LLMs into “green” and “red” tokens for the watermark generation (i.e., token-level watermark), or use the distance of generated sentence embeddings to distinguish the “green” and “red” partitions (i.e., sentence-level watermark). Despite the achiev","url":"https://www.semanticscholar.org/paper/de0e56e50fc26303b1d9b0a7362e5a2eda32334a","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01301","type":"paper","title":"Vision-Language Semantics Guided Model Extraction Attack via Long-Horizon Contrastive Prompt Learning","authors":["Ruinan Ma","Yu-Xi Ma","Meng-Xia Ren","Dehua Zhu","Hong-Yi Liu","Yu-an Tan"],"year":2026,"venue":"Journal on spesial topics in mobile networks and applications","url":"https://www.semanticscholar.org/paper/d3068d4e9c55130f89ec16031e0d8345c7e162d4","categories":["model-extraction"],"reviewed":false},{"id":"llmsec-2026-01302","type":"paper","title":"Optimizing Secure AI Lifecycle Model Management with Innovative Generative AI Strategies","authors":["J. Ponsam","Nalam Siva Bhadra","Jai Kushal Bysani"],"year":2026,"venue":"International Conference on Circuit, Power and Computing Technologies","abstract":"The wide use of the artificial intelligence (AI) in serious applications has led to the problem of the safe management of the cycles models of the AI, such as training, running, and monitoring. The management of the models is usually faced with the problem of the trade-off between robustness, security, and flexibility, particularly when generative AI is integrated into the models. This paper presents a new hybrid synthetic algorithm, Generative Adversarial Networks (GANs) and Reinforcement Learn","url":"https://www.semanticscholar.org/paper/c0f75885aa7e236c348c050ddcefdd3cb8488126","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01303","type":"paper","title":"Whispers of Wealth: A Systematic Red-Teaming Study of the Agent Payments Protocol (AP2)","authors":["Tanusree Debi","Wentian Zhu"],"year":2026,"venue":"2026 International Conference on Intelligent Multimedia, Networking, and Security (IMNS)","abstract":"Large language model (LLM)-based agents are increasingly used to automate financial transactions, but their reliance on contextual reasoning introduces new security risks. The Agent Payments Protocol (AP2) secures agent-mediated purchases through cryptographically signed mandates; however, its robustness against reasoning-layer attacks remains unclear. This paper presents a systematic red-teaming evaluation of AP2 and identifies vulnerabilities arising from prompt injection. Two attack technique","url":"https://www.semanticscholar.org/paper/ad5348c7ef25e740afae709cb6bd799d38b5d7ab","categories":["prompt-injection","red-teaming","threat-modeling"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-01304","type":"paper","title":"Evaluating Prompt Injection Risk and Guardrails in LLM-Enabled Home IoT Assistants","authors":["Shazid Bin Zaman","Sohan Gyawali","C. Popoviciu","Yi-Li Jiang","Jiaqi Huang"],"year":2026,"venue":"2026 International Conference on Intelligent Multimedia, Networking, and Security (IMNS)","abstract":"Smart home virtual assistants are increasingly powered by large language models to enable information retrieval and home device actuation. As a result, intelligent home environments are becoming more exposed to untrusted inputs, increasing their susceptibility to prompt injection, role confusion, and indirect prompt injection through retrieved context. In this paper, we propose a layered architecture that separates LLM-driven intent interpretation from the authorization and safety enforcement me","url":"https://www.semanticscholar.org/paper/a374a751e5fd33edaee8fdcabeac3ef8f5ea590d","categories":["prompt-injection","guardrails","access-control"],"reviewed":false},{"id":"llmsec-2026-01305","type":"paper","title":"UniBreak: A Unified Evolutionary Token-Level Jailbreaking Framework for Large Language Models","authors":["Shen You","Wei Jiang","Hefei Mei","Danei Gong","Zhongshen Li","Jixang Yu","Jun-Kai Ji","Qiuzhen Lin","Xiangtao Li","Ka-chun Wong"],"year":2026,"venue":"IEEE Transactions on Evolutionary Computation","abstract":"Large language models (LLMs) demonstrate promising capabilities in natural language understanding and reasoning with enormous parameter spaces and vast amount of training data. These attributes have facilitated their deployment into diverse application domains. However, the underlying parameters implicitly assume decision-making boundaries, resulting in significant decision space regions not covered by training data. It makes them susceptible to adversarial manipulations through carefully crafte","url":"https://www.semanticscholar.org/paper/9b896fafa251bd1edf23469a7b625b9dbf768770","categories":["jailbreaking"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01306","type":"paper","title":"Prompt-Based Jailbreaking of Leading LLM Chatbots: A Survey of Attacks and Defenses","authors":["Brynn Knowlton","Jovani Campa","Davide Gallo","Khalil Dajani","Nabeel Alzahrani"],"year":2026,"venue":"IEEE Transactions on Artificial Intelligence","abstract":"Generative artificial intelligence (AI) systems—particularly large language models (LLMs)—remain vulnerable to jailbreak attacks: adversarial prompts that bypass safeguards and elicit unsafe or restricted outputs. This survey synthesizes jailbreak research from 2023–2025, covering attack methods, defense strategies, and evaluation frameworks. Jailbreak techniques are grouped into five main categories: prompt-based injections, role-play conditioning, multiturn dialogue, multilingual or multimodal","url":"https://www.semanticscholar.org/paper/9b0b450d7b38f5524e7e6a0aad293f3a521be348","categories":["jailbreaking","guardrails","benchmarks","survey"],"citation_count":5,"reviewed":false},{"id":"llmsec-2026-01307","type":"paper","title":"Governing generative AI in organizations: a design theory and quasi-experimental field study of sociotechnical guardrails","authors":["Maikel Leon"],"year":2026,"venue":"Journal of Supercomputing","abstract":"Generative AI adoption has outpaced organizational governance capabilities. We conceptualize AI guardrails as sociotechnical governance mechanisms, comprising policy, technical, and workflow components that embed organizational norms in deployed AI systems. Extending norm-based coordination accounts, we specify three mechanisms (norm encoding, output monitoring, and escalation) targeting four properties: predictability, fairness, safety, and auditability. We instantiate the theory in a three-lay","url":"https://www.semanticscholar.org/paper/955a0cafb871b88276f7ce9fde7b1bf0a2f68fdf","categories":["guardrails","monitoring-detection"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01308","type":"paper","title":"Multigranularity Adversarial Attacks on Large Language Models Using Genetic Programming","authors":["Wencheng Han","Hao Li","Maoguo Gong","Yu Zhou","Yue Wu","A. Qin","Lining Xing"],"year":2026,"venue":"IEEE Transactions on Evolutionary Computation","abstract":"large language models (LLMs) have demonstrated remarkable capabilities across various natural language processing tasks, but they remain vulnerable to adversarial attacks and pose significant security concerns. Existing attack methods often treat adversarial prompts as flat sequences, neglecting the rich hierarchical structure of natural language, which could limit their effectiveness. Advancing the methodologies for adversarial attacks is crucial for rigorously assessing the security of LLMs an","url":"https://www.semanticscholar.org/paper/8d3e15cda73e865ccfc8867d11be8352fd8a00d9","categories":["adversarial-examples"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01309","type":"paper","title":"MAPLE-Guard: Memory-Aware Link Enforcement Against Memory-Link Poisoning in Multi-Agent Systems","authors":["Wen-Jun Xiong","Yi-Jin Zhou","Jia-Qian Wang","Shangding Gu","Bo Tang","Zhiyu Li","Feiyu Xiong","Ying Wen","Muning Wen"],"year":2026,"abstract":"LLM-based multi-agent systems (MAS) increasingly rely on persistent private and shared memories for long-horizon coordination. This memory layer improves continuity, but it also gives attackers a durable channel: a poisoned memory can be written once, continuously retrieved in later tasks, promoted into shared memory, and reused by other agents. A single poisoned write can therefore steer many later decisions and contaminate agents that never saw the original attack, all while no malicious messa","url":"https://www.semanticscholar.org/paper/758e18c7614f494fe9e97fa9c1eaeb32b4063d1f","categories":["data-poisoning","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01310","type":"paper","title":"A Comparative Survey of Security Risks in AI Systems: From LLMs to AI Agents and Embodied Agents","authors":["Baiqi Wu","Qing-Ming Li","Chun-Yi Zhou","Ting Wang","Shoulin Ji"],"year":2026,"venue":"ACM Computing Surveys","abstract":"Rapid AI development across industries raises pressing security and privacy risks. This work presents a unified comparison of large language models, AI agents, and embodied agents, introducing a taxonomy of risks spanning data, models, systems, content, and applications, alongside a catalog of 24 specific threats. We contrast attack surfaces and methods across the three system types to reveal common patterns and distinctive vulnerabilities. We also survey mainstream AI security assessment framew","url":"https://www.semanticscholar.org/paper/6f3e92cf87e0f99a22425e710009a51958baf627","categories":["membership-inference","survey","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01311","type":"paper","title":"C-P&M: Unified cross-prompt and cross-model adversarial attack on multimodal large language models","authors":["Qiqi Bao","Yujie Yan","Xiaoguo Ding","Yaguan Qian","Zhao-Quan Gu","Shou-Ling Ji","Bin Wang"],"year":2026,"venue":"Neurocomputing","url":"https://www.semanticscholar.org/paper/584885c7d054f13acccd69f11ba3242219e56903","categories":["adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01312","type":"paper","title":"MedCouncil: A Debate-Enabled Multi-Agent Framework for Second-Opinion Clinical Decision Support in General Internal Medicine","authors":["Santipong Thaiprayoon","Phattharat Songthung"],"year":2026,"venue":"2026 7th International Conference on Big Data Analytics and Practices (IBDAP)","abstract":"Large Language Models (LLMs) show promise in healthcare, but single-model systems often suffer from hallucinations, lack transparency, and have limited clinical safety. We present MedCouncil, a multi-agent framework designed to serve as a second-opinion tool for general internal medicine. Inspired by multidisciplinary consultations, MedCouncil orchestrates nine specialized clinical agents through a centralized supervisor agent. The system features a structured, multi-round Debate Mode that mimic","url":"https://www.semanticscholar.org/paper/583373a0f86af92dd2097140781a7c8e4c37519b","categories":["agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01313","type":"paper","title":"Semantic-aware watermarking with adaptive injection for large language models","authors":["Jing Zhao","Hongwei Yang","Heng-Ji Dong","Hui He","Wei-Zhe Zhang","Haoyu Xu"],"year":2026,"venue":"Knowledge-Based Systems","url":"https://www.semanticscholar.org/paper/51e6f9505f72ec24c2dc1a87880a363af2ccb9b5","categories":["watermarking"],"reviewed":false},{"id":"llmsec-2026-01314","type":"paper","title":"HFAVL: Hard-Label Fusion Attack Toward Vision-Language Model","authors":["Hongbo Cao","Yongqi Sun","Li Duan","Yifan Sui","Xisu Wang"],"year":2026,"venue":"IEEE Internet of Things Journal","abstract":"With the advancement of artificial intelligence, vision-language models (VLMs) that integrate text and image modalities have become central to multimodal learning and are increasingly deployed in Internet of Things (IoT) environments such as smart surveillance, autonomous driving, and industrial monitoring. However, VLMs are also highly susceptible to adversarial attacks. Most previous black-box attack methods toward the VLMs rely on either the soft label or substitute models. However, most of t","url":"https://www.semanticscholar.org/paper/2e81df9a879be060866efffa59cd5fec618d8afe","categories":["adversarial-examples","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01315","type":"paper","title":"Prompt injection compromises large language model-based peer review: evidence from randomised controlled trial abstracts from anaesthesia journals.","authors":["A. de Cassai","B. Dost","E. Pistollato","F. Zarantonello","A. Boscolo","P. Navalesi"],"year":2026,"venue":"British Journal of Anaesthesia","url":"https://www.semanticscholar.org/paper/194174119a6ea4cc018234397d662d9d3c78673b","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01316","type":"paper","title":"Analyzing Structural and Semantic Barriers to Automated Adversary Emulation in Active Directory","authors":["Anita Ding","Anthony J. Rose","Mark G. Reith"],"year":2026,"venue":"National Aerospace and Electronics Conference","abstract":"Active Directory (AD) underpins enterprise identity management and is among the most heavily targeted assets in modern enterprise networks. When red teams emulate attackers targeting it, they typically chain BloodHound for relationship mapping, Impacket for protocol manipulation, and Responder for credential interception. Despite rapid advances in Large Language Models (LLMs), reliably automating this BIR workflow is still difficult, and the difficulty is not primarily one of reasoning. We analy","url":"https://www.semanticscholar.org/paper/183de1e3c5e49190a7e327d47f55739e55b4be4f","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-01317","type":"paper","title":"SAFE-HealCloud: Safety-Aware, Agentic Self-Healing for Cloud Infrastructure","authors":["Prudvi Saisaran Ponduru","Pavani Priya Vyshnavi Nandanavanam","S. Ponduru"],"year":2026,"venue":"International Journal of Scientific Research in Computer Science Engineering and Information Technology","abstract":"Cloud infrastructure failures are increasingly difficult to detect, diagnose, and remediate because production environments combine microservices, Kubernetes control loops, service meshes, serverless workloads, infrastructure-as-code, continuous delivery, and heterogeneous telemetry. Reactive monitoring and manual incident response remain necessary, but they do not scale to the volume, velocity, and causal complexity of modern cloud operations. This paper provides a structured synthesis of schol","url":"https://www.semanticscholar.org/paper/04f39380622596e2b33cc72708c6b18dc8b0fb27","categories":["agentic-threats","monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01318","type":"paper","title":"Cognitive Integrity Security: Securing Human-AI Interaction in Large Language Model Systems","authors":["David C. Flynn"],"year":2026,"abstract":"Large language models (LLMs) have expanded far beyond traditional software boundaries, mediating decisions, generating authoritative sounding content, and directly influencing human reasoning across consumer, enterprise, and critical infrastructure domains. The scale and diversity of today's AI landscape-spanning open-weight models, frontier scale systems, local inference runtimes, agentic workflows, and fine-tuned domain models-has created an ecosystem too vast and too fast moving for classical","url":"https://doi.org/10.22541/au.177437389.95026495/v1","categories":["agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01319","type":"paper","title":"Security Risk Assessment and Layered Protection Strategies for Large Language Model Banking Chatbots with Privacy Considerations","authors":["Galal Eltayeb","Abdalilah Alhalangy"],"year":2026,"venue":"The Scholar Journal for Sciences & Technology","abstract":"Abstract But now, given the AI revolution and increased interest in bringing virtual agents and assistants to life banks too are testing LLM-powered AI agents that may assist customers, explain and customize products as well as simplify operational work done by bank employees in the background. But similar systems are susceptible to prompt injection, insecure output handling, and other LLM-specific threats that had only become more prevalent since these publications. Existing surveys and framewo","url":"https://doi.org/10.53348/jsst5st8","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01320","type":"paper","title":"Not All Large Language Model Deployments Are Created Equal: A Taxonomy-Driven Survey of Security, Defense, and Governance","authors":["Nathaniel Kang","Jongho Im"],"year":2026,"abstract":"The rapid enterprise adoption of Large Language Models (LLMs) has generated an expanding attack surface that existing surveys address monolithically, leaving engineering practitioners without deployment-specific guidance. This survey organizes the LLM security and governance landscape along two system design dimensions—user accountability boundary and data exposure temporality—yielding four deployment archetypes: Secured Repository, Managed Pipeline, Fortified Service, and Contested Interface. T","url":"https://doi.org/10.2139/ssrn.6837687","categories":["survey"],"reviewed":false},{"id":"llmsec-2026-01321","type":"paper","title":"HeteroGuard: Constructing Safety Boundaries for Large Language Models via Heterogeneous Weak-Model Collaboration","authors":["Ping Chen","Yin Cai","Shuting Zhang","Yi Wang","Xingqiu Shen","Yuting Shang","Hong Zou"],"year":2026,"venue":"Security and Safety","abstract":"Large language models (LLMs) face safety risks in deployment, but existing evaluation methods rely on single powerful judges and treat safety as binary classification. This paper proposes HeteroGuard, a dynamic heterogeneous redundancy (DHR)-inspired framework that constructs a safety boundary around LLM outputs through heterogeneous weak-model collaboration. HeteroGuard deploys three diverse weak judges and aggregates their decisions via majority voting to define a boundary that organizes respo","url":"https://doi.org/10.1051/sands/2026018","categories":["guardrails"],"reviewed":false},{"id":"llmsec-2026-01322","type":"paper","title":"Countermind: A Multi-Layered Security Architecture for Large Language Models","authors":["Dominik Schwarz"],"year":2025,"abstract":"The security of Large Language Model (LLM) applications is fundamentally challenged by \"formfirst\" attacks like prompt injection and jailbreaking, where malicious instructions are embedded within user inputs. Conventional defenses, which rely on post hoc output filtering, are often brittle and fail to address the root cause: the model's inability to distinguish trusted instructions from untrusted data [1]. This paper proposes Countermind, a multi-layered security architecture intended to shift d","url":"https://doi.org/10.36227/techrxiv.175994550.08962082/v1","categories":["prompt-injection","jailbreaking","output-moderation"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01323","type":"paper","title":"Unified AI Security Posture Management Framework for Multi-Cloud Large Language Model Deployments","authors":["Vaishali Mahavratayajula"],"year":2026,"venue":"International Journal of Computational and Experimental Science and Engineering","abstract":"The rapid proliferation of large language models across multi-cloud environments has introduced new challenges for enterprise security posture management. Existing cloud security posture management tools operate individually, leading to blind spots and delayed threat response for distributed AI workloads. This survey presents emerging Unified AI Security Posture Management (UAI-SPM) frameworks for hosting LLM applications in multi-cloud environments. This article aggregates recent developments i","url":"https://doi.org/10.22399/ijcesen.5279","categories":["cloud-ai-security"],"reviewed":false},{"id":"llmsec-2026-01324","type":"paper","title":"Security and Safety Threats in the Large Language Model Supply Chain: A Systematic Survey and Taxonomy","authors":["Jose Luna","Lili Quan","Ivan Tan","Lingxiao Jiang","Ming Hu","Qiang Hu","Xiaofei Xie"],"year":2026,"abstract":"The rapid advancement and deployment of large language models (LLMs) have led to increasingly complex LLM based systems that operate as multi stage pipelines rather than isolated model instances. These systems span dataset construction, model preparation, system integration, and supporting components, together forming an end-to-end LLM supply chain (LLM-SC). In such a supply chain, threats introduced at any component may propagate downstream and upstream, ultimately compromising applications and","url":"https://doi.org/10.2139/ssrn.6327419","categories":["supply-chain-attacks"],"reviewed":false},{"id":"llmsec-2026-01325","type":"paper","title":"A Safety and Security-Centered Evaluation Framework for Large Language Models via Multi-Model Judgment","authors":["Jinxin Zhang","Yunhao Xia","Hong Zhong","Weichen Lu","Qingwei Deng","Changsheng Wan"],"year":2025,"venue":"Mathematics","abstract":"The pervasive deployment of large language models (LLMs) has given rise to mounting concerns regarding the safety and security of the content generated by these models. Nevertheless, the absence of comprehensive evaluation methods constitutes a substantial obstacle to the effective assessment and enhancement of the safety and security of LLMs. In this paper, we develop the Safety and Security (S&S) Benchmark, integrating multi-source data to ensure comprehensive evaluation. The benchmark compris","url":"https://doi.org/10.3390/math14010090","categories":["benchmarks"],"citation_count":4,"reviewed":false},{"id":"llmsec-2026-01326","type":"paper","title":"Basilisk: An Evolutionary AI Red-Teaming Framework for Systematic Security Evaluation of Large Language Models","authors":["Regaan R"],"year":2026,"abstract":"The rapid deployment of large language models (LLMs) in production environments has introduced a new class of security vulnerabilities that traditional software testing methodologies are ill-equipped to address. I present Basilisk, an opensource AI red-teaming framework that applies evolutionary computation to the systematic discovery of adversarial vulnerabilities in LLMs. At its core, Basilisk introduces Smart Prompt Evolution (SPE-NL), a genetic algorithm that treats adversarial prompts as or","url":"https://doi.org/10.2139/ssrn.6373439","categories":["red-teaming"],"reviewed":false},{"id":"llmsec-2026-01327","type":"paper","title":"Unvalidated Trust: Cross-Stage Vulnerabilities in Large Language Model Architectures","authors":["Dominik Schwarz"],"year":2025,"abstract":"As Large Language Models are increasingly integrated into complex automated pipelines, security vulnerabilities that extend beyond simple input filtering become a practical concern. This work presents a mechanism-centered taxonomy of 41 recurring risk patterns, identified through a standardized, text-only evaluation protocol on commercial LLMs under default settings. Our empirical observations reveal systemic weaknesses in unvalidated trust inheritance across processing stages, which allows infe","url":"https://doi.org/10.36227/techrxiv.176231660.07863161/v1","categories":["input-filtering"],"reviewed":false},{"id":"llmsec-2026-01328","type":"paper","title":"Secure Prompt Engineering: A Practical Framework for Mitigating Prompt Injection and Data Leakage in LLM-based Systems","authors":["Gustavo Viana"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly deployed in production systems, yet their prompt-based interaction paradigm introduces a novel attack surface encompassing prompt injection, instruction hijacking, and sensitive data leakage. This paper proposes and empirically evaluates the Secure Prompt Engineering Framework (SPEF), a four-layer, application-level defensive architecture designed to operate under black-box API conditions without requiring access to model weights or training pipeline","url":"https://doi.org/10.2139/ssrn.6956641","categories":["prompt-injection","membership-inference"],"reviewed":false},{"id":"llmsec-2026-01329","type":"paper","title":"Beyond Prompt Injection: Trust-Boundary Security Assurance for LLM-Integrated and Agentic Applications","authors":["Nazar Waheed"],"year":2026,"abstract":"Abstract Large language model (LLM) applications increasingly combine probabilistic language interpretation with retrieval, persistent memory, external tools, and delegated enterprise credentials. This creates a system-security problem in which untrusted semantic content can cross trust boundaries and acquire operational authority. Existing work provides prompt-injection attacks, agent benchmarks, risk taxonomies, and governance guidance, but less directly addresses how those threats should be t","url":"https://doi.org/10.21203/rs.3.rs-10798245/v1","categories":["prompt-injection","agentic-threats","benchmarks"],"reviewed":false},{"id":"llmsec-2026-01330","type":"paper","title":"ContrastShield: A Contrastive Fine-Tuning Approach for Robust Prompt Injection Detection in LLM Pipelines","authors":["Swethaa R"],"year":2026,"abstract":"Prompt injection attacks pose a critical security threat to large language model (LLM) pipelines, enabling adversaries to hijack model behavior by embedding malicious instructions within user inputs or retrieved tool outputs. Existing defences are primarily rule-based filters and zero-shot classifiers, which are brittle against paraphrased, obfuscated, and indirect injection variants. We present ContrastShield, a binary classifier built on DeBERTa-v3-base and trained with a novel contrastive fin","url":"https://doi.org/10.2139/ssrn.6645520","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01331","type":"paper","title":"LLM-powered SOC Assistants: Prompt Injection Risks in Threat Triage","authors":["Tayyeb Nadeem Somro","Bilal Arshad","Ammara Gul"],"year":2026,"abstract":"Security Operations Centres (SOCs) are increasingly deploying Large Language Model (LLM) assistants to accelerate threat triage, alert prioritisation, and incident response. While these systems offer substantial productivity gains, their integration into security-critical pipelines introduces a novel and underexplored attack surface: prompt injection. This paper investigates how adversaries can manipulate LLM-powered SOC assistants by embedding malicious instructions within security alerts, log ","url":"https://doi.org/10.2139/ssrn.7179658","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01332","type":"paper","title":"Containment over Detection: Cryptographic Boundary Enforcement for Prompt Injection Defense in Agentic LLM Systems","authors":["Saurabh Sharma"],"year":2026,"abstract":"Abstract Prompt injection—the manipulation of an LLM-based agent through adversarial content embedded in user input—poses a critical security risk in production agentic systems with access to sensitive datastores and code execution environments. Detection-based defenses (keyword filtering, learned classifiers) suffer from a fundamental accuracy–latency tradeoff and fail to address the root cause: the absence of a syntactic boundary between instructions and data in LLM context windows. We present","url":"https://doi.org/10.21203/rs.3.rs-10196595/v1","categories":["prompt-injection","agentic-threats","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01333","type":"paper","title":"PIDL: A Prompt Injection Detection Layer for LLM-Integrated Web Applications via Multi-Signal Semantic Analysis","authors":["Jigesh Sheoran"],"year":2026,"abstract":"Large Language Model (LLM) APIs are increasingly embedded into production web applications, creating a new class of security vulnerability: prompt injection. An adversarial user can embed instructions within their input that override the application's system prompt, exfiltrate confidential configuration, or redirect model behavior. No standardized, pre-execution defense mechanism currently exists for this threat class. We propose the Prompt Injection Detection Layer (PIDL), a stateless middlewar","url":"https://doi.org/10.2139/ssrn.6739079","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01334","type":"paper","title":"GraphShield: A Graph-Structured Defense Framework for Prompt Injection in RAG and Multi-Agent LLM Systems","authors":["Shreya Singh"],"year":2026,"abstract":"GraphShield is a graph-structured defense framework for LLMs that represents system prompts, retrieved knowledge, agents, and parsed instructions as directed Trust-Knowledge Graph (TKG). Security is formalized as reachability from an instruction node to policy node in a trust-threshold subgraph, and operationally instantiated through typed instruction-violation signatures that assign zero trust to pattern-matched instructions, together with provenance filtering of retrieved and agent-supplied co","url":"https://doi.org/10.2139/ssrn.7082874","categories":["prompt-injection","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01335","type":"paper","title":"Formal Operational Models for Protecting Web Interfaces of Legal LLM Systems from Prompt Injection and Insecure Output Handling","authors":["Grigorii Danileiko"],"year":2026,"venue":"International Journal of Advanced Artificial Intelligence Research","abstract":"The proliferation of large language model (LLM) systems in legal technology platforms has created a new class of web-interface security vulnerabilities that existing application security frameworks address incompletely. This paper examines prompt injection and insecure output handling as the two primary attack surfaces for legal LLM web applications, with particular attention to contract lifecycle management systems that expose natural-language interfaces to privileged document repositories. Dra","url":"https://doi.org/10.55640/ijaair-v03i05-03","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01336","type":"paper","title":"AgentForensics: Exploring the Real-Time Prompt Injection Detection and Forensics Threats in LLM Agents","authors":["Aparnaa Mahalaxmi Arulljothi","Theepan Kumar Gandhi"],"year":2026,"abstract":"With LLM agents increasingly deployed to autonomously processed external content like web pages, emails, documents, and API responses become targets of indirect prompt injection attacks where malicious instructions embedded in external content hijack the agent's behavior. Existing defenses focus primarily on filtering user-layer inputs, leaving the wider attack surface unaddressed. In this paper, we present AgentForensics, an open-source security framework that monitors entire LLM agent sessions","url":"https://doi.org/10.2139/ssrn.6589479","categories":["prompt-injection","agentic-threats"],"reviewed":false},{"id":"llmsec-2026-01337","type":"paper","title":"guardrail-rs A Fail-Open Reverse Proxy for Prompt-Injection Defense and PII Redaction in LLM Applications","authors":["Min Htet Myet"],"year":2026,"abstract":"Abstract Applications built on large language models (LLMs) typically forward user input to a model provider with no enforcement layer in between, leaving prompt-injection attempts and personally identifiable information (PII) to pass through unfiltered in both directions. We present guardrail-rs, an open-source reverse proxy, implemented in Rust, that sits between an application and its LLM provider and inspects every request and response before it crosses the network boundary. The system is bu","url":"https://doi.org/10.21203/rs.3.rs-10354224/v1","categories":["prompt-injection","guardrails"],"reviewed":false},{"id":"llmsec-2026-01338","type":"paper","title":"Data-Driven Persona Generation via Structured Analysis and LLM Prompt Injection: Comparing Analytical Method Selection and Combination Strategies","authors":["Yujin kim","Jaekwang Kim"],"year":2026,"abstract":"Consumer personas are essential for user understanding and marketing strategy, yet manual construction remains costly and difficult to scale. We propose a data-driven persona generation framework that systematically compares three interpretable text mining methods-SNA, LDA, and K-Means-across eleven experimental configurations and injects the resulting structured representations into LLM prompts. Experiments on two consumer review datasets (VOC: n=965; VAD: n=2,370) with N =20 repeated trials sh","url":"https://doi.org/10.2139/ssrn.7281659","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01339","type":"paper","title":"FEW-AI-SERIAL: A Domain-Agnostic Semantic Compression Protocol for LLM Prompt Injection and IoT Data Encoding","authors":["Vinícius Negrão","Maíra Bocci","Paulo Pitrez"],"year":2026,"abstract":"We present FEW-AI-SERIAL, a domain-agnostic semantic compression protocol that reduces structured data payloads by 68-92% while maintaining full human readability and native interpretability by Large Language Models (LLMs). Unlike binary serialization formats (Protocol Buffers, Avro, CBOR) that achieve comparable compression but produce opaque byte streams, FEW-AI-SERIAL uses positional mnemonic keys (2-3 uppercase characters) and domain-native value notation to produce compact text that both hu","url":"https://doi.org/10.2139/ssrn.6491958","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01340","type":"paper","title":"Kill-Chain Canaries: Stage-Level Tracking of Prompt Injection Across Attack Surfaces and Model Safety Tiers","authors":["Haochuan Wang","Zechen Zhang"],"year":2026,"abstract":"Multi-agent LLM systems are entering productionprocessing documents, managing workflows, acting on behalf of users-yet their resilience to prompt injection is still evaluated with a single binary: did the attack succeed? This leaves architects without the diagnostic information needed to harden real pipelines. We introduce a kill-chain canary methodology that tracks a cryptographic token through four stages (EXPOSED → PERSISTED → RELAYED → EXECUTED) across 950 runs, five frontier LLMs, six attac","url":"https://doi.org/10.2139/ssrn.6604918","categories":["prompt-injection","agent-architecture"],"reviewed":false},{"id":"llmsec-2026-01341","type":"paper","title":"RoLLMRec: a robust LLM-based recommender system for defending against shilling and prompt injection attacks","authors":["Sarama Shehmir","Rasha Kashef"],"year":2026,"venue":"Frontiers in Computer Science","abstract":"Large Language Models (LLMs) are increasingly being integrated into recommender systems, offering contextual reasoning, cross-domain adaptability, and natural language interaction. However, their adoption also introduces vulnerabilities such as prompt injection, semantic poisoning, and shilling attacks, which can distort recommendations and erode user trust. Addressing these risks is essential for the safe deployment of LLM-based recommenders. We propose RoLLMRec, a defense oriented architectura","url":"https://doi.org/10.3389/fcomp.2026.1735253","categories":["prompt-injection","data-poisoning"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01342","type":"paper","title":"LLM Firewall Using Validator Agent for Prevention Against Prompt Injection Attacks","authors":["Michal Podpora","Marek Baranowski","Maciej Chopcian","Lukasz Kwasniewicz","Wojciech Radziewicz"],"year":2025,"venue":"Applied Sciences","abstract":"Large Language Models with Retrieval-Augmented Generation are considered to be modern, chat-native interfaces to enterprise knowledge. However, deploying such systems safely requires precautions more advanced than input filtering. Numerous LLM-related security threats (including the top one: prompt injection attacks) demand robust defense mechanisms beyond input filtering. This paper extends our dual-agent RAG architecture as an LLM firewall with output-level security validation. Similar to netw","url":"https://doi.org/10.3390/app16010085","categories":["prompt-injection","input-filtering"],"citation_count":3,"reviewed":false},{"id":"llmsec-2026-01343","type":"paper","title":"Prompt Injection Attacks Against Clinical LLM Agents Accessing Electronic Health Records: A Survey, Threat Model, Benchmark Specification, and Layered Defense Synthesis","authors":["Divya Pandey","Shivani Manchanda","Gangesh Pathak","Nishant Sonkar"],"year":2026,"abstract":"Clinical large language model (LLM) agents are entering production hospital deployments, where they read longitudinal electronic health records (EHRs), retrieve evidence from clinical knowledge bases, and assist with summarization, dosing, triage, and guideline-based decisions. The same architectural patterns that make these agents useful-instruction-following on retrieved text, tool use over patient data, and multi-turn conversation-also expose them to prompt injection attacks across heterogene","url":"https://doi.org/10.2139/ssrn.6828838","categories":["prompt-injection","agentic-threats","benchmarks","tool-use-security","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01344","type":"paper","title":"Continual Red-Teaming and Guardrail Distillation for Tool-Using LLM Agents: Prompt-Injection Resistance with Utility Preservation","authors":["Wesley Gao"],"year":2026,"venue":"Stout in Computer Science and Technology Studies","abstract":"Tool-using language-model agents can convert indirect prompt injection into consequential actions, making guardrail quality a joint security, utility, and efficiency problem. This study evaluates a ReAct-style control, native tool filtering, deterministic self-verification, a calibrated action gate, and a distilled guardrail on AgentDojo v0.1.22. The evaluation covers 97 benign tasks, 629 canonical attack cases, ten attack formulations, and 32,456 recorded actions, of which 20,953 were assigned ","url":"https://doi.org/10.61424/hewtat90","categories":["prompt-injection","agentic-threats","guardrails","red-teaming"],"reviewed":false},{"id":"llmsec-2026-01345","type":"paper","title":"EvoShield: Selective Test-Time Adaptation for Prompt Injection Detection via Active LLM Querying","authors":["Zanhong Zheng","Jieming Liang","Mengqin Hu","Yijuan Pei","Guobao Xu","Zhenlu Wu"],"year":2026,"venue":"Mathematics","abstract":"Prompt injection detection is commonly studied as a static offline classification problem, yet deployed LLM systems face evolving attacks and distribution shift after deployment. Static detectors are therefore poorly matched to the threat model, while routing every input to a stronger external LLM is costly and defeats the purpose of a local detector. We formulate prompt injection detection as a selective test-time adaptation problem. Our framework combines a prompt-based local detector built on","url":"https://doi.org/10.3390/math14101719","categories":["prompt-injection","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01346","type":"paper","title":"Securing LLM Powered AI Browsers Against Prompt Injection: A Comprehensive Survey, Threat Taxonomy, and Defense Framework","authors":["Sabin Adhikari","Roshan Paudel","Dipesh Gautam","Sanjog Chhetri Sapkota","Biplab Dahal","Roshit Raj Paudel","Anupam Dhakal"],"year":2026,"abstract":"Prompt injection is a serious threat to the security of large language models operating in AI-powered browsers and autonomous web agents, which depend on the ability of those models to interpret instructions correctly as they are used for automated browsing, data extraction or content processing. The inherent shortcomings of these systems to confidently differentiate trusted instructions from malicious ones allow the threat to influence agent behavior, exfiltrate sensitive information, and circu","url":"https://doi.org/10.2139/ssrn.6340078","categories":["prompt-injection","membership-inference"],"reviewed":false},{"id":"llmsec-2026-01347","type":"paper","title":"EchoLeak: The First Real-World Zero-Click Prompt Injection Exploit in a Production LLM System","authors":["Pavan Reddy","Aditya Sanjay Gujral"],"year":2025,"venue":"Proceedings of the AAAI Symposium Series","abstract":"Large language model (LLM) assistants are increasingly integrated into enterprise workflows, raising new security concerns as they bridge internal and external data sources. This paper presents an in-depth case study of EchoLeak (CVE-2025-32711), a zero-click prompt injection vulnerability in Microsoft 365 Copilot that enabled remote, unauthenticated data exfiltration via a single crafted email. By chaining multiple bypasses--evading Microsoft’s XPIA (Cross Prompt Injection Attempt) classifier, ","url":"https://doi.org/10.1609/aaaiss.v7i1.36899","categories":["prompt-injection"],"citation_count":11,"reviewed":false},{"id":"llmsec-2026-01348","type":"paper","title":"EvalHack: Answer-Side Prompt Injection for Probing LLM Exam-Grading Panel Stability","authors":["Catalin Anghel","Marian Viorel Craciun","Adina Cocu","Andreea Alexandra Anghel","Antonio Stefan Balau","Adrian Istrate","Aurelian-Dumitrache Anghele"],"year":2026,"venue":"Information","abstract":"Large language models are increasingly used as automated graders, yet their reliability under answer-side manipulation and their behavior in multi-model panels remain insufficiently understood. This paper introduces EvalHack, a matrix benchmark in which a fixed committee of four LLMs grades university-level machine learning exam answers under a strict integer-only contract (0–10) grounded in instructor-authored rubric artifacts. The dataset comprises 100 students answering 10 short, open-ended i","url":"https://doi.org/10.3390/info17030297","categories":["prompt-injection","benchmarks"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01349","type":"paper","title":"PROMPT PERSISTENCE ATTACKS: LONG-TERM MEMORY POISONING IN LLM-BASED SYSTEMS","authors":["Pranav Bhatnagar"],"year":2026,"abstract":"Large Language Models (LLMs) are increasingly deployed as persistent, interactive systems that retain information across user interactions. These memory mechanisms are designed to enhance personalization, task continuity, and operational efficiency. However, persistence introduces a new and insufficiently examined security risk: the gradual corruption of long-term memory through legitimate interaction. This paper introduces the concept of Prompt Persistence Attacks, a class of long-horizon adver","url":"https://doi.org/10.2139/ssrn.6183548","categories":["data-poisoning","memory-security","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01350","type":"paper","title":"Image-embedded prompt injection vulnerability of vision-language models in dental radiology: a cross-vendor attack–defense evaluation","authors":["Babak Saravi","Daman Deep Singh","Lara Schorn","Andreas Vollmer","Christoph Sproll","Norbert Kübler","Felix Schrader"],"year":2026,"abstract":"Abstract Image-embedded prompt injection — adversarial text rendered into the pixel data of medical images — is an emerging threat to vision-language models (VLMs) used in clinical decision support. We systematically evaluated this vulnerability across four production-tier VLMs (GPT-4o, Gemini 2.5 Flash, Claude Sonnet 4.5, MedGemma 4B) on 270 dental panoramic radiographs from the DenTeX dataset under four attack classes (8 variants per image), comprising 9,720 baseline and 48,600 defense inferen","url":"https://doi.org/10.21203/rs.3.rs-9932271/v1","categories":["prompt-injection"],"reviewed":false},{"id":"llmsec-2026-01351","type":"paper","title":"Artificial Intelligence Security and Adversarial Machine Learning: Threat Models, Defensive Strategies, and Forensic Implications for Trustworthy AI Systems","authors":["James H. Senanu"],"year":2026,"abstract":"The rapid integration of artificial intelligence (AI) systems into security-critical domains has introduced new vulnerabilities, exposing these systems to a growing spectrum of adversarial threats. Adversarial machine learning (AML) has emerged as a key area of research aimed at understanding and mitigating these risks. This paper presents a structured synthesis of AML threat models, defensive mechanisms, and the forensic implications essential for trustworthy AI deployment. We systematically ca","url":"https://doi.org/10.2139/ssrn.6144306","categories":["responsible-ai","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01352","type":"paper","title":"Adversarial Machine Learning Threats To Medical Device AI Controllers","authors":["Venkata Sai Abhinav Piratla -"],"year":2026,"venue":"International Journal of Innovative Research and Creative Technology","abstract":"The integration of artificial intelligence into life-critical medical device controllers—including closed-loop insulin delivery systems and cardiac monitoring devices—introduces adversarial machine learning (AML) attack surfaces that conventional cybersecurity frameworks do not adequately address. Adversarial attacks targeting these systems carry direct patient safety implications, yet no comprehensive, medical-device-specific AML threat taxonomy exists in the current literature. This paper addr","url":"https://doi.org/10.62970/ijirct.v12.i2.2604012","categories":["adversarial-examples","monitoring-detection"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01353","type":"paper","title":"Adversarial Machine Learning: Security Risks and Defense Strategies in AI-Driven Applications","authors":["Harsh Verma"],"year":2026,"venue":"International Journal of Scientific Research and Management (IJSRM)","abstract":"As artificial intelligence becomes woven into critical applications such as healthcare, finance, autonomous systems, and cybersecurity, adversarial threats to machine learning models have grown into one of the most pressing concerns in the field. Adversarial machine learning studies how attackers exploit weaknesses in model architectures and data pipelines, manipulating inputs to trigger misclassification, extract sensitive information, or quietly degrade system performance. This article offers ","url":"https://doi.org/10.18535/ijsrm/v14i02.ec04","categories":["membership-inference","threat-modeling"],"reviewed":false},{"id":"llmsec-2026-01354","type":"paper","title":"Overview of Adversarial AI and Data Poisoning in Federated Learning","authors":["Mohammed Firdos Alam Sheikh"],"year":2026,"venue":"Advances in Computational Intelligence and Robotics","abstract":"Federated Learning (FL) is an emerging decentralized machine learning paradigm that enables multiple clients to collaboratively train a global model without sharing raw data, thereby preserving data privacy. . In adversarial settings, malicious clients can inject carefully crafted inputs or manipulate local training updates to degrade the global model's performance or embed backdoors. Data poisoning attacks, including label flipping and model update manipulation, pose significant threats by subt","url":"https://doi.org/10.4018/979-8-3373-6224-3.ch002","categories":["data-poisoning","federated-learning"],"reviewed":false},{"id":"llmsec-2026-01355","type":"paper","title":"Adversarial Machine Learning: Attack Vectors, Defences, and Robustness","authors":["Rizwan Tanveer"],"year":2026,"abstract":"Background. Adversarial machine learning has progressed from a marginal concern within machine learning research into a first-order discipline for the secure deployment of artificial intelligence systems in regulated and operational environments. The contemporary threat landscape encompasses evasion at inference time, data poisoning across training pipelines, model extraction and inference attacks against deployed systems, and a defence ecosystem whose claimed robustness frequently fails to gene","url":"https://doi.org/10.2139/ssrn.6696178","categories":["data-poisoning","model-extraction"],"reviewed":false},{"id":"llmsec-2026-01356","type":"paper","title":"Adversarial Machine Learning: Attacks, Defenses, and the Path Towards Trustworthy AI","authors":["Jieyao Pang"],"year":2026,"venue":"International Journal of Innovative Science and Research Technology","abstract":"Deep neural networks achieve strong performance on perception tasks but remain vulnerable to adversarial examples—imperceptibly perturbed inputs that induce confident misclassification. This dissertation reviews the adversarial attack–defence landscape and reports CIFAR-10 experiments using ResNet-18. It compares a standard baseline, a PGDadversarially trained model, and a model obtained from RobustBench under FGSM, PGD-20, and AutoAttack.","url":"https://doi.org/10.38124/ijisrt/26jun1631","categories":["adversarial-examples","responsible-ai"],"reviewed":false},{"id":"llmsec-2026-01357","type":"paper","title":"Osprey: Production-ready agentic AI for safety-critical control systems","authors":["Thorsten Hellert","João Montenegro","Antonin Sulc"],"year":2026,"venue":"APL Machine Learning","abstract":"Operating large-scale scientific facilities requires coordinating diverse subsystems, translating operator intent into precise hardware actions, and maintaining strict safety oversight. Language model-driven agents offer a natural interface for these tasks, but most existing approaches are not yet reliable or safe enough for production use. In this paper, we introduce Osprey, a framework for using agentic AI in large, safety-critical facility operations. Osprey is built around the needs of contr","url":"https://doi.org/10.1063/5.0306302","categories":["agentic-threats"],"citation_count":2,"reviewed":false},{"id":"llmsec-2026-01358","type":"paper","title":"Enhancing Network Security through AI-Powered Anomaly Detection Using Generative Adversarial Networks","authors":["C. Satya Kumar","Asha Sunki","Vinith Koppera","Manish Hakeem"],"year":2026,"venue":"Emerging Trends in Machine Learning, Data Science, and Internet of Things","abstract":"Developments in communication technology have facilitated more data sharing in geographically dispersed settings, but they have also enlarged the attack surface, raising questions about network security. Research focuses on AI-based anomaly detection systems to improve Network Intrusion Detection Systems (NIDSs) in order to address this. However, data imbalance makes it more difficult for AI models to learn and effectively identify threats when legitimate traffic outnumbers malicious traffic. To","url":"https://doi.org/10.2174/9798898814717126010016","categories":["monitoring-detection"],"reviewed":false},{"id":"llmsec-2026-01359","type":"paper","title":"Adversarial Machine Learning in Cybersecurity Attacks and Defense Mechanisms","authors":["T Chithralekha","Shivakumar E","N Legapriyadharshini"],"year":2025,"venue":"Machine Learning and Deep Learning Techniques for Cybersecurity Risk Prediction and Anomaly Detection","abstract":"Adversarial machine learning has emerged as a critical challenge in cybersecurity, particularly with the increasing reliance on automated defense systems in modern networks. This chapter explores the evolving landscape of adversarial attacks targeting cybersecurity models, focusing on their impact on real-time threat detection and mitigation strategies. The rise of complex, multi-layered defense mechanisms in smart networks has led to more sophisticated adversarial tactics, which are designed to","url":"https://doi.org/10.71443/9789349552043-07","categories":["adversarial-examples"],"citation_count":1,"reviewed":false},{"id":"llmsec-2026-01360","type":"paper","title":"No Time to Spare: Adversarial Machine Learning at Training and Inference Time","authors":["Xiaoyun Xu"],"year":2026,"abstract":"This thesis addresses the critical challenge of adversarial machine learning in deep learning models, focusing on the defense mechanisms against evasion (adversarial) attacks and backdoor attacks. Part I analyzes evasion attacks through the lens of information bottleneck theory, revealing that compressing redundant information in the input space enhances model robustness. This insight leads to the proposal of novel, theoretically grounded adversarial training methods for stronger defense against","url":"https://doi.org/10.54195/9789465152103","categories":["data-poisoning","adversarial-examples"],"reviewed":false},{"id":"llmsec-2026-01361","type":"paper","title":"Quantum Frontiers","authors":["Shruti Bamboria","Kiran R. Dodiya","Kapil Kumar"],"year":2026,"venue":"Advances in Computational Intelligence and Robotics","abstract":"This chapter examines the intersection of quantum computing, cybersecurity, and adversarial machine learning (AML), outlining both transformative opportunities and emerging risks. Quantum computing, leveraging superposition and entanglement, offers unparalleled computational speed-ups but simultaneously threatens classical cryptographic protocols via Shor's and Grover's algorithms. Meanwhile, AML reveals vulnerabilities in machine learning systems through evasion, poisoning, and inference attack","url":"https://doi.org/10.4018/979-8-3373-4347-1.ch013","categories":["data-poisoning"],"reviewed":false},{"id":"llmsec-2026-01362","type":"paper","title":"Adversarial Machine Learning on Automotive Attack Surfaces: Threats, Intrusion Detection, and Zero-Knowledge Defenses","authors":["Ezekiel Ologunde"],"year":2026,"abstract":"Modern vehicles are distributed embedded computing platforms whose expanding network connectivity-CAN bus, Bluetooth, cellular telematics, and over-the-air (OTA) update channels-exposes them to the same class of adversarial attacks studied in cloud and enterprise environments. Machine learning (ML)-based intrusion detection systems (IDS) have emerged as the primary defensive response, yet these models are themselves vulnerable to adversarial perturbation: a well-crafted malicious CAN frame can e","url":"https://doi.org/10.2139/ssrn.6278899","categories":["adversarial-examples"],"reviewed":false}]