{"slug":"detecting-ai-model-prompt-injection-attacks","source_name":"detecting-ai-model-prompt-injection-attacks","name":"Detecting AI Model Prompt Injection Attacks","description":"Detects prompt injection attacks targeting LLM-based applications using a multi-layered defense combining regex pattern matching for known attack signatures, heuristic scoring for structural anomalies, and transformer-based classification with DeBERTa models. The detector analyzes user inputs before they reach the LLM, flagging direct injections (system prompt overrides, role-play escapes, instruction hijacking) and indirect injections (encoded payloads, multi-language obfuscation, delimiter-based escapes). Based on the OWASP LLM Top 10 (LLM01:2025 Prompt Injection) and Simon Willison's prompt injection taxonomy. Activates for requests involving prompt injection detection, LLM input sanitization, AI security scanning, or prompt attack classification.","version":1,"lift":{"pass_rate_delta_pts":48,"pass_rate_pct":52,"total_cases":25,"passed_cases":13,"tokens_delta_pct":38.9,"turns_delta_pct":0,"verdict":"mixed","benchmark_model":"gemini-3.6-flash","grading_method":"judged","completed_at":"2026-07-28T01:14:59.029555+00:00"},"skill_score":0.52,"benchmark_models":[{"model":"gemini-3.6-flash","headline":true,"delta_pts":48,"with_pass_pct":52,"without_pass_pct":4,"tokens_delta_pct":38.9,"turns_delta_pct":0,"total_cases":25,"cases_aggregated":24,"verdict":"mixed","never_hurt":true,"completed_at":"2026-07-28T01:14:59.029555+00:00","run_id":"4990d58d-0692-4008-a2e9-26ee4cc04297","version_number":1,"is_latest_version":true,"gate":null}],"trust":{"skill_safety":"passed","safety_status":"clean","intent_verdict":null,"content_status":"clean","indexable":true},"license":"Apache-2.0","install_count":0,"manifest_hash":"7edc0248030118e7b9547edff19770cf08b8020d5865571e3c9cc2ab09ccac8f","raw_url":"https://app.decimal.ai/s/detecting-ai-model-prompt-injection-attacks/SKILL.md","scorecard_url":"https://app.decimal.ai/skills/detecting-ai-model-prompt-injection-attacks"}