{
  "hunt": {
    "meta": {
      "tlp": "clear",
      "hunt": {
        "handoff": "promote-to-detection",
        "trigger": "intel-report",
        "methodology": "model-assisted",
        "applicability": "campaign-specific",
        "justification": "AI adoption is outpacing governance; this hunt identifies the misuse of AI runtimes and exfiltration channels that bypass traditional signature-based detection."
      },
      "name": "AI-Driven Persistence and Automated Data Theft",
      "type": "investigation",
      "labels": [
        "hunt",
        "attack.t1041",
        "attack.t1190",
        "attack.t1566",
        "attack.t1543",
        "execution",
        "exfiltration",
        "initial access",
        "persistence"
      ],
      "related": [
        {
          "hunt": "malicious-ai-skill-execution",
          "reason": "This hunt focuses on runtime hijacks (PromptSpy), while execution of malicious plugins belongs in a hunt targeting container and dependency scanning.",
          "relation": "out-of-scope-alternative"
        }
      ],
      "targets": {
        "web": {
          "name": "Web server / proxy logs",
          "category": "siem",
          "telemetry": [
            "network"
          ]
        },
        "hunter": {
          "name": "Hunt agent",
          "agent": true
        },
        "analyst": {
          "name": "Tier-2 analyst",
          "role": "analyst"
        },
        "network": {
          "name": "Network telemetry",
          "category": "network",
          "telemetry": [
            "network"
          ]
        },
        "endpoint": {
          "name": "Endpoint telemetry (hb_ surfaces)",
          "category": "endpoint",
          "telemetry": [
            "endpoint"
          ]
        }
      },
      "analysis": "A standard rule might detect large uploads, but this hunt pivots between high-frequency HTTP traffic, fileless process state (on_disk = 0), and network baseline deviations to distinguish an intrusion from legitimate AI usage.",
      "coverage": [
        {
          "stage": "initial-access-ai-enhanced-phishing-and-exploitation",
          "steps": [
            "ai-service-leads",
            "assess-traffic-volume"
          ],
          "status": "covered"
        },
        {
          "stage": "persistence-ai-runtime-abuse",
          "steps": [
            "ai-runtime-anomalies"
          ],
          "status": "covered"
        },
        {
          "stage": "exfiltration-automated-data-theft",
          "steps": [
            "exfiltration-volume"
          ],
          "status": "covered"
        },
        {
          "stage": "execution-malicious-ai-skills",
          "reason": "Not examined by this hunt; belongs to a separate hunt.",
          "status": "out_of_scope"
        }
      ],
      "scenario": {
        "stages": [
          {
            "name": "AI-Enhanced Phishing and Exploitation",
            "slug": "initial-access-ai-enhanced-phishing-and-exploitation",
            "tactic": "initial-access",
            "techniques": [
              "T1190",
              "T1566"
            ],
            "observables": [
              "automated social media reconnaissance",
              "personalized phishing messages in local languages",
              "rapid exploitation of N-day vulnerabilities in public applications",
              "prompt injection via chatbots"
            ]
          },
          {
            "name": "Execution via Malicious AI Skills",
            "slug": "execution-malicious-ai-skills",
            "tactic": "execution",
            "techniques": [
              "T1204.002"
            ],
            "observables": [
              "malicious AI skills/plugins",
              "download of secondary malware payloads",
              "suspicious agent instructions causing unintended actions"
            ]
          },
          {
            "name": "AI Engine Runtime Persistence",
            "slug": "persistence-ai-runtime-abuse",
            "tactic": "persistence",
            "techniques": [
              "T1543"
            ],
            "observables": [
              "abuse of Google Gemini runtime for persistent execution",
              "AI-powered spyware (PromptSpy) persistence mechanisms"
            ]
          },
          {
            "name": "Automated Data Exfiltration",
            "slug": "exfiltration-automated-data-theft",
            "tactic": "exfiltration",
            "techniques": [
              "T1041"
            ],
            "observables": [
              "data exfiltration over existing C2 channels",
              "AI-driven classification and extraction of stolen data",
              "high-volume traffic to AI skills repositories"
            ]
          }
        ],
        "summary": "Threat actors are leveraging AI to automate victim reconnaissance, accelerate exploit development for public-facing applications, and conduct high-volume personalized phishing. The intrusion lifecycle involves the deployment of malicious AI skills or plugins that execute malware and exfiltrate data via automated classification agents."
      },
      "severity": "medium",
      "rationale": "Begin by scanning all endpoints for high-frequency AI domain traffic. Focus on departments likely to use AI, such as Engineering (coding assistants) and Marketing (content generation), as they are primary targets for AI-based persistence and data theft.",
      "guardrails": {
        "claims": "no_unsupported",
        "evidence": "citation_required",
        "telemetry": "untrusted",
        "missing_data": "not_benign"
      },
      "hypothesis": "An adversary is using compromised AI runtimes to maintain persistence and automate the exfiltration of sensitive data to AI skill repositories.",
      "parameters": {
        "scope_hosts": {
          "type": "list[host]",
          "default": [],
          "description": "Hosts identified in the scoping lead; leave empty to scan the full estate."
        },
        "lookback_days": {
          "type": "number",
          "default": "14",
          "description": "Days of history to examine."
        },
        "ai_service_domains": {
          "from": {
            "ref": "eset-research-2026",
            "kind": "article",
            "observed": "2026-05-01"
          },
          "type": "list[domain]",
          "default": [
            "openai.com",
            "anthropic.com",
            "gemini.google.com",
            "huggingface.co",
            "mistral.ai"
          ],
          "description": "Known AI provider and skill repository domains."
        },
        "ai_parent_processes": {
          "from": {
            "ref": "promptspy-persistence-research",
            "kind": "manual",
            "observed": "2026-06-15"
          },
          "type": "list[string]",
          "default": [
            "chrome.exe",
            "firefox.exe",
            "msedge.exe",
            "gemini.exe",
            "python.exe",
            "python3.exe"
          ],
          "description": "Common parent processes for AI runtimes and browsers."
        }
      },
      "provenance": {
        "authors": [
          {
            "org": "huntbase.io",
            "name": "Huntbase hunt generation"
          }
        ],
        "generated": {
          "by": "huntbase-hunt-generation",
          "from": "https://www.welivesecurity.com/en/business-security/cyberthreats-moving-faster-smbs-readiness-must-accelerate/",
          "gates": [
            "dry-run",
            "lint",
            "critic"
          ],
          "model": "hb_google/gemini-3-flash-preview"
        }
      },
      "references": [
        {
          "url": "https://www.welivesecurity.com/en/business-security/cyberthreats-moving-faster-smbs-readiness-must-accelerate/",
          "name": "Cyberthreats are moving faster than SMBs: Readiness must accelerate"
        }
      ],
      "blind_spots": [
        {
          "id": "no-http-telemetry",
          "risk": "A host connecting to a malicious AI skill repository via an SSH tunnel would be missed in the initial gate.",
          "stage": "initial-access-ai-enhanced-phishing-and-exploitation",
          "question": "Whether the host initiated the connection via a non-HTTP protocol or encrypted tunnel.",
          "requires": "hb_http_activity or proxy logs"
        },
        {
          "id": "no-endpoint-telemetry",
          "risk": "Without endpoint-level 'on_disk' data, fileless persistence within an AI interpreter is indistinguishable from legitimate process execution.",
          "stage": "persistence-ai-runtime-abuse",
          "question": "Whether the malicious runtime hijack was fileless or executed from memory.",
          "requires": "hb_process_activity with on_disk state"
        }
      ]
    },
    "name": "AI-Driven Persistence and Automated Data Theft",
    "description": "This hunt identifies the misuse of AI ecosystem components, such as hijacked Google Gemini runtimes and malicious AI skill repositories. It uses a gated flow to first identify hosts with high-frequency AI service interactions before performing deeper process-level and network-volume forensics. The hunt focuses on detecting 'PromptSpy' style persistence where malicious code runs within an AI interpreter and exfiltrates data at scale."
  },
  "nodes": [
    {
      "id": "hypothesis",
      "type": "hypothesis",
      "label": "Hypothesis",
      "config": {
        "tags": [],
        "coverage": [
          {
            "stage": "initial-access-ai-enhanced-phishing-and-exploitation",
            "steps": [
              "ai-service-leads",
              "assess-traffic-volume"
            ],
            "status": "covered"
          },
          {
            "stage": "persistence-ai-runtime-abuse",
            "steps": [
              "ai-runtime-anomalies"
            ],
            "status": "covered"
          },
          {
            "stage": "exfiltration-automated-data-theft",
            "steps": [
              "exfiltration-volume"
            ],
            "status": "covered"
          },
          {
            "stage": "execution-malicious-ai-skills",
            "reason": "Not examined by this hunt; belongs to a separate hunt.",
            "status": "out_of_scope"
          }
        ],
        "rationale": "An adversary is using compromised AI runtimes to maintain persistence and automate the exfiltration of sensitive data to AI skill repositories.",
        "blind_spots": [
          {
            "id": "no-http-telemetry",
            "risk": "A host connecting to a malicious AI skill repository via an SSH tunnel would be missed in the initial gate.",
            "stage": "initial-access-ai-enhanced-phishing-and-exploitation",
            "question": "Whether the host initiated the connection via a non-HTTP protocol or encrypted tunnel.",
            "requires": "hb_http_activity or proxy logs"
          },
          {
            "id": "no-endpoint-telemetry",
            "risk": "Without endpoint-level 'on_disk' data, fileless persistence within an AI interpreter is indistinguishable from legitimate process execution.",
            "stage": "persistence-ai-runtime-abuse",
            "question": "Whether the malicious runtime hijack was fileless or executed from memory.",
            "requires": "hb_process_activity with on_disk state"
          }
        ],
        "scoping_notes": "Begin by scanning all endpoints for high-frequency AI domain traffic. Focus on departments likely to use AI, such as Engineering (coding assistants) and Marketing (content generation), as they are primary targets for AI-based persistence and data theft.",
        "beyond_detection": "A standard rule might detect large uploads, but this hunt pivots between high-frequency HTTP traffic, fileless process state (on_disk = 0), and network baseline deviations to distinguish an intrusion from legitimate AI usage."
      }
    },
    {
      "id": "ai-service-leads",
      "type": "query",
      "label": "High-frequency AI service interactions",
      "config": {
        "dsl": "sqlite",
        "role": "scoping",
        "source": "web",
        "content": "SELECT device_hostname, url_hostname, COUNT(*) AS request_count, user_agent, MIN(time) AS first_request, MAX(time) AS last_request FROM hb_http_activity WHERE instr(',' || '{{ai_service_domains}}' || ',', ',' || LOWER(url_hostname) || ',') > 0 AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, url_hostname, user_agent HAVING request_count > 100 ORDER BY request_count DESC",
        "surface": "hb_http_activity",
        "description": "Identify hosts with unusual volumes of HTTP traffic to AI providers as a lead for automated agent abuse.",
        "expected_signal": "Hosts with high request counts suggest automated activity rather than manual chat. Silence indicates no large-scale AI service usage observed in HTTP logs."
      },
      "parents": [
        {
          "id": "hypothesis"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "scoping",
        "label": "High-frequency AI service interactions",
        "reads": [
          "device_hostname",
          "url_hostname",
          "user_agent",
          "time"
        ],
        "source": "hb_http_activity",
        "target": "web",
        "content": "SELECT device_hostname, url_hostname, COUNT(*) AS request_count, user_agent, MIN(time) AS first_request, MAX(time) AS last_request FROM hb_http_activity WHERE instr(',' || '{{ai_service_domains}}' || ',', ',' || LOWER(url_hostname) || ',') > 0 AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, url_hostname, user_agent HAVING request_count > 100 ORDER BY request_count DESC",
        "silence": "evidence_of_absence",
        "expected": "Hosts with high request counts suggest automated activity rather than manual chat. Silence indicates no large-scale AI service usage observed in HTTP logs.",
        "verified": "dry-run",
        "verified_at": "2026-09-29"
      }
    },
    {
      "id": "assess-traffic-volume",
      "type": "analytic",
      "label": "Assess traffic automation",
      "config": {
        "cite": "required",
        "tools": [
          "endpoint",
          "network",
          "web"
        ],
        "context": [
          "ai-service-leads"
        ],
        "objective": "Review the request counts and user agents from ai-service-leads. Mark hosts as suspicious if they show persistent, high-frequency requests or use non-standard browser user agents.",
        "description": "Determine if the lead traffic represents an automated AI agent or legitimate user interaction.",
        "max_iterations": 3,
        "expected_signal": "A list of hosts where the traffic volume and user agents suggest automated exfiltration or hijacked sessions.",
        "success_criteria": "A per-host verdict of suspicious or benign."
      },
      "parents": [
        {
          "id": "ai-service-leads"
        }
      ]
    },
    {
      "id": "gate-decision",
      "type": "checkpoint",
      "label": "Gate: Pursue forensics?",
      "config": {
        "fuzzy": true,
        "judge": "hunter",
        "question": "The assessment identifies at least one host where AI interaction is likely automated or malicious.",
        "condition": "The assessment identifies at least one host where AI interaction is likely automated or malicious.",
        "blind_spot": "no-http-telemetry",
        "confidence": "medium",
        "description": "Avoid expensive endpoint and network-wide queries if the HTTP leads are benign.",
        "checkpoint_type": "mandatory"
      },
      "parents": [
        {
          "id": "assess-traffic-volume"
        }
      ]
    },
    {
      "id": "ai-runtime-anomalies",
      "type": "query",
      "label": "AI runtime process anomalies",
      "config": {
        "dsl": "sqlite",
        "role": "detection-candidate",
        "source": "endpoint",
        "content": "SELECT device_hostname, process_name, process_cmd_line, parent_process_name, on_disk, user_name, time FROM hb_process_activity WHERE ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND (instr(',' || '{{ai_parent_processes}}' || ',', ',' || LOWER(process_name) || ',') > 0 OR instr(',' || '{{ai_parent_processes}}' || ',', ',' || LOWER(parent_process_name) || ',') > 0) AND (on_disk = 0 OR on_disk = 'false') AND time >= datetime('now', '-{{lookback_days}} days')",
        "surface": "hb_process_activity",
        "description": "Identify processes with no binary on disk running under AI-related parents, characteristic of PromptSpy.",
        "expected_signal": "A process launched from a browser or AI runtime that has since been deleted from disk. This is a high-confidence signal for fileless execution."
      },
      "parents": [
        {
          "id": "gate-decision",
          "branch": "on_supports"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "detection-candidate",
        "label": "AI runtime process anomalies",
        "reads": [
          "device_hostname",
          "process_name",
          "parent_process_name",
          "on_disk",
          "time"
        ],
        "source": "hb_process_activity",
        "target": "endpoint",
        "content": "SELECT device_hostname, process_name, process_cmd_line, parent_process_name, on_disk, user_name, time FROM hb_process_activity WHERE ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND (instr(',' || '{{ai_parent_processes}}' || ',', ',' || LOWER(process_name) || ',') > 0 OR instr(',' || '{{ai_parent_processes}}' || ',', ',' || LOWER(parent_process_name) || ',') > 0) AND (on_disk = 0 OR on_disk = 'false') AND time >= datetime('now', '-{{lookback_days}} days')",
        "silence": "not_evidence_of_absence",
        "expected": "A process launched from a browser or AI runtime that has since been deleted from disk. This is a high-confidence signal for fileless execution.",
        "verified": "dry-run",
        "verified_at": "2026-09-29"
      }
    },
    {
      "id": "exfiltration-volume",
      "type": "query",
      "label": "Massive data exfiltration to AI",
      "config": {
        "dsl": "sqlite",
        "role": "baseline",
        "source": "network",
        "content": "SELECT device_hostname, dst_endpoint_hostname, SUM(traffic_bytes) AS total_out_bytes, COUNT(*) AS session_count FROM hb_network_connection WHERE ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND instr(',' || '{{ai_service_domains}}' || ',', ',' || LOWER(dst_endpoint_hostname) || ',') > 0 AND direction = 'outbound' AND state_kind = 'log' AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, dst_endpoint_hostname HAVING total_out_bytes > 50000000",
        "surface": "hb_network_connection",
        "description": "Establish a baseline and identify hosts sending unusually large volumes of data to AI domains.",
        "expected_signal": "Hosts sending more than 50MB to AI domains. Silence suggests no bulk exfiltration occurred via these endpoints during the window."
      },
      "parents": [
        {
          "id": "gate-decision",
          "branch": "on_supports"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "baseline",
        "label": "Massive data exfiltration to AI",
        "reads": [
          "device_hostname",
          "dst_endpoint_hostname",
          "traffic_bytes",
          "direction",
          "state_kind",
          "time"
        ],
        "source": "hb_network_connection",
        "target": "network",
        "content": "SELECT device_hostname, dst_endpoint_hostname, SUM(traffic_bytes) AS total_out_bytes, COUNT(*) AS session_count FROM hb_network_connection WHERE ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND instr(',' || '{{ai_service_domains}}' || ',', ',' || LOWER(dst_endpoint_hostname) || ',') > 0 AND direction = 'outbound' AND state_kind = 'log' AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, dst_endpoint_hostname HAVING total_out_bytes > 50000000",
        "silence": "evidence_of_absence",
        "baseline": {
          "window": "{{lookback_days}}d",
          "compare": "prior_equal_window"
        },
        "expected": "Hosts sending more than 50MB to AI domains. Silence suggests no bulk exfiltration occurred via these endpoints during the window.",
        "verified": "dry-run",
        "prevalence": {
          "by": "device_hostname",
          "key": [
            "dst_endpoint_hostname"
          ],
          "rare_below": 3
        },
        "verified_at": "2026-09-29"
      }
    },
    {
      "id": "final-triage",
      "type": "analytic",
      "label": "Final intrusion triage",
      "config": {
        "cite": "required",
        "tools": [
          "endpoint",
          "network",
          "web"
        ],
        "context": [
          "assess-traffic-volume",
          "ai-runtime-anomalies",
          "exfiltration-volume"
        ],
        "objective": "Review the correlated evidence. Confirm a malicious verdict if a host shows suspicious traffic automation (from assess-traffic-volume), a fileless process anomaly (from ai-runtime-anomalies), and a matching exfiltration spike to an AI domain (from exfiltration-volume).",
        "description": "Correlate the HTTP automation leads with process anomalies and exfiltration volume to reach a verdict.",
        "max_iterations": 6,
        "expected_signal": "A definitive verdict on whether the host is compromised by an AI-powered persistent threat.",
        "success_criteria": "A verdict of malicious, suspicious, or benign citing specific event rows."
      },
      "parents": [
        {
          "id": "ai-runtime-anomalies",
          "kind": "merge"
        },
        {
          "id": "exfiltration-volume",
          "kind": "merge"
        }
      ]
    },
    {
      "id": "final-route",
      "type": "checkpoint",
      "label": "Route for containment",
      "config": {
        "fuzzy": true,
        "judge": "hunter",
        "question": "The final triage verdict is malicious for at least one host.",
        "condition": "The final triage verdict is malicious for at least one host.",
        "blind_spot": "no-endpoint-telemetry",
        "confidence": "high",
        "description": "Determine whether to trigger immediate containment or manual investigation.",
        "checkpoint_type": "mandatory"
      },
      "parents": [
        {
          "id": "final-triage"
        }
      ]
    },
    {
      "id": "isolate-compromised-host",
      "type": "action",
      "label": "Isolate compromised host",
      "config": {
        "target": "endpoint",
        "description": "Stop further exfiltration from the AI-hijacked host.",
        "instructions": "Isolate the host immediately. Revoke all active session tokens for AI services (OpenAI, Gemini) used on this device.",
        "action_approval": "required"
      },
      "parents": [
        {
          "id": "final-route",
          "branch": "on_supports"
        }
      ]
    },
    {
      "id": "analyst-confirmation",
      "type": "task",
      "label": "Analyst confirmation",
      "config": {
        "assignee": "analyst",
        "description": "Provide human oversight for suspicious or indeterminate cases.",
        "instructions": "Review the cited process and network volume rows. Verify if the 'on_disk = 0' process correlates with the network spike to the AI provider. Determine if this represents a rogue AI agent or a legitimate developer activity."
      },
      "parents": [
        {
          "id": "gate-decision",
          "branch": "default"
        },
        {
          "id": "gate-decision",
          "branch": "on_unavailable"
        },
        {
          "id": "final-route",
          "branch": "default"
        },
        {
          "id": "final-route",
          "branch": "on_unavailable"
        },
        {
          "id": "isolate-compromised-host"
        }
      ]
    },
    {
      "id": "close-out-report",
      "type": "task",
      "label": "Close out report",
      "config": {
        "assignee": "analyst",
        "description": "Document the findings and update governance notes for AI usage.",
        "instructions": "Summarize the findings. Note any legitimate automated AI tasks that should be excluded from future runs. Update the corporate AI policy if shadow AI usage was discovered."
      },
      "parents": [
        {
          "id": "gate-decision",
          "branch": "on_refutes"
        },
        {
          "id": "final-route",
          "branch": "on_refutes"
        },
        {
          "id": "analyst-confirmation"
        }
      ]
    }
  ]
}