{
  "hunt": {
    "meta": {
      "tlp": "clear",
      "hunt": {
        "handoff": "keep-as-periodic-hunt",
        "trigger": "intel-report",
        "methodology": "model-assisted",
        "applicability": "campaign-specific",
        "justification": "43% of workers lack AI training and 25% admit they would use personal accounts for sensitive contracts; this hunt secures IP and mitigates the external attack surface."
      },
      "name": "Shadow AI Usage and Prompt Injection Exposure",
      "type": "investigation",
      "labels": [
        "hunt",
        "attack.t1190",
        "attack.t1566",
        "collection",
        "exfiltration",
        "initial access"
      ],
      "related": [
        {
          "hunt": "dlp-sensitive-keyword-exfiltration",
          "reason": "Broader exfiltration hunts cover more targets; this is tuned for AI platforms.",
          "relation": "sibling"
        }
      ],
      "targets": {
        "web": {
          "name": "Web server / proxy logs",
          "category": "siem",
          "telemetry": [
            "network"
          ]
        },
        "hunter": {
          "name": "Hunt agent",
          "agent": true
        },
        "analyst": {
          "name": "Tier-2 analyst",
          "role": "analyst"
        },
        "network": {
          "name": "Network telemetry",
          "category": "network",
          "telemetry": [
            "network"
          ]
        },
        "endpoint": {
          "name": "Endpoint telemetry (hb_ surfaces)",
          "category": "endpoint",
          "telemetry": [
            "endpoint"
          ]
        }
      },
      "analysis": "A rule flags AI domains; the hunt pivots to file activity within a narrow temporal window to prove intent, and evaluates the external exposure that standard internal rules miss.",
      "coverage": [
        {
          "stage": "initial-access-prompt-injection",
          "steps": [
            "external-ai-exposure"
          ],
          "status": "covered"
        },
        {
          "stage": "collection-sensitive-file-access",
          "steps": [
            "sensitive-file-activity"
          ],
          "status": "covered"
        },
        {
          "stage": "exfiltration-shadow-ai-usage",
          "steps": [
            "personal-ai-access",
            "network-data-volume"
          ],
          "status": "covered"
        }
      ],
      "scenario": {
        "stages": [
          {
            "name": "Prompt Injection against Public-Facing Chatbots",
            "slug": "initial-access-prompt-injection",
            "tactic": "initial-access",
            "techniques": [
              "T1190"
            ],
            "observables": [
              "public-facing AI chatbot",
              "malicious instructions slipped into prompts",
              "manipulated chatbot responses",
              "credential leakage via AI interface"
            ]
          },
          {
            "name": "Unauthorized Processing of Sensitive Documents",
            "slug": "collection-sensitive-file-access",
            "tactic": "collection",
            "observables": [
              "copying client contracts",
              "client contract summary generation",
              "accessing sensitive information for AI processing"
            ]
          },
          {
            "name": "Data Exfiltration via Personal AI Accounts",
            "slug": "exfiltration-shadow-ai-usage",
            "tactic": "exfiltration",
            "observables": [
              "personal ChatGPT account usage",
              "personal Claude account usage",
              "chatgpt.com",
              "claude.ai",
              "gemini.google.com",
              "uploading data to non-enterprise AI vendors"
            ]
          }
        ],
        "summary": "This analysis identifies two primary AI-related threat vectors: prompt injection attacks against public-facing corporate chatbots used to leak credentials or sensitive data, and the exfiltration of sensitive information by employees using personal AI accounts without authorization. A lack of formal policy and training leaves organizations vulnerable to both external exploitation and internal data leakage via personal accounts."
      },
      "severity": "medium",
      "rationale": "Identify exposed AI assets in the first step to establish the organization's external attack surface. The analyst then uses findings from the shadow AI lead to populate the time window and host parameters for the deeper file-level investigation.",
      "guardrails": {
        "claims": "no_unsupported",
        "evidence": "citation_required",
        "telemetry": "untrusted",
        "missing_data": "not_benign"
      },
      "hypothesis": "Employees are bypassing corporate AI controls by using personal accounts to process sensitive documents, or external attackers are exploiting public-facing AI applications to extract internal data.",
      "parameters": {
        "scope_hosts": {
          "type": "list[host]",
          "default": [],
          "description": "Optional list of hosts to narrow the hunt, populated from the lead results."
        },
        "lookback_days": {
          "type": "number",
          "default": "14",
          "description": "Days of history to examine for AI usage and file access."
        },
        "personal_ai_ips": {
          "from": {
            "ref": "https://www.huntress.com/blog/llm-security-report",
            "kind": "article",
            "observed": "2026-10-02"
          },
          "type": "list[ip]",
          "default": [
            "104.18.37.228",
            "172.64.150.31"
          ],
          "description": "Observed IP addresses for personal AI services to verify data volume."
        },
        "ai_access_timestamp": {
          "type": "string",
          "default": "2026-10-02T12:00:00Z",
          "description": "The UTC timestamp from the lead query result to center the file-activity window."
        },
        "personal_ai_domains": {
          "from": {
            "ref": "https://www.huntress.com/blog/llm-security-report",
            "kind": "article",
            "observed": "2026-10-02"
          },
          "type": "list[domain]",
          "default": [
            "chatgpt.com",
            "claude.ai",
            "gemini.google.com",
            "openai.com",
            "anthropic.com"
          ],
          "description": "Domains associated with personal-tier AI services."
        }
      },
      "provenance": {
        "authors": [
          {
            "org": "huntbase.io",
            "name": "Huntbase hunt generation"
          }
        ],
        "generated": {
          "by": "huntbase-hunt-generation",
          "from": "https://www.huntress.com/blog/llm-security-report",
          "gates": [
            "dry-run",
            "lint"
          ],
          "model": "hb_google/gemini-3-flash-preview"
        }
      },
      "references": [
        {
          "url": "https://www.huntress.com/blog/llm-security-report",
          "name": "Huntress \u2014 Companies Push AI Use But Skip Training and Official Policy"
        }
      ],
      "blind_spots": [
        {
          "id": "no-tls-visibility",
          "risk": "Without TLS decryption, we see the connection but cannot confirm the presence of sensitive text in the prompts.",
          "owner": "Network Engineering",
          "stage": "exfiltration-shadow-ai-usage",
          "question": "What specific text was included in the AI prompts?",
          "requires": "TLS inspection or endpoint proxy logs",
          "remediation": "Enable TLS decryption for AI endpoints."
        },
        {
          "id": "no-clipboard-logs",
          "risk": "We rely on temporal proximity; without clipboard logs, we cannot definitively prove data movement.",
          "owner": "Security Operations",
          "stage": "collection-sensitive-file-access",
          "question": "Did the user copy-paste content from the file into the browser?",
          "requires": "EDR with clipboard telemetry",
          "remediation": "Deploy endpoint monitoring that captures clipboard events."
        }
      ]
    },
    "name": "Shadow AI Usage and Prompt Injection Exposure",
    "description": "The hunt identifies exposed AI assets and detects internal traffic to personal AI platforms. It gates the deeper investigation on initial AI traffic to focus on high-risk hosts where an analyst must confirm whether sensitive files were moved to personal accounts. By separating external prompt-injection threats from internal shadow AI usage, the hunt ensures a focused response to both risk vectors."
  },
  "nodes": [
    {
      "id": "hypothesis",
      "type": "hypothesis",
      "label": "Hypothesis",
      "config": {
        "tags": [],
        "coverage": [
          {
            "stage": "initial-access-prompt-injection",
            "steps": [
              "external-ai-exposure"
            ],
            "status": "covered"
          },
          {
            "stage": "collection-sensitive-file-access",
            "steps": [
              "sensitive-file-activity"
            ],
            "status": "covered"
          },
          {
            "stage": "exfiltration-shadow-ai-usage",
            "steps": [
              "personal-ai-access",
              "network-data-volume"
            ],
            "status": "covered"
          }
        ],
        "rationale": "Employees are bypassing corporate AI controls by using personal accounts to process sensitive documents, or external attackers are exploiting public-facing AI applications to extract internal data.",
        "blind_spots": [
          {
            "id": "no-tls-visibility",
            "risk": "Without TLS decryption, we see the connection but cannot confirm the presence of sensitive text in the prompts.",
            "owner": "Network Engineering",
            "stage": "exfiltration-shadow-ai-usage",
            "question": "What specific text was included in the AI prompts?",
            "requires": "TLS inspection or endpoint proxy logs",
            "remediation": "Enable TLS decryption for AI endpoints."
          },
          {
            "id": "no-clipboard-logs",
            "risk": "We rely on temporal proximity; without clipboard logs, we cannot definitively prove data movement.",
            "owner": "Security Operations",
            "stage": "collection-sensitive-file-access",
            "question": "Did the user copy-paste content from the file into the browser?",
            "requires": "EDR with clipboard telemetry",
            "remediation": "Deploy endpoint monitoring that captures clipboard events."
          }
        ],
        "scoping_notes": "Identify exposed AI assets in the first step to establish the organization's external attack surface. The analyst then uses findings from the shadow AI lead to populate the time window and host parameters for the deeper file-level investigation.",
        "beyond_detection": "A rule flags AI domains; the hunt pivots to file activity within a narrow temporal window to prove intent, and evaluates the external exposure that standard internal rules miss."
      }
    },
    {
      "id": "external-ai-exposure",
      "type": "query",
      "label": "Establish external AI attack surface",
      "config": {
        "dsl": "sqlite",
        "role": "scoping",
        "source": "endpoint",
        "content": "SELECT domain_or_ip, asset_type, product, port, discovered_at FROM hb_exposed_assets WHERE (LOWER(product) LIKE '%ai%' OR LOWER(product) LIKE '%llama%' OR LOWER(product) LIKE '%gpt%' OR LOWER(product) LIKE '%chat%' OR LOWER(product) LIKE '%notebook%' OR LOWER(product) LIKE '%langchain%')",
        "surface": "hb_exposed_assets",
        "description": "Identify public-facing AI applications that attackers might target for prompt injection.",
        "expected_signal": "A list of IP addresses or domains hosting AI-related software exposed to the internet."
      },
      "parents": [
        {
          "id": "hypothesis"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "scoping",
        "label": "Establish external AI attack surface",
        "reads": [
          "domain_or_ip",
          "asset_type",
          "product",
          "discovered_at"
        ],
        "source": "hb_exposed_assets",
        "target": "endpoint",
        "content": "SELECT domain_or_ip, asset_type, product, port, discovered_at FROM hb_exposed_assets WHERE (LOWER(product) LIKE '%ai%' OR LOWER(product) LIKE '%llama%' OR LOWER(product) LIKE '%gpt%' OR LOWER(product) LIKE '%chat%' OR LOWER(product) LIKE '%notebook%' OR LOWER(product) LIKE '%langchain%')",
        "silence": "evidence_of_absence",
        "expected": "A list of IP addresses or domains hosting AI-related software exposed to the internet.",
        "verified": "dry-run",
        "verified_at": "2026-10-04"
      }
    },
    {
      "id": "personal-ai-access",
      "type": "query",
      "label": "Lead: Detect personal AI platform usage",
      "config": {
        "dsl": "sqlite",
        "role": "baseline",
        "source": "web",
        "content": "SELECT device_hostname, actor_user_name, url_hostname, COUNT(*) as request_count, MIN(time) as first_access, MAX(time) as last_access FROM hb_http_activity WHERE (instr(',' || '{{personal_ai_domains}}' || ',', ',' || LOWER(url_hostname) || ',') > 0) AND ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, actor_user_name, url_hostname",
        "surface": "hb_http_activity",
        "description": "Find internal hosts accessing personal AI domains that lack enterprise data protections.",
        "expected_signal": "Users and hosts accessing personal AI tools. Silence indicates absence of traffic to these domains."
      },
      "parents": [
        {
          "id": "external-ai-exposure"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "baseline",
        "label": "Lead: Detect personal AI platform usage",
        "reads": [
          "device_hostname",
          "actor_user_name",
          "url_hostname",
          "time"
        ],
        "source": "hb_http_activity",
        "target": "web",
        "content": "SELECT device_hostname, actor_user_name, url_hostname, COUNT(*) as request_count, MIN(time) as first_access, MAX(time) as last_access FROM hb_http_activity WHERE (instr(',' || '{{personal_ai_domains}}' || ',', ',' || LOWER(url_hostname) || ',') > 0) AND ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, actor_user_name, url_hostname",
        "silence": "evidence_of_absence",
        "baseline": {
          "window": "{{lookback_days}}d",
          "compare": "first_seen"
        },
        "expected": "Users and hosts accessing personal AI tools. Silence indicates absence of traffic to these domains.",
        "verified": "dry-run",
        "prevalence": {
          "by": "device_hostname",
          "key": [
            "url_hostname"
          ],
          "rare_below": 5
        },
        "verified_at": "2026-10-04"
      }
    },
    {
      "id": "evaluate-ai-lead",
      "type": "analytic",
      "label": "Evaluate AI lead",
      "config": {
        "cite": "required",
        "tools": [
          "endpoint",
          "network",
          "web"
        ],
        "context": [
          "personal-ai-access"
        ],
        "objective": "Determine if the HTTP activity in personal-ai-access indicates probable personal account usage.",
        "description": "Determine if the HTTP traffic indicates personal account usage rather than corporate instances.",
        "max_iterations": 3,
        "expected_signal": "A per-host verdict on the likelihood of shadow AI usage.",
        "success_criteria": "A per-host verdict of personal-usage or benign."
      },
      "parents": [
        {
          "id": "personal-ai-access"
        }
      ]
    },
    {
      "id": "gate-on-lead",
      "type": "checkpoint",
      "label": "Gate: Confirm personal usage",
      "config": {
        "fuzzy": true,
        "judge": "hunter",
        "question": "the evaluate-ai-lead verdict identifies personal-usage for at least one host",
        "condition": "the evaluate-ai-lead verdict identifies personal-usage for at least one host",
        "blind_spot": "no-tls-visibility",
        "confidence": "high",
        "description": "Open the expensive file and network volume queries only when personal AI usage is confirmed.",
        "checkpoint_type": "mandatory"
      },
      "parents": [
        {
          "id": "evaluate-ai-lead"
        }
      ]
    },
    {
      "id": "sensitive-file-activity",
      "type": "query",
      "label": "Correlate sensitive document access",
      "config": {
        "dsl": "sqlite",
        "role": "enrichment",
        "source": "endpoint",
        "content": "SELECT device_hostname, actor_user_name, file_name, file_path, time FROM hb_file_activity WHERE (LOWER(file_name) LIKE '%contract%' OR LOWER(file_name) LIKE '%confidential%' OR LOWER(file_name) LIKE '%salary%') AND ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND time BETWEEN datetime('{{ai_access_timestamp}}', '-1 hour') AND datetime('{{ai_access_timestamp}}', '+1 hour')",
        "surface": "hb_file_activity",
        "description": "Identify if hosts accessing personal AI were touching sensitive files within a one-hour window of the access.",
        "expected_signal": "File touch events on sensitive documents occurring near the time of AI platform access."
      },
      "parents": [
        {
          "id": "gate-on-lead",
          "branch": "on_supports"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "enrichment",
        "label": "Correlate sensitive document access",
        "reads": [
          "device_hostname",
          "actor_user_name",
          "file_name",
          "file_path",
          "time"
        ],
        "source": "hb_file_activity",
        "target": "endpoint",
        "content": "SELECT device_hostname, actor_user_name, file_name, file_path, time FROM hb_file_activity WHERE (LOWER(file_name) LIKE '%contract%' OR LOWER(file_name) LIKE '%confidential%' OR LOWER(file_name) LIKE '%salary%') AND ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND time BETWEEN datetime('{{ai_access_timestamp}}', '-1 hour') AND datetime('{{ai_access_timestamp}}', '+1 hour')",
        "silence": "not_evidence_of_absence",
        "expected": "File touch events on sensitive documents occurring near the time of AI platform access.",
        "verified": "dry-run",
        "verified_at": "2026-10-04"
      }
    },
    {
      "id": "network-data-volume",
      "type": "query",
      "label": "Verify data exfiltration volume",
      "config": {
        "dsl": "sqlite",
        "role": "enrichment",
        "source": "network",
        "content": "SELECT device_hostname, dst_endpoint_ip, SUM(traffic_bytes) as total_bytes, MIN(time) as start_time FROM hb_network_connection WHERE (instr(',' || '{{personal_ai_ips}}' || ',', ',' || dst_endpoint_ip || ',') > 0) AND ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, dst_endpoint_ip",
        "surface": "hb_network_connection",
        "description": "Examine network traffic to AI IPs to find large data uploads that confirm document exfiltration.",
        "expected_signal": "High total_bytes counts to known AI infrastructure indicating potential file uploads."
      },
      "parents": [
        {
          "id": "gate-on-lead",
          "branch": "on_supports"
        }
      ],
      "primitive_config": {
        "dsl": "sqlite",
        "role": "enrichment",
        "label": "Verify data exfiltration volume",
        "reads": [
          "device_hostname",
          "dst_endpoint_ip",
          "traffic_bytes",
          "time"
        ],
        "source": "hb_network_connection",
        "target": "network",
        "content": "SELECT device_hostname, dst_endpoint_ip, SUM(traffic_bytes) as total_bytes, MIN(time) as start_time FROM hb_network_connection WHERE (instr(',' || '{{personal_ai_ips}}' || ',', ',' || dst_endpoint_ip || ',') > 0) AND ('{{scope_hosts}}' = '' OR instr(',' || '{{scope_hosts}}' || ',', ',' || device_hostname || ',') > 0) AND time >= datetime('now', '-{{lookback_days}} days') GROUP BY device_hostname, dst_endpoint_ip",
        "silence": "not_evidence_of_absence",
        "expected": "High total_bytes counts to known AI infrastructure indicating potential file uploads.",
        "verified": "dry-run",
        "verified_at": "2026-10-04"
      }
    },
    {
      "id": "triage-ai-risk",
      "type": "analytic",
      "label": "Triage AI risks",
      "config": {
        "cite": "required",
        "tools": [
          "endpoint",
          "network",
          "web"
        ],
        "context": [
          "evaluate-ai-lead",
          "sensitive-file-activity",
          "network-data-volume",
          "external-ai-exposure"
        ],
        "objective": "Determine if any host shows sensitive file access immediately preceding personal AI usage with significant traffic volume.",
        "description": "Assess the combination of personal AI usage, sensitive file access, and traffic volume.",
        "max_iterations": 6,
        "expected_signal": "A risk assessment per host, prioritizing cases where file access and high volume align.",
        "success_criteria": "A verdict of malicious | suspicious | benign per host."
      },
      "parents": [
        {
          "id": "sensitive-file-activity",
          "kind": "merge"
        },
        {
          "id": "network-data-volume",
          "kind": "merge"
        }
      ]
    },
    {
      "id": "route-on-verdict",
      "type": "checkpoint",
      "label": "Route on risk verdict",
      "config": {
        "fuzzy": true,
        "judge": "hunter",
        "question": "the triage-ai-risk verdict identifies malicious data exfiltration for at least one host",
        "condition": "the triage-ai-risk verdict identifies malicious data exfiltration for at least one host",
        "blind_spot": "no-clipboard-logs",
        "confidence": "high",
        "description": "Direct exfiltration findings to isolation and others to remediation.",
        "checkpoint_type": "mandatory"
      },
      "parents": [
        {
          "id": "triage-ai-risk"
        }
      ]
    },
    {
      "id": "isolate-suspect-host",
      "type": "action",
      "label": "Isolate suspect host",
      "config": {
        "target": "endpoint",
        "description": "Halt suspected exfiltration and preserve browser evidence.",
        "instructions": "Isolate the host immediately. Preserve browser cache and local history to reconstruct AI prompts.",
        "action_approval": "required"
      },
      "parents": [
        {
          "id": "route-on-verdict",
          "branch": "on_supports"
        }
      ]
    },
    {
      "id": "analyst-remediation",
      "type": "task",
      "label": "Analyst remediation",
      "config": {
        "assignee": "analyst",
        "description": "Harden exposed assets and verify policy compliance.",
        "instructions": "Check exposed assets from Step 1 for configuration weaknesses. Verify if users in Step 2 violated data handling policies."
      },
      "parents": [
        {
          "id": "gate-on-lead",
          "branch": "default"
        },
        {
          "id": "gate-on-lead",
          "branch": "on_unavailable"
        },
        {
          "id": "route-on-verdict",
          "branch": "default"
        },
        {
          "id": "route-on-verdict",
          "branch": "on_unavailable"
        },
        {
          "id": "route-on-verdict",
          "branch": "on_refutes"
        },
        {
          "id": "isolate-suspect-host"
        }
      ]
    },
    {
      "id": "hunt-close-out",
      "type": "task",
      "label": "Hunt close-out",
      "config": {
        "assignee": "analyst",
        "description": "Finalize documentation and findings.",
        "instructions": "Document the number of identified shadow AI users. Provide the list of exposed assets to the firewall team."
      },
      "parents": [
        {
          "id": "gate-on-lead",
          "branch": "on_refutes"
        },
        {
          "id": "analyst-remediation"
        }
      ]
    }
  ]
}