/** * Security — Input validation, path traversal prevention, and prompt injection guards * * This module centralizes security checks for GSD tooling. Because GSD generates * markdown files that become LLM system prompts (agent instructions, workflow state, * phase plans), any user-controlled text that flows into these files is a potential * indirect prompt injection vector. * * Threat model: * 1. Path traversal: user-supplied file paths escape the project directory * 2. Prompt injection: malicious text in arguments/PRDs embeds LLM instructions * 3. Shell metacharacter injection: user text interpreted by shell * 4. JSON injection: malformed JSON crashes or corrupts state * 5. Regex DoS: crafted input causes catastrophic backtracking * * ADR-457 build-at-publish: the hand-written bin/lib/security.cjs collapsed * to a TypeScript source of truth. Behaviour is preserved byte-for-behaviour * from the prior hand-written .cjs; only types are added. */ import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; // ─── Path Traversal Prevention ────────────────────────────────────────────── /** * Validate that a file path resolves within an allowed base directory. * Prevents path traversal attacks via ../ sequences, symlinks, or absolute paths. */ export function validatePath(filePath: unknown, baseDir: unknown, opts: { allowAbsolute?: boolean } = {}): { safe: boolean; resolved: string; error?: string } { if (!filePath || typeof filePath !== 'string') { return { safe: false, resolved: '', error: 'Empty or invalid file path' }; } if (!baseDir || typeof baseDir !== 'string') { return { safe: false, resolved: '', error: 'Empty or invalid base directory' }; } if (filePath.includes('\0')) { return { safe: false, resolved: '', error: 'Path contains null bytes' }; } let resolvedBase: string; try { resolvedBase = fs.realpathSync(path.resolve(baseDir)); } catch { resolvedBase = path.resolve(baseDir); } let resolvedPath: string; if (path.isAbsolute(filePath)) { if (!opts.allowAbsolute) { return { safe: false, resolved: '', error: 'Absolute paths not allowed' }; } resolvedPath = path.resolve(filePath); } else { resolvedPath = path.resolve(baseDir, filePath); } try { resolvedPath = fs.realpathSync(resolvedPath); } catch { const parentDir = path.dirname(resolvedPath); try { const realParent = fs.realpathSync(parentDir); resolvedPath = path.join(realParent, path.basename(resolvedPath)); } catch { // Parent doesn't exist either — keep the resolved path as-is } } const normalizedBase = resolvedBase + path.sep; const normalizedPath = resolvedPath + path.sep; if (resolvedPath !== resolvedBase && !normalizedPath.startsWith(normalizedBase)) { return { safe: false, resolved: resolvedPath, error: `Path escapes allowed directory: ${resolvedPath} is outside ${resolvedBase}`, }; } return { safe: true, resolved: resolvedPath }; } /** * Load the opt-in trusted global roots allowlist from config. * * Reads `config.agent_skills_security.trusted_global_roots` (an array of * path strings). Each entry is canonicalized via realpathSync: non-strings * are dropped, leading `~/` is expanded to `os.homedir()`, entries that are * not absolute after expansion are dropped (project-relative paths are * rejected as a security boundary), and entries that do not exist on disk are * dropped (a non-existent root is not trustworthy). The canonical realpath is * used for all subsequent checks and as the stored value — this closes the * case-insensitive bypass on macOS APFS (`/users/alice` vs `/Users/alice`) * and ensures trust doesn't drift across re-invocations if a root is * re-created at a different target. Results are de-duplicated by canonical path. */ export function loadTrustedGlobalRoots(config: unknown): string[] { const roots = (config as Record | null | undefined) ?.['agent_skills_security'] as Record | undefined; const raw = roots?.['trusted_global_roots']; if (!Array.isArray(raw)) return []; // Compute canonical homedir once for case-insensitive-safe comparison. let realHome: string; try { realHome = fs.realpathSync(os.homedir()); } catch { realHome = os.homedir(); } const seen = new Set(); const result: string[] = []; for (const entry of raw) { if (typeof entry !== 'string') continue; let expanded: string; if (entry === '~') { expanded = os.homedir(); } else if (entry.startsWith('~/')) { expanded = path.join(os.homedir(), entry.slice(2)); } else { expanded = entry; } if (!path.isAbsolute(expanded)) continue; // reject project-relative // Canonicalize: resolve symlinks and normalise case. If the path doesn't // exist or can't be read, skip it — a non-existent root is not trustworthy. let real: string; try { real = fs.realpathSync(expanded); } catch { continue; // non-existent or unreadable — skip } // Reject dangerously broad roots: filesystem root (e.g. '/' or 'C:\' or UNC '\\server\share'). // Normalize both sides by stripping trailing path separators before comparing so that // Windows UNC shares (where path.parse().root includes a trailing separator) are caught. const stripTrailingSep = (p: string): string => p.replace(/[\\/]+$/, ''); if (stripTrailingSep(path.parse(real).root) === stripTrailingSep(real)) continue; // Reject homedir itself (canonical compare closes case-insensitive bypass). // Apply stripTrailingSep for robustness on platforms where realpathSync may // or may not include a trailing separator on the homedir path. if (stripTrailingSep(real) === stripTrailingSep(realHome)) continue; if (seen.has(real)) continue; seen.add(real); result.push(real); } return result; } /** * Validate a file path and throw on traversal attempt. * Convenience wrapper around validatePath for use in CLI commands. */ export function requireSafePath(filePath: unknown, baseDir: unknown, label: string | null | undefined, opts: { allowAbsolute?: boolean } = {}): string { const result = validatePath(filePath, baseDir, opts); if (!result.safe) { throw new Error(`${label || 'Path'} validation failed: ${result.error}`); } return result.resolved; } // ─── Prompt Injection Detection ──────────────────────────────────────────────────── /** * Patterns that indicate prompt injection attempts in user-supplied text. * These patterns catch common indirect prompt injection techniques where * an attacker embeds LLM instructions in text that will be read by an agent. * * Note: This is defense-in-depth — not a complete solution. The primary defense * is proper input/output boundaries in agent prompts. */ export const INJECTION_PATTERNS: RegExp[] = [ // Direct instruction override attempts /ignore\s+(all\s+)?previous\s+instructions/i, /ignore\s+(all\s+)?above\s+instructions/i, /disregard\s+(all\s+)?previous/i, /forget\s+(all\s+)?(your\s+)?instructions/i, /override\s+(system|previous)\s+(prompt|instructions)/i, // Role/identity manipulation /you\s+are\s+now\s+(?:a|an|the)\s+/i, /act\s+as\s+(?:a|an|the)\s+(?!plan|phase|wave)/i, /pretend\s+(?:you(?:'re| are)\s+|to\s+be\s+)/i, /from\s+now\s+on,?\s+you\s+(?:are|will|should|must)/i, // System prompt extraction /(?:print|output|reveal|show|display|repeat)\s+(?:your\s+)?(?:system\s+)?(?:prompt|instructions)/i, /what\s+(?:are|is)\s+your\s+(?:system\s+)?(?:prompt|instructions)/i, // Hidden instruction markers (XML/HTML tags that mimic system messages) // Note: is excluded — GSD uses it as legitimate prompt structure // Requires > to close the tag (not just whitespace) to avoid matching generic types like Promise /<\/?(?:system|assistant|human)>/i, /\[SYSTEM\]/i, /\[\/?(INST)\]/i, /<<\s*SYS\s*>>/i, // Exfiltration attempts /(?:send|post|fetch|curl|wget)\s+(?:to|from)\s+https?:\/\//i, /(?:base64|btoa|encode)\s+(?:and\s+)?(?:send|exfiltrate|output)/i, // Tool manipulation /(?:run|execute|call|invoke)\s+(?:the\s+)?(?:bash|shell|exec|spawn)\s+(?:tool|command)/i, ]; // Explicit safe-list for data: MIME types that are benign in link targets. // Note: image/svg+xml is intentionally NOT in this list (SVG can host