diff --git a/README.md b/README.md index 6e0a7a3..ec934bb 100644 --- a/README.md +++ b/README.md @@ -88,6 +88,7 @@ codeguard init codeguard validate -config codeguard.yaml codeguard doctor -config codeguard.yaml codeguard scan -config codeguard.yaml +codeguard scan -config codeguard.yaml -memprofile /tmp/codeguard.heap.pprof codeguard scan -config codeguard.yaml -folder ./internal/codeguard codeguard scan -folder ./internal/codeguard -profile startup codeguard scan -folder ./internal/codeguard -profile startup -set checks.quality=true -set output.format=json @@ -108,6 +109,8 @@ By default, `codeguard` looks for `codeguard.yaml`, `codeguard.yml`, or `codegua If you point `-config` at a directory such as `.codeguard`, `codeguard` will look inside it for `codeguard.*` or `config.*` files. +Long-running scans report completed sections and a 30-second heap/GOMEMLIMIT heartbeat on stderr, leaving JSON, SARIF, GitHub, and CycloneDX stdout machine-readable. Pass `-memprofile ` to keep a rolling Go heap profile (written at startup, every heartbeat, and completion), then inspect it with `go tool pprof `. Full scans skip dependency and generated trees named `node_modules`, `vendor`, and `cdk.out` at any depth; use `exclude` for repository-specific generated paths. + Use `codeguard scan -folder ` to scan only one folder. `-path ` is accepted as an alias. If no config file exists and you did not pass `-config`, folder scans use CodeGuard's built-in default config; add `-profile startup`, `-profile strict`, `-profile enterprise`, or `-profile ai-safe` to choose a default profile. Use repeatable `-set key=value` flags to override config values from the terminal without writing a temporary config file. Keys use dotted YAML paths and are validated against the typed CodeGuard config; mistyped fields fail before the scan runs. String lists accept comma-separated values or a JSON string array. diff --git a/docs/features.md b/docs/features.md index b9e8350..30ae1c0 100644 --- a/docs/features.md +++ b/docs/features.md @@ -107,6 +107,24 @@ This page lists the current `codeguard` feature surface and the main config entr See [Production rollout](production.md) for configuration, safe CI usage, review workflow, and exit-code behavior. +## Repository traversal + +Full scans always prune directories named `node_modules` and `cdk.out` at any +depth. Dependency manifests and lockfiles remain in scope for supply-chain, +license, vulnerability, and lockfile-integrity checks; installed and generated +source is not treated as first-party code. + +Directories named `vendor` are also pruned by default. Repositories that commit +or patch vendored source can opt in explicitly while retaining the normal +per-file, corpus-byte, AST, and file-count limits: + +```yaml +scan_vendored_source: true +``` + +Configured `exclude` patterns still take precedence, so an explicitly excluded +vendor subtree remains excluded when vendored-source scanning is enabled. + ## External report ingestion CodeGuard can import findings from scanners that have already run. It does not diff --git a/internal/cli/commands.go b/internal/cli/commands.go index 5b0f62a..cace7fa 100644 --- a/internal/cli/commands.go +++ b/internal/cli/commands.go @@ -79,6 +79,7 @@ func runScan(args []string, stdin io.Reader, stdout io.Writer, stderr io.Writer) pathAlias := fs.String("path", "", "alias for -folder") enableAI := fs.Bool("ai", false, "enable optional AI-assisted analysis") includeSuppressed := fs.Bool("include-suppressed", false, "include individual suppressed findings in JSON output") + heapProfile := fs.String("memprofile", "", "optional path for rolling Go heap profiles during the scan") interactive := fs.Bool("interactive", false, "prompt for scan inputs in the terminal") if ok, code := parseFlags(fs, args, stderr); !ok { return code @@ -116,7 +117,7 @@ func runScan(args []string, stdin io.Reader, stdout io.Writer, stderr io.Writer) return exitError } - if err := executeScan(stdout, cfg, scanMode, strings.TrimSpace(*inputs.baseRef), targetPath, *enableAI, *includeSuppressed); err != nil { + if err := executeScan(stdout, stderr, cfg, scanMode, strings.TrimSpace(*inputs.baseRef), targetPath, *enableAI, *includeSuppressed, *heapProfile); err != nil { _, _ = fmt.Fprintf(stderr, "scan failed: %v\n", err) return exitError } diff --git a/internal/cli/commands_scan_helpers.go b/internal/cli/commands_scan_helpers.go index 578ee4e..341d6f6 100644 --- a/internal/cli/commands_scan_helpers.go +++ b/internal/cli/commands_scan_helpers.go @@ -3,10 +3,12 @@ package cli import ( "bufio" "context" + "errors" "flag" "fmt" "io" "strings" + "time" service "github.com/devr-tools/codeguard/pkg/codeguard" ) @@ -58,16 +60,20 @@ func parseScanMode(mode string) (service.ScanMode, error) { return scanMode, nil } -func executeScan(stdout io.Writer, cfg service.Config, scanMode service.ScanMode, baseRef string, targetPath string, enableAI bool, includeSuppressed bool) error { - report, err := service.RunWithOptions(context.Background(), cfg, service.ScanOptions{ +func executeScan(stdout io.Writer, stderr io.Writer, cfg service.Config, scanMode service.ScanMode, baseRef string, targetPath string, enableAI bool, includeSuppressed bool, heapProfilePath string) error { + monitor := newScanMonitor(stderr, time.Now()) + stopMonitoring := startScanMonitoring(monitor, strings.TrimSpace(heapProfilePath), scanHeartbeatInterval) + report, scanErr := service.RunWithOptions(context.Background(), cfg, service.ScanOptions{ Mode: scanMode, BaseRef: baseRef, TargetPath: targetPath, EnableAI: enableAI, IncludeSuppressed: includeSuppressed, + OnSectionComplete: monitor.writeSectionComplete, }) - if err != nil { - return err + monitorErr := stopMonitoring() + if scanErr != nil { + return scanErr } if err := writeScanMetadata(stdout, cfg.Output.Format, scanMode, baseRef); err != nil { return err @@ -76,10 +82,11 @@ func executeScan(stdout io.Writer, cfg service.Config, scanMode service.ScanMode return fmt.Errorf("write report: %w", err) } writePerformanceUpgradeHint(stdout, cfg) + var findingsErr error if report.Summary.FailedSections > 0 { - return fmt.Errorf("one or more sections failed") + findingsErr = fmt.Errorf("one or more sections failed") } - return nil + return errors.Join(findingsErr, monitorErr) } func scanTargetPath(folderPath string, pathAlias string) (string, error) { diff --git a/internal/cli/scan_diagnostics.go b/internal/cli/scan_diagnostics.go new file mode 100644 index 0000000..dfdccb8 --- /dev/null +++ b/internal/cli/scan_diagnostics.go @@ -0,0 +1,118 @@ +package cli + +import ( + "fmt" + "io" + "os" + "path/filepath" + "runtime" + "runtime/debug" + "runtime/pprof" + "sync" + "time" + + service "github.com/devr-tools/codeguard/pkg/codeguard" +) + +const scanHeartbeatInterval = 30 * time.Second + +type scanMonitor struct { + mu sync.Mutex + writer io.Writer + started time.Time +} + +func newScanMonitor(writer io.Writer, started time.Time) *scanMonitor { + return &scanMonitor{writer: writer, started: started} +} + +func (monitor *scanMonitor) writeStarted() { + monitor.write("scan started\n") +} + +func (monitor *scanMonitor) writeHeartbeat(now time.Time, heapBytes uint64, memoryLimit int64) { + elapsed := now.Sub(monitor.started).Round(time.Second) + monitor.write("scan in progress: elapsed=%s heap=%d MiB memory_limit=%s\n", + elapsed, heapBytes/(1<<20), formatMemoryLimit(memoryLimit)) +} + +func (monitor *scanMonitor) writeSectionComplete(section service.SectionResult) { + monitor.write("completed %s: %s (%d findings)\n", section.Name, section.Status, len(section.Findings)) +} + +func (monitor *scanMonitor) write(format string, args ...any) { + if monitor == nil || monitor.writer == nil { + return + } + monitor.mu.Lock() + defer monitor.mu.Unlock() + _, _ = fmt.Fprintf(monitor.writer, format, args...) +} + +func formatMemoryLimit(limit int64) string { + if limit < 0 || limit >= 1<<62 { + return "unlimited" + } + return fmt.Sprintf("%d MiB", limit/(1<<20)) +} + +func startScanMonitoring(monitor *scanMonitor, heapProfilePath string, interval time.Duration) func() error { + monitor.writeStarted() + if heapProfilePath != "" { + if err := writeHeapProfile(heapProfilePath); err != nil { + monitor.write("heap profile update failed: %v\n", err) + } + } + + stop := make(chan struct{}) + done := make(chan struct{}) + go func() { + defer close(done) + ticker := time.NewTicker(interval) + defer ticker.Stop() + for { + select { + case now := <-ticker.C: + var stats runtime.MemStats + runtime.ReadMemStats(&stats) + monitor.writeHeartbeat(now, stats.HeapAlloc, debug.SetMemoryLimit(-1)) + if heapProfilePath != "" { + if err := writeHeapProfile(heapProfilePath); err != nil { + monitor.write("heap profile update failed: %v\n", err) + } + } + case <-stop: + return + } + } + }() + + return func() error { + close(stop) + <-done + if heapProfilePath != "" { + if err := writeHeapProfile(heapProfilePath); err != nil { + return fmt.Errorf("write final heap profile: %w", err) + } + } + return nil + } +} + +func writeHeapProfile(path string) error { + dir := filepath.Dir(path) + tmp, err := os.CreateTemp(dir, ".codeguard-heap-*.pprof") + if err != nil { + return err + } + tmpPath := tmp.Name() + defer func() { _ = os.Remove(tmpPath) }() + if err := pprof.WriteHeapProfile(tmp); err != nil { + _ = tmp.Close() + return err + } + if err := tmp.Close(); err != nil { + return err + } + return os.Rename(tmpPath, path) +} diff --git a/internal/cli/scan_diagnostics_test.go b/internal/cli/scan_diagnostics_test.go new file mode 100644 index 0000000..cedd82c --- /dev/null +++ b/internal/cli/scan_diagnostics_test.go @@ -0,0 +1,122 @@ +package cli + +import ( + "bytes" + "os" + "path/filepath" + "strings" + "testing" + "time" + + service "github.com/devr-tools/codeguard/pkg/codeguard" +) + +func TestScanMonitorReportsProgressAndMemoryWithoutPollutingReportOutput(t *testing.T) { + var diagnostics bytes.Buffer + monitor := newScanMonitor(&diagnostics, time.Date(2026, 9, 1, 12, 0, 0, 0, time.UTC)) + monitor.writeStarted() + monitor.writeHeartbeat(time.Date(2026, 9, 1, 12, 1, 30, 0, time.UTC), 384<<20, 8<<30) + monitor.writeSectionComplete(service.SectionResult{Name: "Code Quality", Status: service.StatusPass}) + + got := diagnostics.String() + for _, want := range []string{ + "scan started", + "scan in progress: elapsed=1m30s heap=384 MiB memory_limit=8192 MiB", + "completed Code Quality: pass", + } { + if !strings.Contains(got, want) { + t.Fatalf("diagnostics %q do not contain %q", got, want) + } + } +} + +func TestWriteHeapProfileCreatesUsableProfile(t *testing.T) { + path := filepath.Join(t.TempDir(), "scan.heap.pprof") + if err := writeHeapProfile(path); err != nil { + t.Fatalf("write heap profile: %v", err) + } + data, err := os.ReadFile(path) //nolint:gosec // test-owned path under t.TempDir + if err != nil { + t.Fatalf("read heap profile: %v", err) + } + if len(data) < 2 || data[0] != 0x1f || data[1] != 0x8b { + t.Fatalf("heap profile does not have gzip header: %x", data[:min(len(data), 8)]) + } +} + +func TestRunScanStreamsProgressAndWritesRequestedHeapProfile(t *testing.T) { + root := t.TempDir() + if err := os.WriteFile(filepath.Join(root, "main.go"), []byte("package main\n\nfunc main() {}\n"), 0o600); err != nil { + t.Fatalf("write source: %v", err) + } + contextDisabled := false + configPath := filepath.Join(root, "codeguard.json") + cfg := service.Config{ + Name: "diagnostic-scan", + Targets: []service.TargetConfig{{Name: "repo", Path: root, Language: "go"}}, + Checks: service.CheckConfig{ + Quality: true, + Context: &contextDisabled, + }, + Output: service.OutputConfig{Format: "github"}, + } + if err := service.WriteConfigFile(configPath, cfg); err != nil { + t.Fatalf("write config: %v", err) + } + + profilePath := filepath.Join(root, "scan.heap.pprof") + var stdout bytes.Buffer + var stderr bytes.Buffer + exitCode := Run([]string{"scan", "-config", configPath, "-format", "github", "-memprofile", profilePath}, + strings.NewReader(""), &stdout, &stderr) + if exitCode != 0 { + t.Fatalf("scan exit code = %d, want 0; stderr: %s", exitCode, stderr.String()) + } + if !strings.Contains(stderr.String(), "scan started") || !strings.Contains(stderr.String(), "completed Code Quality") { + t.Fatalf("stderr does not contain streamed progress: %q", stderr.String()) + } + if strings.Contains(stdout.String(), "scan started") { + t.Fatalf("machine-readable stdout was polluted by progress: %q", stdout.String()) + } + if info, err := os.Stat(profilePath); err != nil { + t.Fatalf("stat heap profile: %v", err) + } else if info.Size() == 0 { + t.Fatal("heap profile is empty") + } +} + +func TestRunScanStillWritesReportWhenHeapProfileCannotBeWritten(t *testing.T) { + root := t.TempDir() + if err := os.WriteFile(filepath.Join(root, "main.go"), []byte("package main\n\nfunc main() {}\n"), 0o600); err != nil { + t.Fatalf("write source: %v", err) + } + contextDisabled := false + configPath := filepath.Join(root, "codeguard.json") + cfg := service.Config{ + Name: "profile-failure-scan", + Targets: []service.TargetConfig{{Name: "repo", Path: root, Language: "go"}}, + Checks: service.CheckConfig{ + Quality: true, + Context: &contextDisabled, + }, + Output: service.OutputConfig{Format: "github"}, + } + if err := service.WriteConfigFile(configPath, cfg); err != nil { + t.Fatalf("write config: %v", err) + } + + profilePath := filepath.Join(root, "missing", "scan.heap.pprof") + var stdout bytes.Buffer + var stderr bytes.Buffer + exitCode := Run([]string{"scan", "-config", configPath, "-format", "github", "-memprofile", profilePath}, + strings.NewReader(""), &stdout, &stderr) + if exitCode == 0 { + t.Fatal("scan with unwritable requested heap profile returned success") + } + if !strings.Contains(stdout.String(), "::notice title=CodeGuard") { + t.Fatalf("completed scan report was not written: stdout=%q stderr=%q", stdout.String(), stderr.String()) + } + if !strings.Contains(stderr.String(), "heap profile") { + t.Fatalf("heap profile failure was not diagnosed: %q", stderr.String()) + } +} diff --git a/internal/codeguard/checks/quality/quality.go b/internal/codeguard/checks/quality/quality.go index f604b9d..b0b00d6 100644 --- a/internal/codeguard/checks/quality/quality.go +++ b/internal/codeguard/checks/quality/quality.go @@ -16,18 +16,22 @@ func Run(ctx context.Context, env support.Context) core.SectionResult { func runQualitySection(ctx context.Context, env support.Context) core.SectionResult { var unresolved []unresolvedMutationEvidence + var diagnostics []core.Diagnostic findings := support.CollectTargetFindings(ctx, env, func(ctx context.Context, env support.Context, target core.TargetConfig) []core.Finding { analysis := qualityTargetAnalysis(ctx, env, target) unresolved = append(unresolved, analysis.unresolved...) + diagnostics = append(diagnostics, analysis.diagnostics...) return analysis.findings }) findings = append(findings, provenancePolicyFindings(env, findings)...) //nolint:contextcheck // git helpers use a contained timeout; deeper ctx threading is a tracked follow-up - return env.FinalizeSectionWithDiagnostics("quality", "Code Quality", findings, unresolvedMutationDiagnostics(unresolved)) + diagnostics = append(diagnostics, unresolvedMutationDiagnostics(unresolved)...) + return env.FinalizeSectionWithDiagnostics("quality", "Code Quality", findings, diagnostics) } type qualityTargetScan struct { - findings []core.Finding - unresolved []unresolvedMutationEvidence + findings []core.Finding + unresolved []unresolvedMutationEvidence + diagnostics []core.Diagnostic } func qualityTargetAnalysis(ctx context.Context, env support.Context, target core.TargetConfig) qualityTargetScan { @@ -39,7 +43,8 @@ func qualityTargetAnalysis(ctx context.Context, env support.Context, target core findings = append(findings, rustToolchainDeadCodeFindings(ctx, env, target)...) findings = append(findings, cppToolchainDeadCodeFindings(env, target)...) findings = append(findings, pythonToolchainDeadCodeFindings(env, target)...) - findings = append(findings, cloneFindingsForTarget(env, target)...) + cloneAnalysis := cloneFindingsForTarget(env, target) + findings = append(findings, cloneAnalysis.findings...) findings = append(findings, aiTargetFindings(env, target)...) findings = append(findings, semanticFindings(ctx, env, target)...) findings = append(findings, commandFindings(ctx, env, target)...) @@ -50,7 +55,7 @@ func qualityTargetAnalysis(ctx context.Context, env support.Context, target core } maybePutAISlopArtifact(env, target, findings) findings = append(findings, changeRiskFindings(env, target, findings)...) //nolint:contextcheck // git helpers use a contained timeout; deeper ctx threading is a tracked follow-up - return qualityTargetScan{findings: findings, unresolved: language.unresolved} + return qualityTargetScan{findings: findings, unresolved: language.unresolved, diagnostics: cloneAnalysis.diagnostics} } func unresolvedMutationDiagnostics(unresolved []unresolvedMutationEvidence) []core.Diagnostic { diff --git a/internal/codeguard/checks/quality/quality_ai_corpus.go b/internal/codeguard/checks/quality/quality_ai_corpus.go index c433351..19de2e7 100644 --- a/internal/codeguard/checks/quality/quality_ai_corpus.go +++ b/internal/codeguard/checks/quality/quality_ai_corpus.go @@ -29,7 +29,10 @@ func listAITargetFiles(env support.Context, target core.TargetConfig, include fu } return files } - files, err := runnersupport.WalkFiles(target.Path, env.Config.Exclude, include) + files, err := runnersupport.WalkFilesWithOptions(target.Path, env.Config.Exclude, runnersupport.FileWalkOptions{ + LogicalPath: target.LogicalPath, + ScanVendoredSource: env.Config.ScanVendoredSource, + }, include) if err != nil { return nil } diff --git a/internal/codeguard/checks/quality/quality_clone.go b/internal/codeguard/checks/quality/quality_clone.go index 30a73d6..621217f 100644 --- a/internal/codeguard/checks/quality/quality_clone.go +++ b/internal/codeguard/checks/quality/quality_clone.go @@ -4,23 +4,52 @@ import ( "fmt" "path/filepath" "sort" + "strconv" "github.com/devr-tools/codeguard/internal/codeguard/checks/support" "github.com/devr-tools/codeguard/internal/codeguard/core" ) -func cloneFindingsForTarget(env support.Context, target core.TargetConfig) []core.Finding { +type cloneAnalysisResult struct { + findings []core.Finding + diagnostics []core.Diagnostic +} + +type cloneAnalysisBudget struct { + maxSourceBytes int + maxTokens int + maxWindows int +} + +var defaultCloneAnalysisBudget = cloneAnalysisBudget{ + maxSourceBytes: 64 << 20, + maxTokens: 2_000_000, + maxWindows: 1_000_000, +} + +func cloneFindingsForTarget(env support.Context, target core.TargetConfig) cloneAnalysisResult { + return cloneFindingsForTargetWithBudget(env, target, defaultCloneAnalysisBudget) +} + +func cloneFindingsForTargetWithBudget(env support.Context, target core.TargetConfig, budget cloneAnalysisBudget) cloneAnalysisResult { threshold := env.Config.Checks.QualityRules.CloneTokenThreshold if threshold <= 0 { - return nil + return cloneAnalysisResult{} } - docs := cloneDocumentsForTarget(env, target) + docs, truncated := cloneDocumentsForTargetWithBudget(env, target, budget) + result := cloneAnalysisResult{} if len(docs) < 2 { - return nil + if truncated { + result.diagnostics = []core.Diagnostic{cloneBudgetDiagnostic(target, budget)} + } + return result } - candidates := detectCloneCandidates(docs, threshold) + candidates, windowBudgetReached := detectCloneCandidates(docs, threshold, budget.maxWindows) + if truncated || windowBudgetReached { + result.diagnostics = []core.Diagnostic{cloneBudgetDiagnostic(target, budget)} + } findings := make([]core.Finding, 0, len(candidates)*2) for _, candidate := range candidates { left := docs[candidate.LeftDoc] @@ -44,12 +73,16 @@ func cloneFindingsForTarget(env support.Context, target core.TargetConfig) []cor ) findings = append(findings, warnFinding(env, "quality.duplicate-code", right.Path, rightLine, 1, message)) } - return findings + result.findings = findings + return result } -func cloneDocumentsForTarget(env support.Context, target core.TargetConfig) []cloneDocument { +func cloneDocumentsForTargetWithBudget(env support.Context, target core.TargetConfig, budget cloneAnalysisBudget) ([]cloneDocument, bool) { docs := make([]cloneDocument, 0) include := cloneIncludeForLanguage(target.Language) + totalSourceBytes := 0 + totalTokens := 0 + truncated := false // Clone detection builds cross-file state (the document list) rather than // per-file findings, so it must visit every file directly. Routing it // through the per-file findings cache would skip the tokenizer on a cache @@ -57,19 +90,46 @@ func cloneDocumentsForTarget(env support.Context, target core.TargetConfig) []cl env.VisitTargetFiles(target, func(rel string) bool { return include(rel) && !cloneExcludedPath(target.Language, rel) }, func(file string, data []byte) { - tokens := tokenizeNormalizedCloneText(string(data)) + if totalTokens >= budget.maxTokens || totalSourceBytes+len(data) > budget.maxSourceBytes { + truncated = true + return + } + tokens, tokenLimitReached := tokenizeNormalizedCloneTextBounded(string(data), budget.maxTokens-totalTokens) + if tokenLimitReached { + truncated = true + } if len(tokens) > 0 { docs = append(docs, cloneDocument{Path: file, Tokens: tokens}) + totalSourceBytes += len(data) + totalTokens += len(tokens) } }) - return docs + return docs, truncated +} + +func cloneBudgetDiagnostic(target core.TargetConfig, budget cloneAnalysisBudget) core.Diagnostic { + return core.Diagnostic{ + ID: "quality.duplicate-code-budget", + Level: "info", + Kind: "analysis", + Message: fmt.Sprintf( + "duplicate-code analysis for target %q reached its bounded corpus budget; add repository-specific generated paths to exclude if needed", + target.Name, + ), + Metadata: map[string]string{ + "target": target.Name, + "max_source_bytes": strconv.Itoa(budget.maxSourceBytes), + "max_tokens": strconv.Itoa(budget.maxTokens), + "max_windows": strconv.Itoa(budget.maxWindows), + }, + } } -func detectCloneCandidates(docs []cloneDocument, threshold int) []cloneCandidate { - index := cloneWindowIndex(docs, threshold) +func detectCloneCandidates(docs []cloneDocument, threshold int, maxWindows int) ([]cloneCandidate, bool) { + index, truncated := cloneWindowIndexBounded(docs, threshold, maxWindows) candidates := collectCloneCandidates(index, docs, threshold) sortCloneCandidates(candidates, docs) - return candidates + return candidates, truncated } // cloneWindowMultiplier is the odd multiplier for the polynomial rolling @@ -95,7 +155,7 @@ const ( // normalized tokens), and unequal windows that collide are discarded by the // token-by-token verification in sharedCloneLength, so the resulting clone // candidates are identical to the old per-window byte hashing. -func cloneWindowIndex(docs []cloneDocument, threshold int) cloneIndex { +func cloneWindowIndexBounded(docs []cloneDocument, threshold int, maxOccurrences int) (cloneIndex, bool) { index := make(cloneIndex) hasWindow := false for _, doc := range docs { @@ -105,7 +165,10 @@ func cloneWindowIndex(docs []cloneDocument, threshold int) cloneIndex { } } if !hasWindow { - return index + return index, false + } + if maxOccurrences <= 0 { + return index, true } // top = multiplier^(threshold-1), the weight of the token leaving the // window on each slide. @@ -113,6 +176,7 @@ func cloneWindowIndex(docs []cloneDocument, threshold int) cloneIndex { for i := 0; i < threshold-1; i++ { top *= cloneWindowMultiplier } + occurrences := 0 for docIdx, doc := range docs { if len(doc.Tokens) < threshold { continue @@ -121,13 +185,21 @@ func cloneWindowIndex(docs []cloneDocument, threshold int) cloneIndex { for i := 0; i < threshold; i++ { hash = hash*cloneWindowMultiplier + doc.Tokens[i].Hash } + if occurrences >= maxOccurrences { + return index, true + } index[hash] = append(index[hash], cloneOccurrence{DocIndex: docIdx, TokenIndex: 0}) + occurrences++ for tokenIdx := 1; tokenIdx+threshold <= len(doc.Tokens); tokenIdx++ { + if occurrences >= maxOccurrences { + return index, true + } hash = (hash-doc.Tokens[tokenIdx-1].Hash*top)*cloneWindowMultiplier + doc.Tokens[tokenIdx+threshold-1].Hash index[hash] = append(index[hash], cloneOccurrence{DocIndex: docIdx, TokenIndex: tokenIdx}) + occurrences++ } } - return index + return index, false } func collectCloneCandidates(index cloneIndex, docs []cloneDocument, threshold int) []cloneCandidate { diff --git a/internal/codeguard/checks/quality/quality_clone_support.go b/internal/codeguard/checks/quality/quality_clone_support.go index 9b67668..84fed41 100644 --- a/internal/codeguard/checks/quality/quality_clone_support.go +++ b/internal/codeguard/checks/quality/quality_clone_support.go @@ -64,10 +64,18 @@ func cloneIncludeForLanguage(language string) func(string) bool { } } -func tokenizeNormalizedCloneText(source string) []cloneToken { - matches := cloneTokenPattern.FindAllStringIndex(source, -1) +func tokenizeNormalizedCloneTextBounded(source string, maxTokens int) ([]cloneToken, bool) { + matchLimit := maxTokens + if maxTokens >= 0 { + matchLimit++ + } + matches := cloneTokenPattern.FindAllStringIndex(source, matchLimit) if len(matches) == 0 { - return nil + return nil, false + } + truncated := maxTokens >= 0 && len(matches) > maxTokens + if truncated { + matches = matches[:maxTokens] } tokens := make([]cloneToken, 0, len(matches)) line := 1 @@ -82,7 +90,7 @@ func tokenizeNormalizedCloneText(source string) []cloneToken { tokens = append(tokens, cloneToken{Value: value, Hash: cloneTokenHash(value), Line: line}) prev = match[1] } - return tokens + return tokens, truncated } // FNV-1a constants (hash/fnv is not used directly so token hashing can fold diff --git a/internal/codeguard/checks/quality/quality_clone_test.go b/internal/codeguard/checks/quality/quality_clone_test.go index 230bdd2..f406e2c 100644 --- a/internal/codeguard/checks/quality/quality_clone_test.go +++ b/internal/codeguard/checks/quality/quality_clone_test.go @@ -1,6 +1,31 @@ package quality -import "testing" +import ( + "fmt" + "testing" + + checksupport "github.com/devr-tools/codeguard/internal/codeguard/checks/support" + "github.com/devr-tools/codeguard/internal/codeguard/core" +) + +func BenchmarkCloneWindowIndexHighEntropy(b *testing.B) { + for _, tokens := range []int{10_000, 100_000} { + b.Run(fmt.Sprintf("tokens-%d", tokens), func(b *testing.B) { + doc := cloneDocument{Path: "unique.go", Tokens: make([]cloneToken, tokens)} + for i := range doc.Tokens { + doc.Tokens[i] = cloneToken{Value: fmt.Sprintf("token%d", i), Hash: uint64(i + 1)} + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + index, truncated := cloneWindowIndexBounded([]cloneDocument{doc}, 90, defaultCloneAnalysisBudget.maxWindows) + if truncated || len(index) == 0 { + b.Fatalf("unexpected bounded index result: keys=%d truncated=%v", len(index), truncated) + } + } + }) + } +} func TestCollectCloneCandidatesCapsIdenticalDocuments(t *testing.T) { const documentCount = 100 @@ -24,7 +49,59 @@ func TestCloneWindowIndexSkipsMultiplierForOversizedThreshold(t *testing.T) { } threshold := int(^uint(0) >> 1) - if index := cloneWindowIndex(docs, threshold); len(index) != 0 { + if index, _ := cloneWindowIndexBounded(docs, threshold, defaultCloneAnalysisBudget.maxWindows); len(index) != 0 { t.Fatalf("cloneWindowIndex() returned %d windows, want 0", len(index)) } } + +func TestCloneWindowIndexStopsAtOccurrenceBudget(t *testing.T) { + tokens := make([]cloneToken, 10) + for i := range tokens { + tokens[i] = cloneToken{Value: string(rune('a' + i)), Hash: uint64(i + 1)} + } + index, truncated := cloneWindowIndexBounded([]cloneDocument{{Path: "unique.go", Tokens: tokens}}, 1, 5) + if !truncated { + t.Fatal("clone window index did not report the exhausted occurrence budget") + } + occurrences := 0 + for _, bucket := range index { + occurrences += len(bucket) + } + if occurrences != 5 { + t.Fatalf("indexed occurrences = %d, want hard budget 5", occurrences) + } +} + +func TestCloneDocumentsStopAtAggregateTokenBudget(t *testing.T) { + env := checksupport.Context{ + Config: core.Config{Checks: core.CheckConfig{QualityRules: core.QualityRulesConfig{CloneTokenThreshold: 1}}}, + VisitTargetFiles: func(_ core.TargetConfig, _ func(string) bool, visit func(string, []byte)) { + visit("app.ts", []byte("const one = alpha + beta + gamma + delta;")) + visit("more.ts", []byte("const two = epsilon + zeta;")) + }, + } + docs, truncated := cloneDocumentsForTargetWithBudget(env, core.TargetConfig{Language: "typescript"}, cloneAnalysisBudget{ + maxSourceBytes: 1 << 20, + maxTokens: 5, + maxWindows: 5, + }) + if !truncated { + t.Fatal("clone document collection did not report the exhausted token budget") + } + total := 0 + for _, doc := range docs { + total += len(doc.Tokens) + } + if total > 5 { + t.Fatalf("retained clone tokens = %d, budget = 5", total) + } + + analysis := cloneFindingsForTargetWithBudget(env, core.TargetConfig{Name: "repo", Language: "typescript"}, cloneAnalysisBudget{ + maxSourceBytes: 1 << 20, + maxTokens: 5, + maxWindows: 5, + }) + if len(analysis.diagnostics) != 1 || analysis.diagnostics[0].ID != "quality.duplicate-code-budget" { + t.Fatalf("budget diagnostics = %+v, want quality.duplicate-code-budget", analysis.diagnostics) + } +} diff --git a/internal/codeguard/checks/security/security_typescript_bindings.go b/internal/codeguard/checks/security/security_typescript_bindings.go index a51c3ef..aed9e2a 100644 --- a/internal/codeguard/checks/security/security_typescript_bindings.go +++ b/internal/codeguard/checks/security/security_typescript_bindings.go @@ -13,9 +13,46 @@ var ( tsDefaultImportPattern = regexp.MustCompile(`(?m)^\s*import\s+([A-Za-z_$][\w$]*)\s*from\s*["'](?:node:)?%s["']`) tsNamedRequirePattern = regexp.MustCompile(`(?m)^\s*(?:const|let|var)\s+{\s*([^}]+)\s*}\s*=\s*require\(\s*["'](?:node:)?%s["']\s*\)`) tsNamespaceRequirePattern = regexp.MustCompile(`(?m)^\s*(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*require\(\s*["'](?:node:)?%s["']\s*\)`) + tsFixedModulePatterns = buildTypeScriptFixedModulePatterns() ) +type typeScriptModulePatternKey struct { + template *regexp.Regexp + module string +} + +// buildTypeScriptFixedModulePatterns compiles the small, closed set of module +// patterns used by the scanner once. Runtime-derived module names deliberately +// bypass this map so source text cannot grow a process-wide cache. +func buildTypeScriptFixedModulePatterns() map[typeScriptModulePatternKey]*regexp.Regexp { + patterns := []*regexp.Regexp{ + tsNamedImportPattern, + tsNamespaceImportPattern, + tsDefaultImportPattern, + tsNamedRequirePattern, + tsNamespaceRequirePattern, + } + cache := make(map[typeScriptModulePatternKey]*regexp.Regexp, len(patterns)*2) + for _, module := range []string{"child_process", "vm"} { + for _, pattern := range patterns { + key := typeScriptModulePatternKey{template: pattern, module: module} + cache[key] = compileDynamicPattern(strings.ReplaceAll(pattern.String(), "%s", regexp.QuoteMeta(module))) + } + } + return cache +} + +func typeScriptModulePattern(pattern *regexp.Regexp, module string) *regexp.Regexp { + if compiled := tsFixedModulePatterns[typeScriptModulePatternKey{template: pattern, module: module}]; compiled != nil { + return compiled + } + return compileDynamicPattern(strings.ReplaceAll(pattern.String(), "%s", regexp.QuoteMeta(module))) +} + func collectTypeScriptNamedModuleBindings(source string, module string, allowed []string) map[string]string { + if !strings.Contains(source, module) { + return map[string]string{} + } allowedSet := make(map[string]struct{}, len(allowed)) for _, name := range allowed { allowedSet[name] = struct{}{} @@ -31,9 +68,12 @@ func collectTypeScriptNamedModuleBindings(source string, module string, allowed } func collectTypeScriptNamespaceBindings(source string, module string) map[string]struct{} { + if !strings.Contains(source, module) { + return map[string]struct{}{} + } namespaces := make(map[string]struct{}) for _, pattern := range []*regexp.Regexp{tsNamespaceImportPattern, tsDefaultImportPattern, tsNamespaceRequirePattern} { - re := compileDynamicPattern(strings.ReplaceAll(pattern.String(), "%s", regexp.QuoteMeta(module))) + re := typeScriptModulePattern(pattern, module) for _, match := range re.FindAllStringSubmatch(source, -1) { if len(match) > 1 { namespaces[match[1]] = struct{}{} @@ -46,7 +86,7 @@ func collectTypeScriptNamespaceBindings(source string, module string) map[string func collectTypeScriptBindingSpecs(source string, module string, patterns ...*regexp.Regexp) []string { specs := make([]string, 0) for _, pattern := range patterns { - re := compileDynamicPattern(strings.ReplaceAll(pattern.String(), "%s", regexp.QuoteMeta(module))) + re := typeScriptModulePattern(pattern, module) for _, match := range re.FindAllStringSubmatch(source, -1) { if len(match) > 1 { specs = append(specs, splitTypeScriptBindingSpecs(match[1])...) diff --git a/internal/codeguard/checks/security/security_typescript_bindings_cache_test.go b/internal/codeguard/checks/security/security_typescript_bindings_cache_test.go new file mode 100644 index 0000000..7a5a142 --- /dev/null +++ b/internal/codeguard/checks/security/security_typescript_bindings_cache_test.go @@ -0,0 +1,31 @@ +package security + +import "testing" + +func TestTypeScriptModulePatternCachesOnlyFixedScannerModules(t *testing.T) { + t.Parallel() + + trustedFirst := typeScriptModulePattern(tsNamedImportPattern, "child_process") + trustedSecond := typeScriptModulePattern(tsNamedImportPattern, "child_process") + if trustedFirst != trustedSecond { + t.Fatal("fixed scanner module pattern was recompiled") + } + + untrustedFirst := typeScriptModulePattern(tsNamedImportPattern, "attacker-controlled") + untrustedSecond := typeScriptModulePattern(tsNamedImportPattern, "attacker-controlled") + if untrustedFirst == untrustedSecond { + t.Fatal("runtime-derived module pattern was retained globally") + } +} + +func TestTypeScriptModulePatternQuotesRuntimeModuleName(t *testing.T) { + t.Parallel() + + pattern := typeScriptModulePattern(tsNamedImportPattern, `unsafe.*module`) + if !pattern.MatchString(`import { exec } from "unsafe.*module"`) { + t.Fatal("literal runtime module name did not match") + } + if pattern.MatchString(`import { exec } from "unsafe-xyz-module"`) { + t.Fatal("runtime module name was interpreted as a regular expression") + } +} diff --git a/internal/codeguard/config/io.go b/internal/codeguard/config/io.go index e675351..627b1be 100644 --- a/internal/codeguard/config/io.go +++ b/internal/codeguard/config/io.go @@ -210,6 +210,7 @@ func resolveRelativePaths(cfg *core.Config, baseDir string) { if targetPath == "" || filepath.IsAbs(targetPath) { continue } + cfg.Targets[i].LogicalPath = filepath.ToSlash(filepath.Clean(targetPath)) cfg.Targets[i].Path = filepath.Join(baseDir, targetPath) } } diff --git a/internal/codeguard/core/config_types.go b/internal/codeguard/core/config_types.go index dadbff0..f093a65 100644 --- a/internal/codeguard/core/config_types.go +++ b/internal/codeguard/core/config_types.go @@ -12,10 +12,13 @@ type Config struct { ExternalReports []ExternalReportConfig `json:"external_reports,omitempty" yaml:"external_reports,omitempty"` Output OutputConfig `json:"output" yaml:"output"` Exclude []string `json:"exclude,omitempty" yaml:"exclude,omitempty"` - Baseline BaselineConfig `json:"baseline,omitempty" yaml:"baseline,omitempty"` - Waivers []WaiverConfig `json:"waivers,omitempty" yaml:"waivers,omitempty"` - Cache CacheConfig `json:"cache,omitempty" yaml:"cache,omitempty"` - Parsers ParsersConfig `json:"parsers,omitempty" yaml:"parsers,omitempty"` + // ScanVendoredSource includes source beneath directories named vendor. + // Installed node_modules and generated cdk.out trees remain excluded. + ScanVendoredSource bool `json:"scan_vendored_source,omitempty" yaml:"scan_vendored_source,omitempty"` + Baseline BaselineConfig `json:"baseline,omitempty" yaml:"baseline,omitempty"` + Waivers []WaiverConfig `json:"waivers,omitempty" yaml:"waivers,omitempty"` + Cache CacheConfig `json:"cache,omitempty" yaml:"cache,omitempty"` + Parsers ParsersConfig `json:"parsers,omitempty" yaml:"parsers,omitempty"` } // ExternalReportConfig describes a report file produced by another scanner. @@ -56,6 +59,9 @@ type TargetConfig struct { Path string `json:"path" yaml:"path"` Language string `json:"language" yaml:"language"` Entrypoints []string `json:"entrypoints,omitempty" yaml:"entrypoints,omitempty"` + // LogicalPath preserves the target's repository-relative path when a + // folder-scoped scan replaces Path with a narrower filesystem root. + LogicalPath string `json:"-" yaml:"-"` } type CheckConfig struct { diff --git a/internal/codeguard/runner/support/artifacts.go b/internal/codeguard/runner/support/artifacts.go index b8df470..164fd60 100644 --- a/internal/codeguard/runner/support/artifacts.go +++ b/internal/codeguard/runner/support/artifacts.go @@ -63,7 +63,7 @@ func (store *ArtifactStore) List() []core.Artifact { // graphs) always observe every file. It reuses the shared per-scan corpus, so // files are still walked and read only once across the whole scan. func VisitTargetFiles(sc Context, target core.TargetConfig, include func(string) bool, visit func(rel string, data []byte)) { - files, _ := sc.corpusFiles(target.Path) + files, _ := sc.corpusFiles(target) for _, file := range files { if !include(file) { continue @@ -81,7 +81,7 @@ func VisitTargetFiles(sc Context, target core.TargetConfig, include func(string) // VisitTargetFiles iterates). Callers apply their own include filter to the // result. func ListTargetFiles(sc Context, target core.TargetConfig) ([]string, error) { - return sc.corpusFiles(target.Path) + return sc.corpusFiles(target) } // ReadTargetFile returns the bytes of target-root-relative rel via the shared diff --git a/internal/codeguard/runner/support/corpus.go b/internal/codeguard/runner/support/corpus.go index 404d983..d3236ce 100644 --- a/internal/codeguard/runner/support/corpus.go +++ b/internal/codeguard/runner/support/corpus.go @@ -1,6 +1,7 @@ package support import ( + "errors" "fmt" "go/ast" "go/parser" @@ -11,6 +12,12 @@ import ( "sync" checkSupport "github.com/devr-tools/codeguard/internal/codeguard/checks/support" + "github.com/devr-tools/codeguard/internal/codeguard/core" +) + +var ( + errCorpusReadBudget = errors.New("scan corpus read budget exhausted") + errCorpusGoASTBudget = errors.New("scan corpus Go AST budget exhausted") ) // readCappedFile reads path but refuses to buffer more than maxScanFileBytes, @@ -43,16 +50,41 @@ func readCappedFile(path string) ([]byte, error) { // Each cached slot carries its own sync.Once, so concurrent callers racing on a // cold slot compute it exactly once and every caller observes the same result. type fileCorpus struct { - mu sync.Mutex - targets map[string]*targetListing - reads map[string]*fileRead - asts map[string]*goParse - scripts map[string]*scriptParse - scriptBytes int - scriptCount int - scriptParse chan struct{} + mu sync.Mutex + targets map[string]*targetListing + maxFiles int + reads map[string]*fileRead + readBytes int + maxReadBytes int + maxReadEntries int + readExhausted bool + readWork chan struct{} + readFile func(string) ([]byte, error) + asts map[string]*goParse + goASTBytes int + maxGoASTBytes int + maxGoASTEntries int + goASTExhausted bool + goParseWork chan struct{} + scripts map[string]*scriptParse + scriptBytes int + scriptCount int + scriptParse chan struct{} + diagnostics []core.Diagnostic + diagnosticSeen map[string]struct{} } +// The corpus is an optimization, not the owner of repository contents. Bound +// its retained source and Go AST working sets so a full scan can keep making +// progress under GOMEMLIMIT instead of pinning every file until report output. +// Once a budget is reached, later uncached work is skipped and surfaced as an +// informational scan diagnostic rather than repeatedly allocating overflow +// data in concurrent check sections. +const maxCorpusReadBytes = 128 << 20 +const maxCorpusGoASTSourceBytes = 64 << 20 +const maxCorpusReadEntries = 50_000 +const maxCorpusGoASTEntries = 25_000 + // maxTreeSitterScanBytes bounds the source represented by retained script // trees during one scan. Parsing is also serialized because the pure-Go // runtime's transient heap is much larger than its input. @@ -86,19 +118,28 @@ type scriptParse struct { func newFileCorpus() *fileCorpus { return &fileCorpus{ - targets: map[string]*targetListing{}, - reads: map[string]*fileRead{}, - asts: map[string]*goParse{}, - scripts: map[string]*scriptParse{}, - scriptParse: make(chan struct{}, 1), + targets: map[string]*targetListing{}, + maxFiles: maxScanFileCount, + reads: map[string]*fileRead{}, + maxReadBytes: maxCorpusReadBytes, + maxReadEntries: maxCorpusReadEntries, + readWork: make(chan struct{}, 2), + readFile: readCappedFile, + asts: map[string]*goParse{}, + maxGoASTBytes: maxCorpusGoASTSourceBytes, + maxGoASTEntries: maxCorpusGoASTEntries, + goParseWork: make(chan struct{}, 2), + scripts: map[string]*scriptParse{}, + scriptParse: make(chan struct{}, 1), + diagnosticSeen: make(map[string]struct{}), } } // list returns every non-excluded file under root, walking the tree only once // per target. Callers apply their own include filter to the returned slice; the // walk itself is identical regardless of the filter, so sharing it is safe. -func (c *fileCorpus) list(root string, excludes []string) ([]string, error) { - key := filepath.Clean(root) +func (c *fileCorpus) list(root string, excludes []string, opts FileWalkOptions) ([]string, error) { + key := fmt.Sprintf("%s\x00%s\x00%t", filepath.Clean(root), opts.LogicalPath, opts.ScanVendoredSource) c.mu.Lock() entry, ok := c.targets[key] if !ok { @@ -108,7 +149,11 @@ func (c *fileCorpus) list(root string, excludes []string) ([]string, error) { c.mu.Unlock() entry.once.Do(func() { - entry.files, entry.err = WalkFiles(root, excludes, includeAll) + var truncated bool + entry.files, truncated, entry.err = walkFilesBounded(root, excludes, opts, includeAll, c.maxFiles) + if truncated { + c.recordBudget("files", c.maxFiles) + } }) return entry.files, entry.err } @@ -119,13 +164,36 @@ func (c *fileCorpus) read(root string, rel string) ([]byte, error) { c.mu.Lock() entry, ok := c.reads[key] if !ok { + if c.readExhausted || len(c.reads) >= c.maxReadEntries { + c.readExhausted = true + c.recordBudgetLocked("read_entries", c.maxReadEntries) + c.mu.Unlock() + return nil, errCorpusReadBudget + } entry = &fileRead{} c.reads[key] = entry } c.mu.Unlock() entry.once.Do(func() { - entry.data, entry.err = readCappedFile(filepath.Join(root, rel)) + c.readWork <- struct{}{} + entry.data, entry.err = c.readFile(filepath.Join(root, rel)) + <-c.readWork + c.mu.Lock() + defer c.mu.Unlock() + if entry.err == nil && c.readBytes+len(entry.data) <= c.maxReadBytes { + c.readBytes += len(entry.data) + return + } + if entry.err == nil { + entry.data = nil + entry.err = errCorpusReadBudget + c.readExhausted = true + c.recordBudgetLocked("read_bytes", c.maxReadBytes) + } + if c.reads[key] == entry { + delete(c.reads, key) + } }) return entry.data, entry.err } @@ -140,12 +208,25 @@ func (c *fileCorpus) parseGo(path string, data []byte) (*token.FileSet, *ast.Fil c.mu.Lock() entry, ok := c.asts[key] if !ok { + if c.goASTExhausted || len(c.asts) >= c.maxGoASTEntries || c.goASTBytes+len(data) > c.maxGoASTBytes { + c.goASTExhausted = true + if len(c.asts) >= c.maxGoASTEntries { + c.recordBudgetLocked("go_ast_entries", c.maxGoASTEntries) + } else { + c.recordBudgetLocked("go_ast_bytes", c.maxGoASTBytes) + } + c.mu.Unlock() + return nil, nil, errCorpusGoASTBudget + } entry = &goParse{} c.asts[key] = entry + c.goASTBytes += len(data) } c.mu.Unlock() entry.once.Do(func() { + c.goParseWork <- struct{}{} + defer func() { <-c.goParseWork }() fset := token.NewFileSet() file, err := parser.ParseFile(fset, path, data, parser.ParseComments) entry.fset, entry.file, entry.err = fset, file, err @@ -153,6 +234,40 @@ func (c *fileCorpus) parseGo(path string, data []byte) (*token.FileSet, *ast.Fil return entry.fset, entry.file, entry.err } +func (c *fileCorpus) recordBudget(kind string, limit int) { + c.mu.Lock() + defer c.mu.Unlock() + c.recordBudgetLocked(kind, limit) +} + +func (c *fileCorpus) recordBudgetLocked(kind string, limit int) { + if _, exists := c.diagnosticSeen[kind]; exists { + return + } + c.diagnosticSeen[kind] = struct{}{} + c.diagnostics = append(c.diagnostics, core.Diagnostic{ + ID: "scan.corpus-budget", + Level: "info", + Kind: "analysis", + Message: "repository analysis reached a bounded corpus budget; remaining uncached work was skipped", + Metadata: map[string]string{ + "budget": kind, + "limit": fmt.Sprintf("%d", limit), + }, + }) +} + +func (c *fileCorpus) takeDiagnostics() []core.Diagnostic { + if c == nil { + return nil + } + c.mu.Lock() + defer c.mu.Unlock() + diagnostics := append([]core.Diagnostic(nil), c.diagnostics...) + c.diagnostics = nil + return diagnostics +} + // parseScript returns a shared tree-sitter syntax tree for the given script // source, mirroring parseGo: keyed by path plus content hash (so diff-mode // patched content reparses) with a sync.Once per slot, so N rules across N diff --git a/internal/codeguard/runner/support/corpus_access.go b/internal/codeguard/runner/support/corpus_access.go index f30a503..c6014e5 100644 --- a/internal/codeguard/runner/support/corpus_access.go +++ b/internal/codeguard/runner/support/corpus_access.go @@ -7,15 +7,20 @@ import ( "path/filepath" checkSupport "github.com/devr-tools/codeguard/internal/codeguard/checks/support" + "github.com/devr-tools/codeguard/internal/codeguard/core" ) func includeAll(string) bool { return true } -func (sc Context) corpusFiles(root string) ([]string, error) { +func (sc Context) corpusFiles(target core.TargetConfig) ([]string, error) { + opts := FileWalkOptions{ + LogicalPath: target.LogicalPath, + ScanVendoredSource: sc.Cfg.ScanVendoredSource, + } if sc.corpus != nil { - return sc.corpus.list(root, sc.Cfg.Exclude) + return sc.corpus.list(target.Path, sc.Cfg.Exclude, opts) } - return WalkFiles(root, sc.Cfg.Exclude, includeAll) + return WalkFilesWithOptions(target.Path, sc.Cfg.Exclude, opts, includeAll) } func (sc Context) corpusRead(root string, rel string) ([]byte, error) { diff --git a/internal/codeguard/runner/support/corpus_memory_test.go b/internal/codeguard/runner/support/corpus_memory_test.go new file mode 100644 index 0000000..c500032 --- /dev/null +++ b/internal/codeguard/runner/support/corpus_memory_test.go @@ -0,0 +1,201 @@ +package support + +import ( + "errors" + "os" + "path/filepath" + "sync" + "sync/atomic" + "testing" +) + +func TestFileCorpusDoesNotRetainReadsBeyondAggregateBudget(t *testing.T) { + root := t.TempDir() + first := []byte("package first\n") + second := []byte("package second\n") + for name, data := range map[string][]byte{"first.go": first, "second.go": second} { + if err := os.WriteFile(filepath.Join(root, name), data, 0o600); err != nil { + t.Fatalf("write %s: %v", name, err) + } + } + + corpus := newFileCorpus() + corpus.maxReadBytes = len(first) + if _, err := corpus.read(root, "first.go"); err != nil { + t.Fatalf("read first file: %v", err) + } + if _, err := corpus.read(root, "second.go"); !errors.Is(err, errCorpusReadBudget) { + t.Fatalf("read second file error = %v, want corpus budget error", err) + } + + if corpus.readBytes > corpus.maxReadBytes { + t.Fatalf("retained read bytes = %d, budget = %d", corpus.readBytes, corpus.maxReadBytes) + } + if got := len(corpus.reads); got != 1 { + t.Fatalf("retained read entries = %d, want 1", got) + } + + if _, err := corpus.read(root, "second.go"); !errors.Is(err, errCorpusReadBudget) { + t.Fatalf("repeated overflow read error = %v, want corpus budget error", err) + } +} + +func TestFileCorpusDoesNotRetainGoASTsBeyondAggregateBudget(t *testing.T) { + corpus := newFileCorpus() + first := []byte("package first\n") + second := []byte("package second\n") + corpus.maxGoASTBytes = len(first) + + if _, _, err := corpus.parseGo("first.go", first); err != nil { + t.Fatalf("parse first file: %v", err) + } + if _, _, err := corpus.parseGo("second.go", second); !errors.Is(err, errCorpusGoASTBudget) { + t.Fatalf("parse second file error = %v, want corpus budget error", err) + } + + if corpus.goASTBytes > corpus.maxGoASTBytes { + t.Fatalf("retained Go AST source bytes = %d, budget = %d", corpus.goASTBytes, corpus.maxGoASTBytes) + } + if got := len(corpus.asts); got != 1 { + t.Fatalf("retained Go AST entries = %d, want 1", got) + } +} + +func TestFileCorpusBoundsZeroByteReadEntries(t *testing.T) { + root := t.TempDir() + for _, name := range []string{"first", "second", "third"} { + if err := os.WriteFile(filepath.Join(root, name), nil, 0o600); err != nil { + t.Fatalf("write %s: %v", name, err) + } + } + + corpus := newFileCorpus() + corpus.maxReadEntries = 2 + for _, name := range []string{"first", "second"} { + if _, err := corpus.read(root, name); err != nil { + t.Fatalf("read %s: %v", name, err) + } + } + if _, err := corpus.read(root, "third"); !errors.Is(err, errCorpusReadBudget) { + t.Fatalf("third read error = %v, want corpus budget error", err) + } + if got := len(corpus.reads); got != 2 { + t.Fatalf("retained read entries = %d, want 2", got) + } +} + +func TestFileCorpusBoundsFileListingsAndReportsDiagnostic(t *testing.T) { + root := t.TempDir() + for _, name := range []string{"first.go", "second.go", "third.go"} { + if err := os.WriteFile(filepath.Join(root, name), []byte("package fixture\n"), 0o600); err != nil { + t.Fatalf("write %s: %v", name, err) + } + } + + corpus := newFileCorpus() + corpus.maxFiles = 2 + files, err := corpus.list(root, nil, FileWalkOptions{}) + if err != nil { + t.Fatalf("list files: %v", err) + } + if got := len(files); got != 2 { + t.Fatalf("listed files = %d, want 2", got) + } + section := FinalizeSection(Context{corpus: corpus}, "quality", "Code Quality", nil) + if len(section.Diagnostics) != 1 || section.Diagnostics[0].ID != "scan.corpus-budget" { + t.Fatalf("diagnostics = %#v, want one corpus budget diagnostic", section.Diagnostics) + } + if again := corpus.takeDiagnostics(); len(again) != 0 { + t.Fatalf("diagnostic was emitted more than once: %#v", again) + } +} + +func TestFileCorpusRejectsConcurrentGoParsesBeyondBudget(t *testing.T) { + corpus := newFileCorpus() + corpus.maxGoASTBytes = 1 + data := []byte("package fixture\n") + + const callers = 32 + var wg sync.WaitGroup + errs := make(chan error, callers) + for i := 0; i < callers; i++ { + wg.Add(1) + go func() { + defer wg.Done() + _, _, err := corpus.parseGo("large.go", data) + errs <- err + }() + } + wg.Wait() + close(errs) + for err := range errs { + if !errors.Is(err, errCorpusGoASTBudget) { + t.Fatalf("parse error = %v, want corpus budget error", err) + } + } + if got := len(corpus.asts); got != 0 { + t.Fatalf("retained Go AST entries = %d, want 0", got) + } +} + +func TestFileCorpusSingleflightsConcurrentOverflowReads(t *testing.T) { + corpus := newFileCorpus() + corpus.maxReadBytes = 1 + var readCalls atomic.Int32 + releaseRead := make(chan struct{}) + corpus.readFile = func(string) ([]byte, error) { + readCalls.Add(1) + <-releaseRead + return []byte("too large"), nil + } + + const callers = 32 + start := make(chan struct{}) + var entered sync.WaitGroup + var done sync.WaitGroup + entered.Add(callers) + done.Add(callers) + errs := make(chan error, callers) + for i := 0; i < callers; i++ { + go func() { + defer done.Done() + <-start + entered.Done() + _, err := corpus.read("root", "large.go") + errs <- err + }() + } + close(start) + entered.Wait() + close(releaseRead) + done.Wait() + close(errs) + + for err := range errs { + if !errors.Is(err, errCorpusReadBudget) { + t.Fatalf("read error = %v, want corpus budget error", err) + } + } + if got := readCalls.Load(); got != 1 { + t.Fatalf("overflow file reads = %d, want one singleflight read", got) + } +} + +func TestFileCorpusBoundsGoASTEntriesIndependentOfBytes(t *testing.T) { + corpus := newFileCorpus() + corpus.maxGoASTBytes = 1 << 20 + corpus.maxGoASTEntries = 2 + data := []byte("package fixture\n") + + for _, name := range []string{"first.go", "second.go"} { + if _, _, err := corpus.parseGo(name, data); err != nil { + t.Fatalf("parse %s: %v", name, err) + } + } + if _, _, err := corpus.parseGo("third.go", data); !errors.Is(err, errCorpusGoASTBudget) { + t.Fatalf("third parse error = %v, want corpus budget error", err) + } + if got := len(corpus.asts); got != 2 { + t.Fatalf("retained Go AST entries = %d, want 2", got) + } +} diff --git a/internal/codeguard/runner/support/findings_scan.go b/internal/codeguard/runner/support/findings_scan.go index 92ecd76..62affa9 100644 --- a/internal/codeguard/runner/support/findings_scan.go +++ b/internal/codeguard/runner/support/findings_scan.go @@ -69,7 +69,7 @@ func scanTargetFiles(sc Context, target core.TargetConfig, spec fileScanSpec) [] } func selectedTargetFiles(sc Context, target core.TargetConfig, include func(string) bool) []string { - files, _ := sc.corpusFiles(target.Path) + files, _ := sc.corpusFiles(target) selected := make([]string, 0, len(files)) for _, file := range files { if include(file) { diff --git a/internal/codeguard/runner/support/findings_section.go b/internal/codeguard/runner/support/findings_section.go index 94e9057..25583bc 100644 --- a/internal/codeguard/runner/support/findings_section.go +++ b/internal/codeguard/runner/support/findings_section.go @@ -99,6 +99,9 @@ func FinalizeSection(sc Context, id string, name string, findings []core.Finding } func FinalizeSectionWithDiagnostics(sc Context, id string, name string, findings []core.Finding, diagnostics []core.Diagnostic) core.SectionResult { + if sc.corpus != nil { + diagnostics = append(diagnostics, sc.corpus.takeDiagnostics()...) + } section := core.SectionResult{ID: id, Name: name, Status: core.StatusPass} active := make([]core.Finding, 0, len(findings)) for _, finding := range findings { diff --git a/internal/codeguard/runner/support/target_path.go b/internal/codeguard/runner/support/target_path.go index f6710f2..5b6d66d 100644 --- a/internal/codeguard/runner/support/target_path.go +++ b/internal/codeguard/runner/support/target_path.go @@ -13,6 +13,7 @@ import ( // folder. The path is a runtime option rather than config, so it may point at // any local directory the caller intentionally chose. func ApplyTargetPath(cfg *core.Config, targetPath string) error { + initializeTargetLogicalPaths(cfg) targetPath = strings.TrimSpace(targetPath) if targetPath == "" { return nil @@ -32,6 +33,11 @@ func ApplyTargetPath(cfg *core.Config, targetPath string) error { switch { case samePath(scanPath, targetAbs), isSubpath(scanPath, targetAbs): + rel, relErr := filepath.Rel(targetAbs, scanPath) + if relErr != nil { + return fmt.Errorf("scan path relative to target %q: %w", target.Name, relErr) + } + target.LogicalPath = joinLogicalPath(target.LogicalPath, rel) target.Path = scanPath matched = append(matched, target) case isSubpath(targetAbs, scanPath): @@ -53,6 +59,35 @@ func ApplyTargetPath(cfg *core.Config, targetPath string) error { return nil } +func initializeTargetLogicalPaths(cfg *core.Config) { + for i := range cfg.Targets { + if cfg.Targets[i].LogicalPath != "" { + continue + } + path := filepath.Clean(cfg.Targets[i].Path) + if !filepath.IsAbs(path) { + cfg.Targets[i].LogicalPath = filepath.ToSlash(path) + continue + } + switch base := filepath.Base(path); base { + case "vendor", "node_modules", "cdk.out": + cfg.Targets[i].LogicalPath = base + } + } +} + +func joinLogicalPath(base, rel string) string { + base = filepath.ToSlash(filepath.Clean(base)) + rel = filepath.ToSlash(filepath.Clean(rel)) + if base == "." || base == "" { + return rel + } + if rel == "." || rel == "" { + return base + } + return base + "/" + rel +} + func absoluteDir(path string) (string, error) { abs, err := filepath.Abs(path) if err != nil { diff --git a/internal/codeguard/runner/support/utils.go b/internal/codeguard/runner/support/utils.go index fcd52ad..13426c6 100644 --- a/internal/codeguard/runner/support/utils.go +++ b/internal/codeguard/runner/support/utils.go @@ -17,6 +17,12 @@ import ( // are far smaller than this; oversized inputs are almost always generated blobs // or vendored bundles that are not useful to scan. const maxScanFileBytes = 32 << 20 // 32 MiB +const maxScanFileCount = 100_000 + +type FileWalkOptions struct { + LogicalPath string + ScanVendoredSource bool +} func SummarizeSections(sections []core.SectionResult) core.ReportSummary { var summary core.ReportSummary @@ -36,12 +42,26 @@ func SummarizeSections(sections []core.SectionResult) core.ReportSummary { } func WalkFiles(root string, excludes []string, include func(string) bool) ([]string, error) { + return WalkFilesWithOptions(root, excludes, FileWalkOptions{}, include) +} + +func WalkFilesWithOptions(root string, excludes []string, opts FileWalkOptions, include func(string) bool) ([]string, error) { + files, _, err := walkFilesBounded(root, excludes, opts, include, maxScanFileCount) + return files, err +} + +func walkFilesBounded(root string, excludes []string, opts FileWalkOptions, include func(string) bool, maxFiles int) ([]string, bool, error) { var files []string + truncated := false err := filepath.WalkDir(root, func(path string, d fs.DirEntry, err error) error { if err != nil { return err } if path == root { + logicalRoot := filepath.ToSlash(filepath.Clean(opts.LogicalPath)) + if logicalRoot != "." && shouldExclude(logicalRoot, excludes, opts) { + return fs.SkipAll + } return nil } rel, err := filepath.Rel(root, path) @@ -49,7 +69,9 @@ func WalkFiles(root string, excludes []string, include func(string) bool) ([]str return err } rel = filepath.ToSlash(rel) - if ShouldExclude(rel, excludes) { + logicalRel := joinLogicalPath(opts.LogicalPath, rel) + if shouldExclude(rel, excludes, opts) || + (logicalRel != rel && shouldExclude(logicalRel, excludes, opts)) { if d.IsDir() { return filepath.SkipDir } @@ -67,11 +89,15 @@ func WalkFiles(root string, excludes []string, include func(string) bool) ([]str return nil } if include(rel) { + if len(files) >= maxFiles { + truncated = true + return fs.SkipAll + } files = append(files, rel) } return nil }) - return files, err + return files, truncated, err } // CountLines reports how many lines data spans without allocating. It diff --git a/internal/codeguard/runner/support/utils_match.go b/internal/codeguard/runner/support/utils_match.go index fb2049a..41b9dbb 100644 --- a/internal/codeguard/runner/support/utils_match.go +++ b/internal/codeguard/runner/support/utils_match.go @@ -32,6 +32,13 @@ func IsSDKFacadeFile(path string) bool { } func ShouldExclude(rel string, excludes []string) bool { + return shouldExclude(rel, excludes, FileWalkOptions{}) +} + +func shouldExclude(rel string, excludes []string, opts FileWalkOptions) bool { + if isDefaultDependencyOrBuildPath(rel, opts.ScanVendoredSource) { + return true + } defaults := []string{".git/**", ".gocache/**", ".gomodcache/**", ".codeguard/**", "dist/**"} for _, pattern := range append(defaults, excludes...) { if MatchPattern(pattern, rel) { @@ -41,6 +48,18 @@ func ShouldExclude(rel string, excludes []string) bool { return false } +func isDefaultDependencyOrBuildPath(rel string, scanVendoredSource bool) bool { + for _, segment := range strings.Split(filepath.ToSlash(rel), "/") { + switch segment { + case "node_modules", "cdk.out": + return true + case "vendor": + return !scanVendoredSource + } + } + return false +} + func MatchPattern(pattern string, value string) bool { pattern = filepath.ToSlash(strings.TrimSpace(pattern)) value = filepath.ToSlash(strings.TrimSpace(value)) diff --git a/tests/codeguard/full_scan_benchmark_test.go b/tests/codeguard/full_scan_benchmark_test.go new file mode 100644 index 0000000..cf2169d --- /dev/null +++ b/tests/codeguard/full_scan_benchmark_test.go @@ -0,0 +1,159 @@ +package codeguard_test + +import ( + "context" + "fmt" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/devr-tools/codeguard/pkg/codeguard" +) + +func BenchmarkFullScanMixedGoTypeScriptRepository(b *testing.B) { + root := buildFullScanBenchmarkRepository(b, 64) + for _, tc := range []struct { + name string + cacheEnabled bool + warmup bool + scanVendoredSource bool + }{ + {name: "cold", cacheEnabled: false}, + {name: "warm-cache", cacheEnabled: true, warmup: true}, + {name: "cold-vendored-source", cacheEnabled: false, scanVendoredSource: true}, + } { + b.Run(tc.name, func(b *testing.B) { + cfg := fullScanBenchmarkConfig(root, tc.cacheEnabled, tc.scanVendoredSource) + if tc.scanVendoredSource { + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + b.Fatalf("verify vendored benchmark fixture: %v", err) + } + if !benchmarkReportHasVendorFinding(report) { + b.Fatal("vendored benchmark case did not analyze vendored source") + } + } + if tc.warmup { + if _, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}); err != nil { + b.Fatalf("warm benchmark cache: %v", err) + } + } + b.ReportAllocs() + b.ResetTimer() + for i := 0; i < b.N; i++ { + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + b.Fatalf("full scan: %v", err) + } + if len(report.Sections) < 10 { + b.Fatalf("full scan returned only %d sections", len(report.Sections)) + } + } + }) + } +} + +func buildFullScanBenchmarkRepository(b *testing.B, filesPerLanguage int) string { + b.Helper() + root := b.TempDir() + for i := 0; i < filesPerLanguage; i++ { + goSource := fmt.Sprintf(`package service + +import "context" + +type Handler%d struct{} + +func (Handler%d) Execute(ctx context.Context, values []int) int { + total := 0 + for _, value := range values { + if value > 0 { + total += value + } + } + return total +} +`, i, i) + writeBenchmarkFile(b, filepath.Join(root, "app", fmt.Sprintf("handler_%03d.go", i)), goSource) + + tsSource := fmt.Sprintf(`export interface Request%d { values: number[] } + +export function execute%d(request: Request%d): number { + let total = 0; + for (const value of request.values) { + if (value > 0) total += value; + } + return total; +} +`, i, i, i) + writeBenchmarkFile(b, filepath.Join(root, "infrastructure", "src", fmt.Sprintf("stack-%03d.ts", i)), tsSource) + } + + generated := strings.Repeat("export const generatedValue = 1;\n", 32) + for i := 0; i < filesPerLanguage*4; i++ { + writeBenchmarkFile(b, filepath.Join(root, "infrastructure", "node_modules", "library", fmt.Sprintf("generated-%03d.ts", i)), generated) + writeBenchmarkFile(b, filepath.Join(root, "infrastructure", "cdk.out", fmt.Sprintf("asset.%03d", i), "index.js"), generated) + writeBenchmarkFile(b, filepath.Join(root, "vendor", "example.com", "dependency", fmt.Sprintf("generated_%03d.go", i)), "package dependency\n") + } + writeBenchmarkFile(b, filepath.Join(root, "vendor", "example.com", "dependency", "oversized.go"), "package dependency\n"+strings.Repeat("// benchmark vendored source\n", 600)) + writeBenchmarkFile(b, filepath.Join(root, "go.mod"), "module example.com/fullscanbench\n\ngo 1.23\n") + writeBenchmarkFile(b, filepath.Join(root, "package.json"), `{"name":"full-scan-benchmark","private":true}`) + return root +} + +func benchmarkReportHasVendorFinding(report codeguard.Report) bool { + for _, section := range report.Sections { + for _, finding := range section.Findings { + if strings.HasPrefix(filepath.ToSlash(finding.Path), "vendor/") { + return true + } + } + } + return false +} + +func fullScanBenchmarkConfig(root string, cacheEnabled bool, scanVendoredSource bool) codeguard.Config { + enabled := true + disabled := false + cfg := codeguard.ExampleConfig() + cfg.Name = "mixed-full-scan-benchmark" + cfg.ScanVendoredSource = scanVendoredSource + cfg.Targets = []codeguard.TargetConfig{ + {Name: "go-app", Path: root, Language: "go", Entrypoints: []string{"app"}}, + {Name: "typescript-infrastructure", Path: root, Language: "typescript", Entrypoints: []string{"infrastructure/src"}}, + } + cfg.Checks.Quality = true + cfg.Checks.Performance = &enabled + cfg.Checks.Reliability = &enabled + cfg.Checks.Data = &enabled + cfg.Checks.Observability = &enabled + cfg.Checks.Operations = &enabled + cfg.Checks.Design = true + cfg.Checks.Security = true + cfg.Checks.Prompts = true + cfg.Checks.CI = true + cfg.Checks.SupplyChain = true + cfg.Checks.Context = &enabled + cfg.Checks.Change = &disabled + cfg.Checks.Contracts = &disabled + cfg.Checks.Delivery = &disabled + cfg.Checks.QualityRules.DeadCode.Enabled = &disabled + cfg.Checks.QualityRules.LocalPrecision = &disabled + cfg.Checks.QualityRules.AIProvenance.Enabled = &disabled + cfg.Checks.QualityRules.AIChangeRisk.Enabled = &disabled + cfg.Checks.SecurityRules.GovulncheckMode = "off" + cfg.Cache.Enabled = &cacheEnabled + cfg.Cache.Path = filepath.Join(root, ".codeguard", "benchmark-cache.json") + cfg.Output.Format = "json" + return cfg +} + +func writeBenchmarkFile(tb testing.TB, path string, content string) { + tb.Helper() + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + tb.Fatalf("mkdir %s: %v", filepath.Dir(path), err) + } + if err := os.WriteFile(path, []byte(content), 0o600); err != nil { + tb.Fatalf("write %s: %v", path, err) + } +} diff --git a/tests/codeguard/yaml_config_helpers_test.go b/tests/codeguard/yaml_config_helpers_test.go index 98f3dd8..cb02892 100644 --- a/tests/codeguard/yaml_config_helpers_test.go +++ b/tests/codeguard/yaml_config_helpers_test.go @@ -10,6 +10,7 @@ import ( func yamlRoundTripConfig() codeguard.Config { cfg := codeguard.ExampleConfig() + cfg.ScanVendoredSource = true cfg.Checks.SupplyChain = true cfg.Checks.QualityRules.LanguageCommands = map[string][]codeguard.CommandCheckConfig{ "typescript": {{Name: "tsc", Command: "npx", Args: []string{"tsc", "--noEmit"}}}, @@ -45,7 +46,7 @@ func assertYAMLSchemaMarkers(t *testing.T, path string) { t.Fatalf("read yaml: %v", err) } rendered := string(data) - for _, want := range []string{"supply_chain:", "quality_rules:", "max_file_lines:", "language_commands:", "naming:", "allowed_abbreviations:", "role_suffix_warn_threshold:", "ci_rules:", "required_workflow_files:", "hybrid_triage:", "candidate_sections:", "function_contract:", "test_commands:", "rule_packs:"} { + for _, want := range []string{"scan_vendored_source: true", "supply_chain:", "quality_rules:", "max_file_lines:", "language_commands:", "naming:", "allowed_abbreviations:", "role_suffix_warn_threshold:", "ci_rules:", "required_workflow_files:", "hybrid_triage:", "candidate_sections:", "function_contract:", "test_commands:", "rule_packs:"} { if !strings.Contains(rendered, want) { t.Fatalf("written yaml missing %q:\n%s", want, rendered) } @@ -57,6 +58,9 @@ func assertYAMLRoundTripConfig(t *testing.T, loaded codeguard.Config, want codeg if loaded.Name != want.Name { t.Fatalf("loaded name = %q, want %q", loaded.Name, want.Name) } + if loaded.ScanVendoredSource != want.ScanVendoredSource { + t.Fatalf("scan_vendored_source = %t, want %t", loaded.ScanVendoredSource, want.ScanVendoredSource) + } assertYAMLCommand(t, loaded.Checks.QualityRules.LanguageCommands["typescript"][0].Command, "npx", "loaded command") if loaded.Checks.QualityRules.Naming.RoleSuffixWarnThreshold != want.Checks.QualityRules.Naming.RoleSuffixWarnThreshold { t.Fatalf("role_suffix_warn_threshold = %d, want %d", loaded.Checks.QualityRules.Naming.RoleSuffixWarnThreshold, want.Checks.QualityRules.Naming.RoleSuffixWarnThreshold) @@ -77,6 +81,7 @@ func assertYAMLCommand(t *testing.T, got string, want string, label string) { func snakeCaseYAMLFixture() string { return `name: snake-case-config +scan_vendored_source: true targets: - name: repo path: . @@ -143,6 +148,9 @@ output: func assertSnakeCaseYAMLLoaded(t *testing.T, loaded codeguard.Config) { t.Helper() + if !loaded.ScanVendoredSource { + t.Fatal("expected scan_vendored_source to load from snake_case yaml") + } assertSnakeCaseChecks(t, loaded) assertSnakeCaseAI(t, loaded) assertSnakeCaseRulePack(t, loaded) diff --git a/tests/security/scan_file_cap_test.go b/tests/security/scan_file_cap_test.go index 8559edb..eb4085b 100644 --- a/tests/security/scan_file_cap_test.go +++ b/tests/security/scan_file_cap_test.go @@ -1,13 +1,356 @@ package security_test import ( + "context" + "fmt" "os" "path/filepath" + "reflect" + "sort" + "strings" "testing" runnersupport "github.com/devr-tools/codeguard/internal/codeguard/runner/support" + "github.com/devr-tools/codeguard/pkg/codeguard" ) +// A repository-wide scan must not analyze dependency or generated build trees. +// Go vendor trees and TypeScript/CDK output can contain hundreds of thousands +// of source-shaped files; admitting them makes full-scan work and retained +// corpus memory proportional to installed dependencies instead of owned code. +func TestWalkFilesSkipsDependencyAndCDKOutputTrees(t *testing.T) { + root := t.TempDir() + files := map[string]string{ + "cmd/service/main.go": "package main\n", + "infrastructure/app.ts": "export const app = {};\n", + "vendor/example.com/dependency/dependency.go": "package dependency\n", + "infrastructure/node_modules/library/src/index.ts": "export const dependency = {};\n", + "infrastructure/cdk.out/asset.1234567890abcdef/index.js": "exports.generated = true;\n", + } + for rel, content := range files { + path := filepath.Join(root, filepath.FromSlash(rel)) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatalf("mkdir %s: %v", rel, err) + } + if err := os.WriteFile(path, []byte(content), 0o600); err != nil { + t.Fatalf("write %s: %v", rel, err) + } + } + + got, err := runnersupport.WalkFiles(root, nil, func(string) bool { return true }) + if err != nil { + t.Fatalf("walk: %v", err) + } + sort.Strings(got) + want := []string{"cmd/service/main.go", "infrastructure/app.ts"} + if !reflect.DeepEqual(got, want) { + t.Fatalf("walked files = %v, want only owned source %v", got, want) + } +} + +func TestWalkFilesCanIncludeVendoredSourceWithoutIncludingInstalledOrGeneratedTrees(t *testing.T) { + root := t.TempDir() + files := map[string]string{ + "app/main.go": "package app\n", + "vendor/example.com/dependency/dependency.go": "package dependency\n", + "nested/vendor/local/patch.go": "package local\n", + "node_modules/library/index.js": "exports.installed = true;\n", + "nested/node_modules/library/index.js": "exports.installed = true;\n", + "infrastructure/cdk.out/asset/index.js": "exports.generated = true;\n", + } + for rel, content := range files { + path := filepath.Join(root, filepath.FromSlash(rel)) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatalf("mkdir %s: %v", rel, err) + } + if err := os.WriteFile(path, []byte(content), 0o600); err != nil { + t.Fatalf("write %s: %v", rel, err) + } + } + + got, err := runnersupport.WalkFilesWithOptions(root, []string{"nested/vendor/**"}, runnersupport.FileWalkOptions{ + ScanVendoredSource: true, + }, func(string) bool { return true }) + if err != nil { + t.Fatalf("walk: %v", err) + } + sort.Strings(got) + want := []string{ + "app/main.go", + "vendor/example.com/dependency/dependency.go", + } + if !reflect.DeepEqual(got, want) { + t.Fatalf("walked files = %v, want owned and vendored source %v", got, want) + } +} + +func TestWalkFilesAppliesDependencyPolicyToLogicalTargetRoot(t *testing.T) { + root := t.TempDir() + if err := os.WriteFile(filepath.Join(root, "dependency.go"), []byte("package dependency\n"), 0o600); err != nil { + t.Fatalf("write dependency source: %v", err) + } + + for _, tc := range []struct { + name string + logicalPath string + scanVendoredSource bool + wantFiles []string + }{ + {name: "vendor-default", logicalPath: "vendor/example.com/dependency"}, + {name: "vendor-opt-in", logicalPath: "vendor/example.com/dependency", scanVendoredSource: true, wantFiles: []string{"dependency.go"}}, + {name: "nested-node-modules", logicalPath: "web/node_modules/library", scanVendoredSource: true}, + {name: "nested-cdk-output", logicalPath: "infrastructure/cdk.out/asset", scanVendoredSource: true}, + } { + t.Run(tc.name, func(t *testing.T) { + got, err := runnersupport.WalkFilesWithOptions(root, nil, runnersupport.FileWalkOptions{ + LogicalPath: tc.logicalPath, + ScanVendoredSource: tc.scanVendoredSource, + }, func(string) bool { return true }) + if err != nil { + t.Fatalf("walk: %v", err) + } + if !reflect.DeepEqual(got, tc.wantFiles) { + t.Fatalf("walked files = %v, want %v", got, tc.wantFiles) + } + }) + } +} + +func TestFullScanIncludesVendoredSourceWhenConfigured(t *testing.T) { + root := t.TempDir() + vendoredPath := filepath.Join(root, "vendor", "example.com", "dependency", "dependency.go") + if err := os.MkdirAll(filepath.Dir(vendoredPath), 0o755); err != nil { + t.Fatalf("mkdir vendor: %v", err) + } + if err := os.WriteFile(vendoredPath, []byte(strings.Repeat("// vendored source\n", 8)), 0o600); err != nil { + t.Fatalf("write vendored source: %v", err) + } + + cfg := codeguard.ExampleConfig() + cfg.Name = "vendored-source-opt-in" + cfg.Targets = []codeguard.TargetConfig{{Name: "repo", Path: root, Language: "go"}} + cfg.ScanVendoredSource = true + cfg.Checks.Quality = true + cfg.Checks.QualityRules.MaxFileLines = 2 + cfg.Checks.QualityRules.MaxFunctionLines = 100 + cfg.Checks.SecurityRules.GovulncheckMode = "off" + cacheEnabled := true + cfg.Cache.Enabled = &cacheEnabled + cfg.Cache.Path = filepath.Join(root, ".codeguard", "vendored-source-cache.json") + + cfg.ScanVendoredSource = false + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + t.Fatalf("default full scan: %v", err) + } + if reportHasFindingPath(report, "vendor/example.com/dependency/dependency.go") { + t.Fatal("default full scan included vendored source") + } + + cfg.ScanVendoredSource = true + report, err = codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + t.Fatalf("opt-in full scan: %v", err) + } + if !reportHasFindingPath(report, "vendor/example.com/dependency/dependency.go") { + t.Fatal("opt-in full scan did not report the oversized vendored source after a default cached scan") + } + + cfg.ScanVendoredSource = false + report, err = codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + t.Fatalf("second default full scan: %v", err) + } + if reportHasFindingPath(report, "vendor/example.com/dependency/dependency.go") { + t.Fatal("default full scan replayed a cached vendored-source finding") + } +} + +func TestConfigFileTargetRootHonorsVendoredSourcePolicy(t *testing.T) { + root := t.TempDir() + if err := os.MkdirAll(filepath.Join(root, "vendor"), 0o755); err != nil { + t.Fatalf("mkdir vendor: %v", err) + } + if err := os.WriteFile(filepath.Join(root, "vendor", "dependency.go"), []byte(strings.Repeat("// vendored source\n", 8)), 0o600); err != nil { + t.Fatalf("write vendored source: %v", err) + } + + for _, tc := range []struct { + name string + enabled bool + }{ + {name: "default"}, + {name: "opt-in", enabled: true}, + } { + t.Run(tc.name, func(t *testing.T) { + configPath := filepath.Join(root, "codeguard-"+tc.name+".json") + configJSON := `{ + "name": "vendored-target-root", + "targets": [{"name": "vendor", "path": "vendor", "language": "go"}], + "scan_vendored_source": ` + fmt.Sprintf("%t", tc.enabled) + `, + "checks": {"quality": true, "quality_rules": {"max_file_lines": 2}, "security_rules": {"govulncheck_mode": "off"}}, + "output": {"format": "json"} +}` + if err := os.WriteFile(configPath, []byte(configJSON), 0o600); err != nil { + t.Fatalf("write config: %v", err) + } + cfg, err := codeguard.LoadConfigFile(configPath) + if err != nil { + t.Fatalf("load config: %v", err) + } + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + t.Fatalf("full scan: %v", err) + } + if got := reportHasFindingPath(report, "dependency.go"); got != tc.enabled { + t.Fatalf("vendored finding present = %t, want %t", got, tc.enabled) + } + }) + } +} + +func TestRelativeConfigTargetPreservesTargetRelativeExcludes(t *testing.T) { + root := t.TempDir() + for rel, content := range map[string]string{ + "services/api/main.go": strings.Repeat("// owned source\n", 8), + "services/api/dist/bundle.go": strings.Repeat("// distribution output\n", 8), + "services/api/generated/code.go": strings.Repeat("// configured exclusion\n", 8), + } { + path := filepath.Join(root, filepath.FromSlash(rel)) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatalf("mkdir %s: %v", rel, err) + } + if err := os.WriteFile(path, []byte(content), 0o600); err != nil { + t.Fatalf("write %s: %v", rel, err) + } + } + configPath := filepath.Join(root, "codeguard.json") + configJSON := `{ + "name": "relative-target-excludes", + "targets": [{"name": "api", "path": "services/api", "language": "go"}], + "exclude": ["generated/**"], + "checks": {"quality": true, "quality_rules": {"max_file_lines": 2}, "security_rules": {"govulncheck_mode": "off"}}, + "output": {"format": "json"} +}` + if err := os.WriteFile(configPath, []byte(configJSON), 0o600); err != nil { + t.Fatalf("write config: %v", err) + } + cfg, err := codeguard.LoadConfigFile(configPath) + if err != nil { + t.Fatalf("load config: %v", err) + } + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + t.Fatalf("full scan: %v", err) + } + if !reportHasFindingPath(report, "main.go") { + t.Fatal("owned source was not scanned") + } + for _, excluded := range []string{"dist/bundle.go", "generated/code.go"} { + if reportHasFindingPath(report, excluded) { + t.Fatalf("target-relative exclusion did not exclude %s", excluded) + } + } +} + +func TestAbsoluteTargetRootHonorsDependencyPolicyWithoutInspectingAncestors(t *testing.T) { + root := t.TempDir() + for _, rel := range []string{ + "vendor/dependency.go", + "node_modules/dependency.go", + "cdk.out/dependency.go", + "unrelated/vendor/projects/app/main.go", + } { + path := filepath.Join(root, filepath.FromSlash(rel)) + if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { + t.Fatalf("mkdir %s: %v", rel, err) + } + if err := os.WriteFile(path, []byte(strings.Repeat("// source\n", 8)), 0o600); err != nil { + t.Fatalf("write %s: %v", rel, err) + } + } + + for _, tc := range []struct { + name string + target string + scanVendoredSource bool + wantFinding bool + }{ + {name: "vendor-default", target: "vendor"}, + {name: "vendor-opt-in", target: "vendor", scanVendoredSource: true, wantFinding: true}, + {name: "node-modules-hard-exclusion", target: "node_modules", scanVendoredSource: true}, + {name: "cdk-output-hard-exclusion", target: "cdk.out", scanVendoredSource: true}, + {name: "unrelated-vendor-ancestor", target: "unrelated/vendor/projects/app", wantFinding: true}, + } { + t.Run(tc.name, func(t *testing.T) { + cfg := codeguard.ExampleConfig() + cfg.Name = "absolute-target-root-" + tc.name + cfg.Targets = []codeguard.TargetConfig{{Name: "repo", Path: filepath.Join(root, filepath.FromSlash(tc.target)), Language: "go"}} + cfg.ScanVendoredSource = tc.scanVendoredSource + cfg.Checks.Quality = true + cfg.Checks.QualityRules.MaxFileLines = 2 + cfg.Checks.SecurityRules.GovulncheckMode = "off" + + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{Mode: codeguard.ScanModeFull}) + if err != nil { + t.Fatalf("full scan: %v", err) + } + findingPath := "dependency.go" + if tc.name == "unrelated-vendor-ancestor" { + findingPath = "main.go" + } + if got := reportHasFindingPath(report, findingPath); got != tc.wantFinding { + t.Fatalf("finding present = %t, want %t", got, tc.wantFinding) + } + }) + } +} + +func TestFullScanFolderCannotBypassVendoredSourceDefault(t *testing.T) { + root := t.TempDir() + vendorRoot := filepath.Join(root, "nested", "vendor") + vendoredPath := filepath.Join(vendorRoot, "dependency", "dependency.go") + if err := os.MkdirAll(filepath.Dir(vendoredPath), 0o755); err != nil { + t.Fatalf("mkdir vendor: %v", err) + } + if err := os.WriteFile(vendoredPath, []byte(strings.Repeat("// vendored source\n", 8)), 0o600); err != nil { + t.Fatalf("write vendored source: %v", err) + } + + cfg := codeguard.ExampleConfig() + cfg.Name = "vendored-source-folder-default" + cfg.Targets = []codeguard.TargetConfig{{Name: "repo", Path: root, Language: "go"}} + cfg.Checks.Quality = true + cfg.Checks.QualityRules.MaxFileLines = 2 + cfg.Checks.SecurityRules.GovulncheckMode = "off" + + report, err := codeguard.RunWithOptions(context.Background(), cfg, codeguard.ScanOptions{ + Mode: codeguard.ScanModeFull, + TargetPath: vendorRoot, + }) + if err != nil { + t.Fatalf("full folder scan: %v", err) + } + for _, section := range report.Sections { + for _, finding := range section.Findings { + if finding.Path == "dependency/dependency.go" { + t.Fatal("folder scan bypassed the default vendored-source exclusion") + } + } + } +} + +func reportHasFindingPath(report codeguard.Report, path string) bool { + for _, section := range report.Sections { + for _, finding := range section.Findings { + if finding.Path == path { + return true + } + } + } + return false +} + // An untrusted repository must not be able to exhaust scan memory with an // oversized file: files above the scan size cap are skipped by the walk so no // section ever reads them into memory. Normal-sized files are still listed.