layerEntropy function

List<RedactionMatch> layerEntropy(
  1. String text,
  2. RedactionConfig cfg, {
  3. List<RedactionMatch> prior = const [],
})

Finds high-entropy ASCII token spans in text.

prior holds matches from higher-priority layers; tokens inside those spans are skipped.

Implementation

List<RedactionMatch> layerEntropy(
  String text,
  RedactionConfig cfg, {
  List<RedactionMatch> prior = const [],
}) {
  final minLength = cfg.minLength;
  if (text.length < minLength) return const [];
  final pattern = RegExp('[A-Za-z0-9+=_-]{$minLength,}');
  final matches = <RedactionMatch>[];
  for (final m in pattern.allMatches(text)) {
    final token = m.group(0)!;
    // Alphabet-length guard: a real secret never uses a tiny alphabet.
    if (token.codeUnits.toSet().length < 8) continue;
    if (shannonEntropy(token) < cfg.minEntropy) continue;
    if (isStructuredIdentifier(token)) continue;
    // A token glued after `/` is a path component, not a standalone
    // value - real secrets do not live in directory names.
    if (m.start > 0 && text[m.start - 1] == '/') continue;
    final match = RedactionMatch(
      start: m.start,
      end: m.end,
      layer: RedactionLayer.entropy,
      kindLabel: highEntropyLabel,
    );
    if (prior.any(match.overlaps) || matches.any(match.overlaps)) continue;
    matches.add(match);
  }
  return matches;
}