layerEntropy function
List<RedactionMatch>
layerEntropy(
- String text,
- RedactionConfig cfg, {
- List<
RedactionMatch> prior = const [],
Finds high-entropy ASCII token spans in text.
prior holds matches from higher-priority layers; tokens inside those
spans are skipped.
Implementation
List<RedactionMatch> layerEntropy(
String text,
RedactionConfig cfg, {
List<RedactionMatch> prior = const [],
}) {
final minLength = cfg.minLength;
if (text.length < minLength) return const [];
final pattern = RegExp('[A-Za-z0-9+=_-]{$minLength,}');
final matches = <RedactionMatch>[];
for (final m in pattern.allMatches(text)) {
final token = m.group(0)!;
// Alphabet-length guard: a real secret never uses a tiny alphabet.
if (token.codeUnits.toSet().length < 8) continue;
if (shannonEntropy(token) < cfg.minEntropy) continue;
if (isStructuredIdentifier(token)) continue;
// A token glued after `/` is a path component, not a standalone
// value - real secrets do not live in directory names.
if (m.start > 0 && text[m.start - 1] == '/') continue;
final match = RedactionMatch(
start: m.start,
end: m.end,
layer: RedactionLayer.entropy,
kindLabel: highEntropyLabel,
);
if (prior.any(match.overlaps) || matches.any(match.overlaps)) continue;
matches.add(match);
}
return matches;
}