All files / src TemplateMiner.ts

96.95% Statements 191/197
83.65% Branches 87/104
100% Functions 22/22
97.9% Lines 187/191

Press n or j to go to the next uncovered block, b, p or k for the previous block.

1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771                                                                    12x 12x             42x                         14x   14x 14x                                                                                                             132x                 132x                                                     132x 132x     132x     132x 132x                                           132x             132x   132x           132x           132x 1x         132x         132x 132x 2x           132x 132x 132x     132x           132x 28x 28x 4x 1x   3x 3x 2x   1x         24x                                                           3x 3x 3x   3x                                                                   1x     1x 1x 2x     2x 2x 2x       1x                             2x                                           6368x       6368x     6368x 6368x 6368x     6368x 6368x 1x 1x       1x       6368x 6368x 6368x       6368x 6368x       6368x 163x   6367x   6367x 6367x   6367x                                             6x     6x     6x 6x 1x 1x       1x     6x                                                                     28x         28x 28x   2x     28x   28x 28x   28x 25x 25x 25x 25x 25x     28x 28x   23x 23x 34x 34x 34x             23x                           2x 4x                                                             25x 25x   25x 30x 30x 30x     25x 25x                 25x 25x   25x 42x 42x   11x 11x       31x 30x   31x   31x 31x   1x 1x 1x     30x 30x   30x   30x   15x 15x 15x   14x 14x 14x     14x 14x 14x 14x       1x 1x         25x 25x     25x   25x               163x   27x     28x             27x 27x     27x 5x 5x         27x 27x 3x 2x 2x 1x   1x                                                     25x         7x 2x 2x   5x     7x   7x   7x     7x 6x     7x 8x 8x 8x 8x       7x 1x 1x 1x                           6368x 162x     6206x 6206x 6206x 1x 1x     6205x      
import { Drain } from "./core/Drain.js";
import { JaccardDrain } from "./core/JaccardDrain.js";
import type { DrainBase } from "./core/DrainBase.js";
import { LogCluster } from "./core/LogCluster.js";
import { LogMasker } from "./masker/LogMasker.js";
import type { MaskingInstruction } from "./masker/MaskingInstruction.js";
import { TemplateMinerConfig } from "./TemplateMinerConfig.js";
import { LRUCache } from "./LRUCache.js";
import { SimpleProfiler, NullProfiler, type Profiler } from "./Profiler.js";
import type { PersistenceHandler } from "./persistence/PersistenceHandler.js";
import {
  ChangeType,
  MatchStrategy,
  type AddLogResult,
  type MatchStrategy as IMatchStrategy,
  type ExtractedParameter,
} from "./core/types.js";
import type { LogCluster as ILogCluster } from "./core/LogCluster.js";
import {
  AdjacentConstantFusion,
  RegexCollapseNormalizer,
  RegexSubstitutionNormalizer,
  TokenNormalizerPipeline,
} from "./core/TokenNormalizer.js";
import {
  ClusterMergePipeline,
  PositionDiffMergeStrategy,
} from "./core/ClusterMergeStrategy.js";
import * as zlib from "node:zlib";
 
// ============================================================
// Helpers
// ============================================================
 
const encoder = new TextEncoder();
const decoder = new TextDecoder();
 
/**
 * Escapes special regex characters in a string.
 * Equivalent to Python's `re.escape()`.
 */
function escapeRegex(str: string): string {
  return str.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}
 
/**
 * Sanitizes a regex pattern for use inside a larger capture group.
 *
 * Python: Drain3's `_get_template_parameter_extraction_regex`
 * handles this by:
 * - Converting named groups `(?P<name>...)` to non-capturing groups `(?:...)`
 * - Converting numeric backreferences `\1` to `(?:.+?)`
 */
function sanitizeRegexForCapture(pattern: string): string {
  // Replace Python-style named groups: (?P<name>...) → (?:...)
  let sanitized = pattern.replace(/\(\?P<[^>]*>/g, "(?:");
  // Replace numeric backreferences \1, \2, etc. (exclude \0)
  sanitized = sanitized.replace(/\\(?!0)\d{1,2}/g, "(?:.+?)");
  return sanitized;
}
 
// ============================================================
// TemplateMiner
// ============================================================
 
/**
 * Main user-facing facade for log template mining.
 *
 * Maps 1:1 to Python `TemplateMiner` class (drain3/template_miner.py).
 *
 * TemplateMiner integrates the Drain clustering engine with the masking
 * preprocessor, optional persistence, and parameter extraction. It is
 * the single entry point that users should instantiate.
 *
 * Usage:
 * ```typescript
 * const miner = new TemplateMiner({
 *   config: TemplateMinerConfig.from({ simTh: 0.5 }),
 * });
 *
 * const result = miner.addLogMessage("user alice logged in from 192.168.1.1");
 * console.log(result.templateMined); // "user alice logged in from <IP>"
 * ```
 */
export class TemplateMiner {
  /** Configuration snapshot. */
  readonly config: TemplateMinerConfig;
 
  /** The Drain clustering engine (Drain or JaccardDrain). */
  readonly drain: DrainBase;
 
  /** The log masking preprocessor. */
  readonly masker: LogMasker;
 
  /** Optional persistence handler for state save/load. */
  private readonly _persistence: PersistenceHandler | null;
 
  /** Pre-clustering token normalization pipeline. */
  private readonly _normalizerPipeline: TokenNormalizerPipeline;
 
  /** Post-training cluster merge pipeline. */
  private readonly _mergePipeline: ClusterMergePipeline;
 
  /** LRU cache for parameter extraction regexes: (template, exactMatching) → compiled RegExp. */
  private readonly _extractionCache: LRUCache<string, RegExp>;
 
  /** LRU cache for param-name-to-mask-name mappings. Keyed same as _extractionCache. */
  private readonly _extractionMappingCache: LRUCache<string, Record<string, string>>;
 
  /** Profiler instance (NullProfiler by default, SimpleProfiler when enabled). */
  readonly profiler: Profiler;
 
  /** Timestamp (seconds) of the last snapshot save. Initialized to now to prevent immediate periodic save. */
  private _lastSnapshotTimestamp: number = Date.now() / 1000;
 
  /**
   * Promise that resolves when async state loading completes.
   * `null` if no persistence handler or if loading was synchronous.
   *
   * When using an async PersistenceHandler, prefer `TemplateMiner.create()`
   * over the constructor to ensure the model is fully loaded before use.
   */
  readonly initPromise: Promise<void> | null = null;
 
  /**
   * Creates a TemplateMiner.
   *
   * **Important**: When using an async `PersistenceHandler` (e.g., Redis,
   * Kafka), prefer the static `TemplateMiner.create()` factory method instead
   * of the constructor. The constructor returns immediately before async
   * state loading completes, which can cause a race condition if you call
   * `addLogMessage()` right away.
   *
   * @param options.config - Configuration object (defaults used if omitted).
   * @param options.persistenceHandler - Optional persistence backend.
   *
   * @example
   * ```typescript
   * // For async persistence, use the factory:
   * const miner = await TemplateMiner.create({ persistenceHandler: myRedisHandler });
   * ```
   */
  constructor({
    config = new TemplateMinerConfig(),
    persistenceHandler = null,
  }: {
    config?: TemplateMinerConfig;
    persistenceHandler?: PersistenceHandler | null | undefined;
  } = {}) {
    this.config = config;
    this._persistence = persistenceHandler;
 
    // Build paramStr from mask prefix/suffix: "<*>" by default
    const paramStr = `${config.maskPrefix}*${config.maskSuffix}`;
 
    // Create the Drain engine (Drain or JaccardDrain based on config.engine)
    const DrainCtor = config.engine === "JaccardDrain" ? JaccardDrain : Drain;
    this.drain = new DrainCtor({
      depth: config.depth,
      simTh: config.simTh,
      maxChildren: config.maxChildren,
      maxClusters: config.maxClusters,
      extraDelimiters: config.drainExtraDelimiters,
      paramStr,
      parametrizeNumericTokens: config.parametrizeNumericTokens,
      // Pass strategy chain configuration (conditionally for exactOptionalPropertyTypes)
      ...(config.templatePatternStrategies !== undefined
        ? { templatePatternStrategies: config.templatePatternStrategies }
        : {}),
      enableAffixPreserving: config.enableAffixPreserving,
      minAffixLength: config.minAffixLength,
      customRegexPatterns: config.customRegexPatterns,
      enableParamBinning: config.enableParamBinning,
      enableMaskParamGeneralization: config.enableMaskParamGeneralization,
      enableAELSimilarity: config.enableAELSimilarity,
      maxDiffRatio: config.maxDiffRatio,
    });
 
    // Create the masker with the configured instructions
    this.masker = new LogMasker(
      config.maskingInstructions,
      config.maskPrefix,
      config.maskSuffix,
    );
 
    // Build the token normalizer pipeline
    this._normalizerPipeline = new TokenNormalizerPipeline();
    // Phase 0: AEL-style regex substitution (runs first — per-token)
    Iif (config.aelRegexSubstitution.length > 0) {
      this._normalizerPipeline.register(
        new RegexSubstitutionNormalizer(config.aelRegexSubstitution),
      );
    }
    // Phase 1: Regex collapse (runs second — across joined tokens)
    Iif (config.regexCollapsePatterns.length > 0) {
      this._normalizerPipeline.register(
        new RegexCollapseNormalizer(config.regexCollapsePatterns),
      );
    }
    // Phase 2: Adjacent constant fusion (auto-detects and fuses constant pairs)
    if (config.enableAdjacentFusion) {
      this._normalizerPipeline.register(
        new AdjacentConstantFusion(config.minFusionTokenLength),
      );
    }
    // Phase 3: User-defined normalizers (runs last)
    for (const normalizer of config.tokenNormalizers) {
      this._normalizerPipeline.register(normalizer);
    }
 
    // Build the cluster merge pipeline
    this._mergePipeline = new ClusterMergePipeline();
    if (config.enableClusterMerge) {
      this._mergePipeline.register(
        new PositionDiffMergeStrategy(config.clusterMergePercent),
      );
    }
 
    // Initialize regex caches for parameter extraction
    const cacheCapacity = config.parameterExtractionCacheCapacity;
    this._extractionCache = new LRUCache(cacheCapacity);
    this._extractionMappingCache = new LRUCache(cacheCapacity);
 
    // Initialize profiler
    this.profiler = config.profilingEnabled
      ? new SimpleProfiler()
      : new NullProfiler();
 
    // Restore state from persistence if available.
    // If the handler is async, capture the loading promise so callers can await it.
    if (this._persistence) {
      const loadResult = this._persistence.loadState();
      if (loadResult instanceof Promise) {
        this.initPromise = loadResult.then(
          (buf) => { this._doLoad(buf); },
          (err: unknown) => {
            const error = err instanceof Error ? err : new Error(String(err));
            if (this.config.onError) {
              this.config.onError("loadState", error);
            } else {
              console.error("[drain-ts] Failed to load state:", error.message);
            }
          },
        );
      } else {
        this._doLoad(loadResult);
      }
    }
  }
 
  /**
   * Async factory — creates a fully-initialized TemplateMiner.
   *
   * Use this instead of the constructor when using an async
   * `PersistenceHandler` (e.g., Redis, Kafka, S3). This method
   * awaits state loading before returning, eliminating the race
   * condition between construction and `addLogMessage()` calls.
   *
   * For sync `PersistenceHandler` (FilePersistence, MemoryPersistence),
   * the constructor and `create()` behave identically.
   *
   * @example
   * ```typescript
   * const miner = await TemplateMiner.create({
   *   config: TemplateMinerConfig.from({ simTh: 0.5 }),
   *   persistenceHandler: new FilePersistence("/path/to/snapshot.json"),
   * });
   * // Model is fully loaded — safe to call addLogMessage() immediately.
   * miner.addLogMessage("first message");
   * ```
   */
  static async create(options: {
    config?: TemplateMinerConfig;
    persistenceHandler?: PersistenceHandler | null | undefined;
  } = {}): Promise<TemplateMiner> {
    const miner = new TemplateMiner(options);
    Eif (miner.initPromise) {
      await miner.initPromise;
    }
    return miner;
  }
 
  // ============================================================
  // learnTokens — batch learning for token normalizers
  // ============================================================
 
  /**
   * Learns token patterns from a batch of raw log messages.
   *
   * Must be called BEFORE processing any messages when using
   * normalizers that require a learning phase (e.g., AdjacentConstantFusion).
   *
   * This method tokenizes all messages, runs the normalizer's learn phase,
   * and resets the Drain engine. Messages are NOT added to clusters.
   *
   * Call this once, then call addLogMessage() for each message.
   *
   * @param messages - Raw log messages to learn from
   *
   * @example
   * ```typescript
   * const miner = new TemplateMiner({
   *   config: TemplateMinerConfig.from({
   *     enableAdjacentFusion: true,
   *   }),
   * });
   * miner.learnTokens(allLogMessages);
   * for (const msg of allLogMessages) {
   *   miner.addLogMessage(msg); // tokens are now normalized
   * }
   * ```
   */
  learnTokens(messages: readonly string[]): void {
    Iif (this._normalizerPipeline.isEmpty) return;
 
    // Tokenize all messages (with preprocessor + extra delimiters + masking)
    const tokenized: string[][] = [];
    for (const msg of messages) {
      const preprocessed = this.config.preprocessor
        ? this.config.preprocessor(msg)
        : msg;
      const masked = this.masker.mask(preprocessed);
      const tokens = this.drain.getContentAsTokens(masked);
      tokenized.push(tokens);
    }
 
    // Let normalizers learn from the batch
    this._normalizerPipeline.learn(tokenized);
  }
 
  /**
   * Post-training cluster merge (AEL reconcile).
   *
   * Applies the configured ClusterMergePipeline to merge near-identical
   * clusters that were split during training. This is a non-destructive
   * operation that only consolidates clusters — it never creates new ones.
   *
   * Call this after all addLogMessage() calls to improve Grouping Accuracy.
   *
   * @returns Number of clusters merged
   */
  mergeClusters(): number {
    return this._mergePipeline.size > 0
      ? this.drain.mergeClusters(this._mergePipeline)
      : 0;
  }
 
  // ============================================================
  // addLogMessage — maps to Python TemplateMiner.add_log_message()
  // ============================================================
 
  /**
   * Processes a log message (training mode).
   *
   * The message is first preprocessed and masked, then token-normalized,
   * then passed to the Drain engine for clustering.
   *
   * State may be persisted if a PersistenceHandler is configured and
   * a snapshot trigger condition is met.
   *
   * Python: TemplateMiner.add_log_message(log_message) → dict
   */
  addLogMessage(logMessage: string): AddLogResult {
    // Phase 0: Preprocess (dataset-specific normalization)
    const preprocessed = this.config.preprocessor
      ? this.config.preprocessor(logMessage)
      : logMessage;
 
    this.profiler.startSection("total");
 
    // Phase 1: Mask
    this.profiler.startSection("mask");
    const maskedContent = this.masker.mask(preprocessed);
    this.profiler.endSection("mask");
 
    // Phase 1.5: Token normalization (pre-clustering)
    let clusterInput = maskedContent;
    if (!this._normalizerPipeline.isEmpty) {
      const tokens = this.drain.getContentAsTokens(maskedContent);
      const normalized = this._normalizerPipeline.normalize(
        tokens,
        `${this.config.maskPrefix}*${this.config.maskSuffix}`,
      );
      clusterInput = normalized.tokens.join(" ");
    }
 
    // Phase 2: Cluster
    this.profiler.startSection("drain");
    const { cluster, changeType } = this.drain.addLogMessage(clusterInput);
    this.profiler.endSection("drain");
 
    // Phase 3: Conditional persistence
    // Python: self.profiler.start_section("save_state")
    this.profiler.startSection("save_state");
    const snapshotReason = this._getSnapshotReason(
      changeType,
      cluster.clusterId,
    );
    if (snapshotReason !== null) {
      this._saveState(snapshotReason);
    }
    this.profiler.endSection("save_state");
 
    this.profiler.endSection("total");
    this.profiler.report(this.config.profilingReportSec);
 
    return {
      changeType,
      clusterId: cluster.clusterId,
      clusterSize: cluster.size,
      templateMined: cluster.getTemplate(),
      clusterCount: this.drain.idToCluster.size,
    };
  }
 
  // ============================================================
  // match — maps to Python TemplateMiner.match()
  // ============================================================
 
  /**
   * Matches a log message against existing clusters (inference mode).
   *
   * Unlike `addLogMessage`, this does NOT create new clusters or modify
   * templates.
   */
  match(
    logMessage: string,
    fullSearchStrategy: IMatchStrategy = MatchStrategy.Never,
  ): ILogCluster | null {
    const preprocessed = this.config.preprocessor
      ? this.config.preprocessor(logMessage)
      : logMessage;
    const maskedContent = this.masker.mask(preprocessed);
 
    // Apply token normalization
    let matchInput = maskedContent;
    if (!this._normalizerPipeline.isEmpty) {
      const tokens = this.drain.getContentAsTokens(maskedContent);
      const normalized = this._normalizerPipeline.normalize(
        tokens,
        `${this.config.maskPrefix}*${this.config.maskSuffix}`,
      );
      matchInput = normalized.tokens.join(" ");
    }
 
    return this.drain.match(matchInput, fullSearchStrategy);
  }
 
  // ============================================================
  // extractParameters — maps to Python TemplateMiner.extract_parameters()
  // ============================================================
 
  /**
   * Extracts variable parameters from a log message based on its template.
   *
   * Python: TemplateMiner.extract_parameters(template, log_line, exact_matching)
   *
   * Given a mined template like `"user <*:> logged in from <:IP:>"` and the
   * original log message `"user alice logged in from 192.168.1.1"`, this
   * method returns the extracted parameter values with their mask names:
   *
   * ```
   * [
   *   { value: "alice", maskName: "*" },
   *   { value: "192.168.1.1", maskName: "IP" }
   * ]
   * ```
   *
   * @param logTemplate - The mined template string (from `addLogMessage` result).
   * @param logMessage - The original (unmasked) log message.
   * @param exactMatching - If true, uses the masking instruction regex patterns.
   *                        If false, uses non-whitespace matching `.+?` for all params.
   * @returns Ordered list of extracted parameters.
   */
  extractParameters(
    logTemplate: string,
    logMessage: string,
    exactMatching: boolean = true,
  ): ExtractedParameter[] {
    // Phase 0: Preprocess
    const preprocessed = this.config.preprocessor
      ? this.config.preprocessor(logMessage)
      : logMessage;
    // Preprocess: replace extra delimiters with spaces
    // Python: for delimiter in self.config.drain_extra_delimiters: log_message = re.sub(delimiter, " ", log_message)
    let processedMessage = preprocessed;
    for (const delimiter of this.config.drainExtraDelimiters) {
      // Use split+join instead of regex replace for plain string delimiters
      processedMessage = processedMessage.split(delimiter).join(" ");
    }
 
    const cacheKey = `${logTemplate}\x00${String(exactMatching)}`;
 
    let regex = this._extractionCache.get(cacheKey);
    let paramNameToMaskName = this._extractionMappingCache.get(cacheKey);
 
    if (!regex || !paramNameToMaskName) {
      const built = this._buildExtractionRegex(logTemplate, exactMatching);
      regex = built.regex;
      paramNameToMaskName = built.paramNameToMaskName;
      this._extractionCache.set(cacheKey, regex);
      this._extractionMappingCache.set(cacheKey, paramNameToMaskName);
    }
 
    const match = regex.exec(processedMessage);
    if (!match || !match.groups) return [];
 
    const result: ExtractedParameter[] = [];
    for (const paramName of Object.keys(paramNameToMaskName)) {
      const value = match.groups[paramName];
      Eif (value !== undefined) {
        result.push({
          value,
          maskName: paramNameToMaskName[paramName]!,
        });
      }
    }
 
    return result;
  }
 
  /**
   * Deprecated: use extractParameters() instead.
   *
   * Python: TemplateMiner.get_parameter_list(template, log_line)
   *
   * Extracts parameter VALUES only (no mask names) using inexact matching.
   * Provided for compatibility with Drain3 API.
   *
   * @deprecated Use `extractParameters()` for full ExtractedParameter[] results.
   */
  getParameterList(logTemplate: string, logMessage: string): string[] {
    const params = this.extractParameters(logTemplate, logMessage, false);
    return params.map((p) => p.value);
  }
 
  // ============================================================
  // Parameter extraction regex builder
  // ============================================================
 
  /**
   * Builds a compiled RegExp and param-name-to-mask-name mapping for
   * a given template.
   *
   * Python: TemplateMiner._get_template_parameter_extraction_regex()
   *
   * Algorithm:
   * 1. Escape the template for regex.
   * 2. For each known mask name, find `<MASK_NAME>` placeholders.
   * 3. Replace each placeholder with a named capture group:
   *    - Exact matching: use the MaskingInstruction's regex pattern(s).
   *    - Inexact matching or `*`: use `.+?`.
   * 4. Replace spaces with `\s+` to handle multiple spaces.
   * 5. Anchor with `^...$`.
   *
   * @returns Compiled regex and mapping from param group name to mask name.
   */
  private _buildExtractionRegex(
    template: string,
    exactMatching: boolean,
  ): {
    regex: RegExp;
    paramNameToMaskName: Record<string, string>;
  } {
    const paramNameToMaskName: Record<string, string> = {};
    let paramCounter = 0;
 
    const getNextParamName = (): string => {
      const name = `p_${paramCounter}`;
      paramCounter += 1;
      return name;
    };
 
    const prefix = this.config.maskPrefix;
    const suffix = this.config.maskSuffix;
 
    // Build the regex by splitting the template into parts:
    // literal text parts (escaped) and placeholder parts (replaced with capture groups).
    //
    // Strategy: tokenize the template at `<...>` boundaries, escape the literal
    // segments, and replace each placeholder with a named capture group.
    // This avoids the double-escaping problem that occurs when escaping the
    // entire template first and then trying to find placeholders within it.
    const parts: string[] = [];
    let remaining = template;
 
    while (remaining.length > 0) {
      const openIdx = remaining.indexOf(prefix);
      if (openIdx === -1) {
        // No more placeholders — escape the rest
        parts.push(escapeRegex(remaining));
        break;
      }
 
      // Literal text before placeholder
      if (openIdx > 0) {
        parts.push(escapeRegex(remaining.slice(0, openIdx)));
      }
      remaining = remaining.slice(openIdx + prefix.length);
 
      const closeIdx = remaining.indexOf(suffix);
      if (closeIdx === -1) {
        // No closing suffix — treat rest as literal
        parts.push(escapeRegex(prefix + remaining));
        remaining = "";
        break;
      }
 
      const maskName = remaining.slice(0, closeIdx);
      remaining = remaining.slice(closeIdx + suffix.length);
 
      const paramGroupName = getNextParamName();
 
      if (maskName === "*" || !exactMatching) {
        // Universal wildcard or inexact mode: match any characters
        paramNameToMaskName[paramGroupName] = maskName;
        parts.push(`(?<${paramGroupName}>.+?)`);
      } else if (this.masker.maskNames.includes(maskName)) {
        // Known mask name with exact matching
        paramNameToMaskName[paramGroupName] = maskName;
        const instructions = this.masker.instructionsByMaskName(maskName);
        Iif (instructions.length === 0) {
          parts.push(`(?<${paramGroupName}>.+?)`);
        } else {
          const patterns = instructions
            .filter((inst): inst is MaskingInstruction => "regexPattern" in inst)
            .map((inst) => sanitizeRegexForCapture(inst.regexPattern));
          parts.push(`(?<${paramGroupName}>${patterns.join("|")})`);
        }
      } else {
        // Unknown mask name — treat as generic wildcard
        paramNameToMaskName[paramGroupName] = maskName;
        parts.push(`(?<${paramGroupName}>.+?)`);
      }
    }
 
    // Join parts and replace spaces with \s+
    let templateRegex = parts.join("");
    templateRegex = templateRegex.replace(/ /g, "\\s+");
 
    // Anchor to start and end
    const finalRegex = new RegExp(`^${templateRegex}$`);
 
    return { regex: finalRegex, paramNameToMaskName };
  }
 
  // ============================================================
  // Persistence — maps to Python TemplateMiner.save_state/load_state
  // ============================================================
 
  private _saveState(snapshotReason: string): void {
    if (!this._persistence) return;
 
    const snapshot = {
      version: "0.1.1",
      clusters_counter: this.drain.clustersCounter,
      clusters: [...this.drain.idToCluster.values()].map((c) => ({
        cluster_id: c.clusterId,
        log_template_tokens: c.logTemplateTokens,
        size: c.size,
      })),
    };
 
    const json = JSON.stringify(snapshot);
    let state = encoder.encode(json);
 
    // Python: if config.snapshot_compress_state → zlib.compress + base64.b64encode
    if (this.config.snapshotCompressState) {
      const compressed = zlib.deflateSync(state);
      state = encoder.encode(
        Buffer.from(compressed).toString("base64"),
      );
    }
 
    const result = this._persistence.saveState(state);
    if (result instanceof Promise) {
      result.catch((err: unknown) => {
        const error = err instanceof Error ? err : new Error(String(err));
        if (this.config.onError) {
          this.config.onError(`saveState(${snapshotReason})`, error);
        } else {
          console.error(
            `[drain-ts] Failed to save state (${snapshotReason}):`,
            error.message,
          );
        }
      });
    }
  }
 
  /**
   * Restores state from a serialized snapshot.
   *
   * **Rebuild behavior**: The snapshot stores only per-cluster metadata
   * (`cluster_id`, `log_template_tokens`, `size`). The prefix tree is NOT
   * preserved — on load, each cluster is re-inserted via
   * `addSeqToPrefixTree()`, rebuilding the tree from scratch.
   *
   * This means:
   * - Load time is O(n) where n is the number of clusters
   * - The resulting tree structure may differ from the tree at save time
   *   (insertion order affects prefix tree shape)
   * - This is a deliberate simplification vs. Drain3's `jsonpickle` approach
   *   which serializes the complete Python object graph including tree structure.
   *
   * @param stateBuffer - Raw snapshot bytes (possibly compressed).
   */
  private _doLoad(stateBuffer: Uint8Array | null): void {
    if (!stateBuffer || stateBuffer.length === 0) return;
 
    let json: string;
 
    // Python: if compressed → zlib.decompress(base64.b64decode(state))
    if (this.config.snapshotCompressState) {
      const decoded = Buffer.from(decoder.decode(stateBuffer), "base64");
      json = decoder.decode(zlib.inflateSync(decoded));
    } else {
      json = decoder.decode(stateBuffer);
    }
 
    const snapshot = JSON.parse(json);
 
    Iif (!snapshot.clusters || !Array.isArray(snapshot.clusters)) return;
 
    this.drain.idToCluster.clear();
 
    // Restore clusters counter from snapshot (avoids O(n) re-scan)
    if (typeof snapshot.clusters_counter === "number") {
      this.drain.clustersCounter = snapshot.clusters_counter;
    }
 
    for (const c of snapshot.clusters) {
      const cluster = new LogCluster(c.log_template_tokens, c.cluster_id);
      cluster.size = c.size;
      this.drain.idToCluster.set(c.cluster_id, cluster);
      this.drain.addSeqToPrefixTree(this.drain.rootNode, cluster);
    }
 
    // Fallback for pre-0.1.1 snapshots without clusters_counter
    if (typeof snapshot.clusters_counter !== "number" && this.drain.clustersCounter === 0) {
      for (const c of snapshot.clusters) {
        Eif (c.cluster_id > this.drain.clustersCounter) {
          this.drain.clustersCounter = c.cluster_id;
        }
      }
    }
  }
 
  // ============================================================
  // Snapshot trigger logic
  // ============================================================
 
  private _getSnapshotReason(
    changeType: typeof ChangeType[keyof typeof ChangeType],
    clusterId: number,
  ): string | null {
    if (changeType !== ChangeType.None) {
      return `${changeType} (${clusterId})`;
    }
 
    const now = Date.now() / 1000;
    const elapsed = now - this._lastSnapshotTimestamp;
    if (elapsed >= this.config.snapshotIntervalMinutes * 60) {
      this._lastSnapshotTimestamp = now;
      return "periodic";
    }
 
    return null;
  }
}