All files / lib/bundler scanner.js

98.23% Statements 389/396
94.4% Branches 118/125
100% Functions 8/8
98.23% Lines 389/396

Press n or j to go to the next uncovered block, b, p or k for the previous block.

1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 14115x 14115x 3592x 3592x 14115x     14115x 14115x 14115x 13423x 13423x 692x 14115x 605x 605x 4274x 4274x 605x 605x 87x 87x 14115x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 18047x 18047x 18047x 18047x 18047x 18047x 70561273x 70561273x 70561273x 139573x 139573x 8443081x 8443081x 139573x 139573x 139573x 70421700x 70561273x 192384x 192384x 192384x 45653175x 45653175x 192384x 192384x 192384x 192384x 70229316x 70561273x 545835x 545835x 545835x 545835x 9450953x 29617x 29617x 29617x 9450953x 545834x 545834x 545834x 8875502x 8875502x 8875502x 9450953x 1x 1x 8875501x 8875501x 545835x 545835x 545835x 69683481x 70561273x 29366x 29366x 29366x 29366x 2383938x 2546x 2546x 2546x 2383938x 41343x 41343x 41343x 41343x 2383938x 500197x 500196x 458852x 110x 110x 110x 110x 110x 110x 110x 500087x 500087x 500087x 2383938x 29343x 29343x 29343x 1810509x 1810509x 29366x 29366x 29366x 69654115x 70561273x 13678x 13678x 13678x 13678x 203425x 203425x 19487x 19487x 19487x 203425x 2x 2x 203425x 176794x 169907x 13676x 13676x 10148x 10148x 13676x 13676x 170260x 170260x 13678x 13678x 13678x 69640437x 69640437x 69640437x 18047x 18047x 18047x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 17937x 911938x 911938x 911938x 911938x 66272043x 66272043x 911938x 17937x 281x 281x 281x 281x 281x 281x 281x 73491630x 73491630x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 5219675x 5219675x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 6577x 6577x 453782x 278974x 278974x 453782x 29103849x 29103849x 174808x 6577x 6577x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 3798x 3798x 3798x 3798x 3798x 161905x 161905x 161905x 158107x 3798x 3798x 3798x 161905x   3798x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 281x 36892x 36892x 36892x 36892x 55377573x 19739569x 19739569x 19739569x 35638004x 35638004x 55377573x 1053144x 55377573x 1053144x 1053144x 53495x 53495x 53495x 53495x 53495x 53495x 24693x 24693x 53495x 1053144x     34584860x 3824x 3824x 35601112x 35601112x 35601112x     36892x  
/**
 * @file scanner.js
 * @description Source-position primitives shared by every part of the bundler.
 *
 * ## Why Avenx scans rather than parses
 *
 * Everything else in this compiler reads source with a purpose-built scanner —
 * the HTML tokenizer, the declaration reader, the Atlas reference scanner — and
 * the bundler follows the same rule for the same reason: a dependency-free
 * build is a property of the project, not an accident of what was convenient.
 *
 * The bundler does not need a full ECMAScript parser. It needs to know three
 * things about a module: which `import`/`export` declarations it contains,
 * where each top-level statement begins and ends, and which top-level names it
 * declares. All three are answerable from a scanner that knows where source
 * text *is not* code.
 *
 * ## What "is not code" means here
 *
 * String literals, template literals (including nested `${}` expressions),
 * regular-expression literals, line comments and block comments. Every one of
 * them can contain the characters this file searches for — a `{` inside a
 * string would otherwise corrupt every depth calculation downstream, and the
 * word `import` inside a comment would otherwise become a phantom dependency.
 *
 * Distinguishing `/` as division from `/` as the start of a regex is the one
 * genuinely ambiguous case in JavaScript lexing. {@link regexAllowedAfter}
 * resolves it the way every hand-written JS lexer does: by looking at the last
 * significant token.
 * @module lib/bundler/scanner
 */
 
/**
 * Characters that can legally precede a regular-expression literal.
 *
 * After any of these, a `/` starts a regex; after an identifier, a number, or a
 * closing bracket it is division. The exceptions to "closing bracket means
 * division" are keywords such as `return` and `typeof`, handled by
 * {@link KEYWORDS_BEFORE_REGEX}.
 * @type {Set<string>}
 */
const PUNCTUATORS_BEFORE_REGEX = new Set([
  '(', ',', '=', ':', '[', '!', '&', '|', '?', '{', '}', ';', '+', '-', '*', '%', '^', '~', '<', '>',
]);
 
/**
 * Keywords after which a `/` begins a regular expression rather than division.
 * @type {Set<string>}
 */
const KEYWORDS_BEFORE_REGEX = new Set([
  'return', 'typeof', 'instanceof', 'in', 'of', 'new', 'delete', 'void', 'throw',
  'case', 'do', 'else', 'yield', 'await',
]);
 
/**
 * Decides whether a `/` at `index` opens a regular-expression literal.
 * @param {string} source - The full source text.
 * @param {number} index - Offset of the `/`.
 * @returns {boolean} True when a regex literal starts here.
 */
export function regexAllowedAfter(source, index) {
  let i = index - 1;
  while (i >= 0 && /\s/.test(source[i])) {
    i -= 1;
  }
  if (i < 0) {
    return true;
  }
 
  const char = source[i];
  if (PUNCTUATORS_BEFORE_REGEX.has(char)) {
    return true;
  }
 
  if (/[\w$]/.test(char)) {
    let start = i;
    while (start >= 0 && /[\w$]/.test(source[start])) {
      start -= 1;
    }
    return KEYWORDS_BEFORE_REGEX.has(source.slice(start + 1, i + 1));
  }
 
  return false;
}
 
/**
 * A single lexical region the scanner recognises.
 * @typedef {object} Region
 * @property {'code'|'string'|'template'|'regex'|'line-comment'|'block-comment'} kind - What the region is.
 * @property {number} start - Inclusive start offset.
 * @property {number} end - Exclusive end offset.
 */
 
/**
 * Walks a source and reports every non-code region.
 *
 * Template literals are reported as a single region spanning the whole literal,
 * including any `${}` substitutions. Code inside a substitution is therefore
 * not scanned at all — which is correct for this bundler's purposes, because an
 * `import` declaration cannot appear inside an expression and a top-level
 * statement boundary cannot fall inside a template literal.
 * @param {string} source - The source text.
 * @returns {Region[]} Regions in ascending order of `start`.
 */
export function scanRegions(source) {
  /** @type {Region[]} */
  const regions = [];
  const length = source.length;
  let i = 0;
 
  while (i < length) {
    const char = source[i];
 
    if (char === '/' && source[i + 1] === '/') {
      const start = i;
      while (i < length && source[i] !== '\n') {
        i += 1;
      }
      regions.push({ kind: 'line-comment', start, end: i });
      continue;
    }
 
    if (char === '/' && source[i + 1] === '*') {
      const start = i;
      i += 2;
      while (i < length && !(source[i] === '*' && source[i + 1] === '/')) {
        i += 1;
      }
      i = Math.min(i + 2, length);
      regions.push({ kind: 'block-comment', start, end: i });
      continue;
    }
 
    if (char === '"' || char === "'") {
      const start = i;
      const quote = char;
      i += 1;
      while (i < length) {
        if (source[i] === '\\') {
          i += 2;
          continue;
        }
        if (source[i] === quote) {
          i += 1;
          break;
        }
        // An unterminated string literal cannot span a line. Stopping at the
        // newline keeps a malformed file from swallowing the rest of the
        // module and reporting nonsense about its imports.
        if (source[i] === '\n') {
          break;
        }
        i += 1;
      }
      regions.push({ kind: 'string', start, end: i });
      continue;
    }
 
    if (char === '`') {
      const start = i;
      i += 1;
      let depth = 0;
      while (i < length) {
        if (source[i] === '\\') {
          i += 2;
          continue;
        }
        if (depth === 0 && source[i] === '$' && source[i + 1] === '{') {
          depth += 1;
          i += 2;
          continue;
        }
        if (depth > 0) {
          if (source[i] === '{') depth += 1;
          else if (source[i] === '}') depth -= 1;
          else if (source[i] === '`') {
            // A nested template inside a substitution. Skip it whole so its
            // own braces cannot unbalance this one.
            const nested = scanRegions(source.slice(i));
            const first = nested[0];
            i += first && first.kind === 'template' && first.start === 0 ? first.end : 1;
            continue;
          }
          i += 1;
          continue;
        }
        if (source[i] === '`') {
          i += 1;
          break;
        }
        i += 1;
      }
      regions.push({ kind: 'template', start, end: i });
      continue;
    }
 
    if (char === '/' && regexAllowedAfter(source, i)) {
      const start = i;
      i += 1;
      let inClass = false;
      while (i < length) {
        const current = source[i];
        if (current === '\\') {
          i += 2;
          continue;
        }
        if (current === '\n') {
          break;
        }
        if (current === '[') inClass = true;
        else if (current === ']') inClass = false;
        else if (current === '/' && !inClass) {
          i += 1;
          while (i < length && /[a-z]/i.test(source[i])) {
            i += 1;
          }
          break;
        }
        i += 1;
      }
      regions.push({ kind: 'regex', start, end: i });
      continue;
    }
 
    i += 1;
  }
 
  return regions;
}
 
/**
 * Mask value for a comment: not code, and its whitespace carries no meaning.
 * @type {number}
 */
const COMMENT = 1;
 
/**
 * Mask value for a string, template or regex literal: not code, and its text
 * is data that must survive verbatim.
 * @type {number}
 */
const LITERAL = 2;
 
/**
 * A lookup answering "is this offset inside a string, comment or literal?".
 *
 * Built once per module and consulted many times. A flat `Uint8Array` costs one
 * byte per source character and turns every query into an array read, which
 * matters because the export scanner asks the question for every identifier in
 * a module.
 */
export class CodeMask {
  /**
   * @param {string} source - The source the mask describes.
   */
  constructor(source) {
    /** @type {string} */
    this.source = source;
    /** @type {Region[]} */
    this.regions = scanRegions(source);
    // One byte per character: 0 for code, 1 for a comment, 2 for a literal
    // whose text is data. Distinguishing the last two is what lets a minifier
    // blank a comment's whitespace and leave a template literal's alone, and it
    // has to be a lookup rather than a search -- the question is asked once per
    // character of the bundle.
    /** @type {Uint8Array} */
    this.mask = new Uint8Array(source.length);
 
    for (const region of this.regions) {
      const end = Math.min(region.end, source.length);
      const value =
        region.kind === 'string' || region.kind === 'template' || region.kind === 'regex' ? LITERAL : COMMENT;
      for (let i = region.start; i < end; i += 1) {
        this.mask[i] = value;
      }
    }
  }
 
  /**
   * Whether an offset holds executable code rather than literal or comment text.
   * @param {number} index - The offset to test.
   * @returns {boolean} True when the offset is code.
   */
  isCode(index) {
    return index >= 0 && index < this.mask.length && this.mask[index] === 0;
  }
 
  /**
   * Whether an offset sits inside a literal whose text is data.
   *
   * Distinct from {@link CodeMask#isCode}: a comment is not code, but its
   * whitespace is still safe to remove, whereas a string or template literal
   * carries its spacing as a value. Anything that rewrites whitespace has to
   * tell those two apart.
   * @param {number} index - The offset to test.
   * @returns {boolean} True inside a string, template or regex literal.
   */
  isLiteral(index) {
    return index >= 0 && index < this.mask.length && this.mask[index] === LITERAL;
  }
 
  /**
   * Returns the source with every comment replaced by equivalent whitespace.
   *
   * Offsets are preserved, so a position found in the blanked text is valid in
   * the original. String and template contents are left alone: they are values,
   * not noise, and blanking them would corrupt the module.
   * @returns {string} The comment-free source, same length as the original.
   */
  withoutComments() {
    const chars = this.source.split('');
    for (const region of this.regions) {
      if (region.kind !== 'line-comment' && region.kind !== 'block-comment') {
        continue;
      }
      for (let i = region.start; i < Math.min(region.end, chars.length); i += 1) {
        chars[i] = chars[i] === '\n' ? '\n' : ' ';
      }
    }
    return chars.join('');
  }
}
 
/**
 * Finds the offset just past a balanced bracket run starting at `index`.
 * @param {string} source - The source text.
 * @param {CodeMask} mask - Mask for the same source.
 * @param {number} index - Offset of the opening bracket.
 * @returns {number} Offset one past the matching close, or `source.length`.
 */
export function matchBracket(source, mask, index) {
  const open = source[index];
  const close = open === '(' ? ')' : open === '[' ? ']' : '}';
  let depth = 0;
 
  for (let i = index; i < source.length; i += 1) {
    if (!mask.isCode(i)) continue;
    const char = source[i];
    if (char === open) depth += 1;
    else if (char === close) {
      depth -= 1;
      if (depth === 0) return i + 1;
    }
  }
  return source.length;
}
 
/**
 * Finds the end of the top-level statement that begins at `start`.
 *
 * "End" means the offset one past its terminating semicolon, or one past the
 * closing brace of a block-bodied declaration, or the end of the line for an
 * ASI-terminated statement. Only bracket depth at code positions is consulted,
 * so a `;` inside a string or an object literal never terminates anything.
 * @param {string} source - The source text.
 * @param {CodeMask} mask - Mask for the same source.
 * @param {number} start - Offset of the statement's first character.
 * @returns {number} Exclusive end offset.
 */
export function statementEnd(source, mask, start) {
  let depth = 0;
  let i = start;
 
  while (i < source.length) {
    if (!mask.isCode(i)) {
      i += 1;
      continue;
    }
    const char = source[i];
 
    if (char === '(' || char === '[' || char === '{') {
      depth += 1;
    } else if (char === ')' || char === ']' || char === '}') {
      depth -= 1;
      if (depth === 0) {
        // A declaration whose body is a block ends at that block's close,
        // unless a semicolon or an operator follows on the same statement.
        let j = i + 1;
        while (j < source.length && /[ \t]/.test(source[j])) j += 1;
        if (source[j] === ';') return j + 1;
        if (j >= source.length || source[j] === '\n' || source[j] === '\r') {
          return i + 1;
        }
      }
      if (depth < 0) {
        return i;
      }
    } else if (char === ';' && depth === 0) {
      return i + 1;
    }
 
    i += 1;
  }

  return source.length;
}