From 61977f1d92a1c78a7bb4d3acd0d4729773c21181 Mon Sep 17 00:00:00 2001 From: Simon Holthausen Date: Thu, 11 May 2023 14:04:23 +0200 Subject: [PATCH] Convert src/compiler/parse/utils/html.ts to JavaScript --- package.json | 1 + pnpm-lock.yaml | 17 +++ src/compiler/parse/utils/html.ts | 230 +++++++++++++++---------------- 3 files changed, 133 insertions(+), 115 deletions(-) diff --git a/package.json b/package.json index a10a52feab..f28d2076e2 100644 --- a/package.json +++ b/package.json @@ -150,6 +150,7 @@ "prettier-plugin-svelte": "^2.10.0", "puppeteer": "^19.8.5", "rollup": "^3.20.2", + "rollup-plugin-dts": "^5.3.0", "source-map": "^0.7.4", "source-map-support": "^0.5.21", "tiny-glob": "^0.2.9", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index ed6da5e726..59e896a1da 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -115,6 +115,9 @@ devDependencies: rollup: specifier: ^3.20.2 version: 3.20.2 + rollup-plugin-dts: + specifier: ^5.3.0 + version: 5.3.0(rollup@3.20.2)(typescript@5.0.4) source-map: specifier: ^0.7.4 version: 0.7.4 @@ -2736,6 +2739,20 @@ packages: glob: 7.2.3 dev: true + /rollup-plugin-dts@5.3.0(rollup@3.20.2)(typescript@5.0.4): + resolution: {integrity: sha512-8FXp0ZkyZj1iU5klkIJYLjIq/YZSwBoERu33QBDxm/1yw5UU4txrEtcmMkrq+ZiKu3Q4qvPCNqc3ovX6rjqzbQ==} + engines: {node: '>=v14'} + peerDependencies: + rollup: ^3.0.0 + typescript: ^4.1 || ^5.0 + dependencies: + magic-string: 0.30.0 + rollup: 3.20.2 + typescript: 5.0.4 + optionalDependencies: + '@babel/code-frame': 7.21.4 + dev: true + /rollup@3.20.2: resolution: {integrity: sha512-3zwkBQl7Ai7MFYQE0y1MeQ15+9jsi7XxfrqwTb/9EK8D9C9+//EBR4M+CuA1KODRaNbFez/lWxA5vhEGZp4MUg==} engines: {node: '>=14.18.0', npm: '>=8.0.0'} diff --git a/src/compiler/parse/utils/html.ts b/src/compiler/parse/utils/html.ts index b569dd84d6..588c9aba78 100644 --- a/src/compiler/parse/utils/html.ts +++ b/src/compiler/parse/utils/html.ts @@ -1,140 +1,140 @@ -import entities from './entities'; - +import entities from './entities.js'; const windows_1252 = [ - 8364, 129, 8218, 402, 8222, 8230, 8224, 8225, 710, 8240, 352, 8249, 338, 141, 381, 143, 144, 8216, - 8217, 8220, 8221, 8226, 8211, 8212, 732, 8482, 353, 8250, 339, 157, 382, 376 + 8364, 129, 8218, 402, 8222, 8230, 8224, 8225, 710, 8240, 352, 8249, 338, 141, 381, 143, 144, 8216, + 8217, 8220, 8221, 8226, 8211, 8212, 732, 8482, 353, 8250, 339, 157, 382, 376 ]; -function reg_exp_entity(entity_name: string, is_attribute_value: boolean) { - // https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state - // doesn't decode the html entity which not ends with ; and next character is =, number or alphabet in attribute value. - if (is_attribute_value && !entity_name.endsWith(';')) { - return `${entity_name}\\b(?!=)`; - } - return entity_name; +/** + * @param {string} entity_name + * @param {boolean} is_attribute_value + */ +function reg_exp_entity(entity_name, is_attribute_value) { + // https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state + // doesn't decode the html entity which not ends with ; and next character is =, number or alphabet in attribute value. + if (is_attribute_value && !entity_name.endsWith(';')) { + return `${entity_name}\\b(?!=)`; + } + return entity_name; } -function get_entity_pattern(is_attribute_value: boolean) { - const reg_exp_num = '#(?:x[a-fA-F\\d]+|\\d+)(?:;)?'; - const reg_exp_entities = Object.keys(entities).map((entity_name) => - reg_exp_entity(entity_name, is_attribute_value) - ); - - const entity_pattern = new RegExp(`&(${reg_exp_num}|${reg_exp_entities.join('|')})`, 'g'); - - return entity_pattern; +/** + * @param {boolean} is_attribute_value + */ +function get_entity_pattern(is_attribute_value) { + const reg_exp_num = '#(?:x[a-fA-F\\d]+|\\d+)(?:;)?'; + const reg_exp_entities = Object.keys(entities).map((entity_name) => reg_exp_entity(entity_name, is_attribute_value)); + const entity_pattern = new RegExp(`&(${reg_exp_num}|${reg_exp_entities.join('|')})`, 'g'); + return entity_pattern; } - const entity_pattern_content = get_entity_pattern(false); const entity_pattern_attr_value = get_entity_pattern(true); -export function decode_character_references(html: string, is_attribute_value: boolean) { - const entity_pattern = is_attribute_value ? entity_pattern_attr_value : entity_pattern_content; - return html.replace(entity_pattern, (match, entity) => { - let code; - - // Handle named entities - if (entity[0] !== '#') { - code = entities[entity]; - } else if (entity[1] === 'x') { - code = parseInt(entity.substring(2), 16); - } else { - code = parseInt(entity.substring(1), 10); - } - - if (!code) { - return match; - } - - return String.fromCodePoint(validate_code(code)); - }); +/** + * @param {string} html + * @param {boolean} is_attribute_value + */ +export function decode_character_references(html, is_attribute_value) { + const entity_pattern = is_attribute_value ? entity_pattern_attr_value : entity_pattern_content; + return html.replace(entity_pattern, (match, entity) => { + let code; + // Handle named entities + if (entity[0] !== '#') { + code = entities[entity]; + } + else if (entity[1] === 'x') { + code = parseInt(entity.substring(2), 16); + } + else { + code = parseInt(entity.substring(1), 10); + } + if (!code) { + return match; + } + return String.fromCodePoint(validate_code(code)); + }); } - const NUL = 0; - // some code points are verboten. If we were inserting HTML, the browser would replace the illegal // code points with alternatives in some cases - since we're bypassing that mechanism, we need // to replace them ourselves // // Source: http://en.wikipedia.org/wiki/Character_encodings_in_HTML#Illegal_characters -function validate_code(code: number) { - // line feed becomes generic whitespace - if (code === 10) { - return 32; - } - - // ASCII range. (Why someone would use HTML entities for ASCII characters I don't know, but...) - if (code < 128) { - return code; - } - - // code points 128-159 are dealt with leniently by browsers, but they're incorrect. We need - // to correct the mistake or we'll end up with missing € signs and so on - if (code <= 159) { - return windows_1252[code - 128]; - } - // basic multilingual plane - if (code < 55296) { - return code; - } - - // UTF-16 surrogate halves - if (code <= 57343) { - return NUL; - } - - // rest of the basic multilingual plane - if (code <= 65535) { - return code; - } - - // supplementary multilingual plane 0x10000 - 0x1ffff - if (code >= 65536 && code <= 131071) { - return code; - } - - // supplementary ideographic plane 0x20000 - 0x2ffff - if (code >= 131072 && code <= 196607) { - return code; - } - - return NUL; +/** + * @param {number} code + */ +function validate_code(code) { + // line feed becomes generic whitespace + if (code === 10) { + return 32; + } + // ASCII range. (Why someone would use HTML entities for ASCII characters I don't know, but...) + if (code < 128) { + return code; + } + // code points 128-159 are dealt with leniently by browsers, but they're incorrect. We need + // to correct the mistake or we'll end up with missing € signs and so on + if (code <= 159) { + return windows_1252[code - 128]; + } + // basic multilingual plane + if (code < 55296) { + return code; + } + // UTF-16 surrogate halves + if (code <= 57343) { + return NUL; + } + // rest of the basic multilingual plane + if (code <= 65535) { + return code; + } + // supplementary multilingual plane 0x10000 - 0x1ffff + if (code >= 65536 && code <= 131071) { + return code; + } + // supplementary ideographic plane 0x20000 - 0x2ffff + if (code >= 131072 && code <= 196607) { + return code; + } + return NUL; } - // based on http://developers.whatwg.org/syntax.html#syntax-tag-omission const disallowed_contents = new Map([ - ['li', new Set(['li'])], - ['dt', new Set(['dt', 'dd'])], - ['dd', new Set(['dt', 'dd'])], - [ - 'p', - new Set( - 'address article aside blockquote div dl fieldset footer form h1 h2 h3 h4 h5 h6 header hgroup hr main menu nav ol p pre section table ul'.split( - ' ' - ) - ) - ], - ['rt', new Set(['rt', 'rp'])], - ['rp', new Set(['rt', 'rp'])], - ['optgroup', new Set(['optgroup'])], - ['option', new Set(['option', 'optgroup'])], - ['thead', new Set(['tbody', 'tfoot'])], - ['tbody', new Set(['tbody', 'tfoot'])], - ['tfoot', new Set(['tbody'])], - ['tr', new Set(['tr', 'tbody'])], - ['td', new Set(['td', 'th', 'tr'])], - ['th', new Set(['td', 'th', 'tr'])] + ['li', new Set(['li'])], + ['dt', new Set(['dt', 'dd'])], + ['dd', new Set(['dt', 'dd'])], + [ + 'p', + new Set('address article aside blockquote div dl fieldset footer form h1 h2 h3 h4 h5 h6 header hgroup hr main menu nav ol p pre section table ul'.split(' ')) + ], + ['rt', new Set(['rt', 'rp'])], + ['rp', new Set(['rt', 'rp'])], + ['optgroup', new Set(['optgroup'])], + ['option', new Set(['option', 'optgroup'])], + ['thead', new Set(['tbody', 'tfoot'])], + ['tbody', new Set(['tbody', 'tfoot'])], + ['tfoot', new Set(['tbody'])], + ['tr', new Set(['tr', 'tbody'])], + ['td', new Set(['td', 'th', 'tr'])], + ['th', new Set(['td', 'th', 'tr'])] ]); - // can this be a child of the parent element, or does it implicitly // close it, like `
  • one
  • two`? -export function closing_tag_omitted(current: string, next?: string) { - if (disallowed_contents.has(current)) { - if (!next || disallowed_contents.get(current).has(next)) { - return true; - } - } - return false; +/** + * @param {string} current + * @param {string} next + */ +export function closing_tag_omitted(current, next) { + if (disallowed_contents.has(current)) { + if (!next || disallowed_contents.get(current).has(next)) { + return true; + } + } + return false; } + + + +