@ -1,7 +1,7 @@
import entities from './entities.js' ;
const windows _1252 = [
8364 , 129 , 8218 , 402 , 8222 , 8230 , 8224 , 8225 , 710 , 8240 , 352 , 8249 , 338 , 141 , 381 , 143 , 144 , 8216 ,
8217 , 8220 , 8221 , 8226 , 8211 , 8212 , 732 , 8482 , 353 , 8250 , 339 , 157 , 382 , 376
8364 , 129 , 8218 , 402 , 8222 , 8230 , 8224 , 8225 , 710 , 8240 , 352 , 8249 , 338 , 141 , 381 , 143 , 144 , 8216 ,
8217 , 8220 , 8221 , 8226 , 8211 , 8212 , 732 , 8482 , 353 , 8250 , 339 , 157 , 382 , 376
] ;
/ * *
@ -9,22 +9,24 @@ const windows_1252 = [
* @ param { boolean } is _attribute _value
* /
function reg _exp _entity ( entity _name , is _attribute _value ) {
// https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state
// doesn't decode the html entity which not ends with ; and next character is =, number or alphabet in attribute value.
if ( is _attribute _value && ! entity _name . endsWith ( ';' ) ) {
return ` ${ entity _name } \\ b(?!=) ` ;
}
return entity _name ;
// https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state
// doesn't decode the html entity which not ends with ; and next character is =, number or alphabet in attribute value.
if ( is _attribute _value && ! entity _name . endsWith ( ';' ) ) {
return ` ${ entity _name } \\ b(?!=) ` ;
}
return entity _name ;
}
/ * *
* @ param { boolean } is _attribute _value
* /
function get _entity _pattern ( is _attribute _value ) {
const reg _exp _num = '#(?:x[a-fA-F\\d]+|\\d+)(?:;)?' ;
const reg _exp _entities = Object . keys ( entities ) . map ( ( entity _name ) => reg _exp _entity ( entity _name , is _attribute _value ) ) ;
const entity _pattern = new RegExp ( ` &( ${ reg _exp _num } | ${ reg _exp _entities . join ( '|' ) } ) ` , 'g' ) ;
return entity _pattern ;
const reg _exp _num = '#(?:x[a-fA-F\\d]+|\\d+)(?:;)?' ;
const reg _exp _entities = Object . keys ( entities ) . map ( ( entity _name ) =>
reg _exp _entity ( entity _name , is _attribute _value )
) ;
const entity _pattern = new RegExp ( ` &( ${ reg _exp _num } | ${ reg _exp _entities . join ( '|' ) } ) ` , 'g' ) ;
return entity _pattern ;
}
const entity _pattern _content = get _entity _pattern ( false ) ;
const entity _pattern _attr _value = get _entity _pattern ( true ) ;
@ -34,24 +36,22 @@ const entity_pattern_attr_value = get_entity_pattern(true);
* @ param { boolean } is _attribute _value
* /
export function decode _character _references ( html , is _attribute _value ) {
const entity _pattern = is _attribute _value ? entity _pattern _attr _value : entity _pattern _content ;
return html . replace ( entity _pattern , ( match , entity ) => {
let code ;
// Handle named entities
if ( entity [ 0 ] !== '#' ) {
code = entities [ entity ] ;
}
else if ( entity [ 1 ] === 'x' ) {
code = parseInt ( entity . substring ( 2 ) , 16 ) ;
}
else {
code = parseInt ( entity . substring ( 1 ) , 10 ) ;
}
if ( ! code ) {
return match ;
}
return String . fromCodePoint ( validate _code ( code ) ) ;
} ) ;
const entity _pattern = is _attribute _value ? entity _pattern _attr _value : entity _pattern _content ;
return html . replace ( entity _pattern , ( match , entity ) => {
let code ;
// Handle named entities
if ( entity [ 0 ] !== '#' ) {
code = entities [ entity ] ;
} else if ( entity [ 1 ] === 'x' ) {
code = parseInt ( entity . substring ( 2 ) , 16 ) ;
} else {
code = parseInt ( entity . substring ( 1 ) , 10 ) ;
}
if ( ! code ) {
return match ;
}
return String . fromCodePoint ( validate _code ( code ) ) ;
} ) ;
}
const NUL = 0 ;
// some code points are verboten. If we were inserting HTML, the browser would replace the illegal
@ -64,60 +64,64 @@ const NUL = 0;
* @ param { number } code
* /
function validate _code ( code ) {
// line feed becomes generic whitespace
if ( code === 10 ) {
return 32 ;
}
// ASCII range. (Why someone would use HTML entities for ASCII characters I don't know, but...)
if ( code < 128 ) {
return code ;
}
// code points 128-159 are dealt with leniently by browsers, but they're incorrect. We need
// to correct the mistake or we'll end up with missing € signs and so on
if ( code <= 159 ) {
return windows _1252 [ code - 128 ] ;
}
// basic multilingual plane
if ( code < 55296 ) {
return code ;
}
// UTF-16 surrogate halves
if ( code <= 57343 ) {
return NUL ;
}
// rest of the basic multilingual plane
if ( code <= 65535 ) {
return code ;
}
// supplementary multilingual plane 0x10000 - 0x1ffff
if ( code >= 65536 && code <= 131071 ) {
return code ;
}
// supplementary ideographic plane 0x20000 - 0x2ffff
if ( code >= 131072 && code <= 196607 ) {
return code ;
}
return NUL ;
// line feed becomes generic whitespace
if ( code === 10 ) {
return 32 ;
}
// ASCII range. (Why someone would use HTML entities for ASCII characters I don't know, but...)
if ( code < 128 ) {
return code ;
}
// code points 128-159 are dealt with leniently by browsers, but they're incorrect. We need
// to correct the mistake or we'll end up with missing € signs and so on
if ( code <= 159 ) {
return windows _1252 [ code - 128 ] ;
}
// basic multilingual plane
if ( code < 55296 ) {
return code ;
}
// UTF-16 surrogate halves
if ( code <= 57343 ) {
return NUL ;
}
// rest of the basic multilingual plane
if ( code <= 65535 ) {
return code ;
}
// supplementary multilingual plane 0x10000 - 0x1ffff
if ( code >= 65536 && code <= 131071 ) {
return code ;
}
// supplementary ideographic plane 0x20000 - 0x2ffff
if ( code >= 131072 && code <= 196607 ) {
return code ;
}
return NUL ;
}
// based on http://developers.whatwg.org/syntax.html#syntax-tag-omission
const disallowed _contents = new Map ( [
[ 'li' , new Set ( [ 'li' ] ) ] ,
[ 'dt' , new Set ( [ 'dt' , 'dd' ] ) ] ,
[ 'dd' , new Set ( [ 'dt' , 'dd' ] ) ] ,
[
'p' ,
new Set ( 'address article aside blockquote div dl fieldset footer form h1 h2 h3 h4 h5 h6 header hgroup hr main menu nav ol p pre section table ul' . split ( ' ' ) )
] ,
[ 'rt' , new Set ( [ 'rt' , 'rp' ] ) ] ,
[ 'rp' , new Set ( [ 'rt' , 'rp' ] ) ] ,
[ 'optgroup' , new Set ( [ 'optgroup' ] ) ] ,
[ 'option' , new Set ( [ 'option' , 'optgroup' ] ) ] ,
[ 'thead' , new Set ( [ 'tbody' , 'tfoot' ] ) ] ,
[ 'tbody' , new Set ( [ 'tbody' , 'tfoot' ] ) ] ,
[ 'tfoot' , new Set ( [ 'tbody' ] ) ] ,
[ 'tr' , new Set ( [ 'tr' , 'tbody' ] ) ] ,
[ 'td' , new Set ( [ 'td' , 'th' , 'tr' ] ) ] ,
[ 'th' , new Set ( [ 'td' , 'th' , 'tr' ] ) ]
[ 'li' , new Set ( [ 'li' ] ) ] ,
[ 'dt' , new Set ( [ 'dt' , 'dd' ] ) ] ,
[ 'dd' , new Set ( [ 'dt' , 'dd' ] ) ] ,
[
'p' ,
new Set (
'address article aside blockquote div dl fieldset footer form h1 h2 h3 h4 h5 h6 header hgroup hr main menu nav ol p pre section table ul' . split (
' '
)
)
] ,
[ 'rt' , new Set ( [ 'rt' , 'rp' ] ) ] ,
[ 'rp' , new Set ( [ 'rt' , 'rp' ] ) ] ,
[ 'optgroup' , new Set ( [ 'optgroup' ] ) ] ,
[ 'option' , new Set ( [ 'option' , 'optgroup' ] ) ] ,
[ 'thead' , new Set ( [ 'tbody' , 'tfoot' ] ) ] ,
[ 'tbody' , new Set ( [ 'tbody' , 'tfoot' ] ) ] ,
[ 'tfoot' , new Set ( [ 'tbody' ] ) ] ,
[ 'tr' , new Set ( [ 'tr' , 'tbody' ] ) ] ,
[ 'td' , new Set ( [ 'td' , 'th' , 'tr' ] ) ] ,
[ 'th' , new Set ( [ 'td' , 'th' , 'tr' ] ) ]
] ) ;
// can this be a child of the parent element, or does it implicitly
// close it, like `<li>one<li>two`?
@ -127,14 +131,10 @@ const disallowed_contents = new Map([
* @ param { string } next
* /
export function closing _tag _omitted ( current , next ) {
if ( disallowed _contents . has ( current ) ) {
if ( ! next || disallowed _contents . get ( current ) . has ( next ) ) {
return true ;
}
}
return false ;
if ( disallowed _contents . has ( current ) ) {
if ( ! next || disallowed _contents . get ( current ) . has ( next ) ) {
return true ;
}
}
return false ;
}