Skip to content

Commit 1f54598

Browse files
committed
feat(compiler): tokenize interpolations in escapable raw text
Expose interpolation boundaries in textarea and title so consumers can replace regular expressions with lexer tokens. Preserve RCDATA entity validation and closing-tag handling, and add public parser regressions.
1 parent 4762d60 commit 1f54598

4 files changed

Lines changed: 181 additions & 17 deletions

File tree

‎packages/angular-html-parser/readme.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -97,6 +97,7 @@ interface Options {
9797
- support [bogus comments](https://www.w3.org/TR/html5/syntax.html#bogus-comment-state) (`<!...>`, `<?...>`)
9898
- ~~support full [named entities](https://html.spec.whatwg.org/multipage/entities.json)~~ (fixed upstream)
9999
- add `type` property to nodes
100+
- tokenize interpolation expressions inside escapable raw text (`textarea` and HTML `title`)
100101
- value span for attributes includes quotes
101102

102103
## Development

‎packages/angular-html-parser/test/index_spec.ts‎

Lines changed: 124 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -298,3 +298,127 @@ describe("public token API", () => {
298298
expect(token.parts).toEqual(["{{", ' "}}" ', "}}"]);
299299
});
300300
});
301+
302+
describe("escapable raw text interpolation", () => {
303+
describe.each(["textarea", "title"])("<%s>", (tagName) => {
304+
it.each([
305+
{
306+
name: "multiple expressions",
307+
content: "before {{value}} after {{other}}",
308+
value: "before {{value}} after {{other}}",
309+
expressions: ["value", "other"],
310+
},
311+
{
312+
name: "double braces inside a string",
313+
content: 'before {{ "{{first}} / {{second}}" }} after',
314+
value: 'before {{ "{{first}} / {{second}}" }} after',
315+
expressions: [' "{{first}} / {{second}}" '],
316+
},
317+
{
318+
name: "literal markup inside and outside an expression",
319+
content: '<b>{{ "<span>" }}</b>',
320+
value: '<b>{{ "<span>" }}</b>',
321+
expressions: [' "<span>" '],
322+
},
323+
{
324+
name: "entities inside and outside an expression",
325+
content: '&lt;{{ "&amp;" }}&#32;{{value}}&gt;',
326+
value: '<{{ "&" }} {{value}}>',
327+
expressions: [' "&amp;" ', "value"],
328+
},
329+
{
330+
name: "a leading LF",
331+
content: "\n{{value}}\n",
332+
value: "\n{{value}}\n",
333+
expressions: ["value"],
334+
},
335+
{
336+
name: "CRLF inside an expression",
337+
content: "{{a\r\n+b}}tail",
338+
value: "{{a\n+b}}tail",
339+
expressions: ["a\n+b"],
340+
},
341+
{
342+
name: "adjacent and empty expressions",
343+
content: "{{one}}{{two}}{{}}",
344+
value: "{{one}}{{two}}{{}}",
345+
expressions: ["one", "two", ""],
346+
},
347+
])("should tokenize $name", ({ content, value, expressions }) => {
348+
const result = parse(`<${tagName}>${content}</${tagName}>`);
349+
expect(result.errors).toEqual([]);
350+
const element = result.rootNodes[0] as ast.Element;
351+
expect(element.children).toHaveLength(1);
352+
const text = element.children[0] as ast.Text;
353+
expect(text.value).toBe(
354+
tagName === "textarea" && value.startsWith("\n")
355+
? value.slice(1)
356+
: value,
357+
);
358+
expect(
359+
text.tokens
360+
.filter((token) => token.type === TokenType.INTERPOLATION)
361+
.map((token) => token.parts),
362+
).toEqual(expressions.map((expression) => ["{{", expression, "}}"]));
363+
expect(
364+
text.tokens.map((token) => token.sourceSpan.toString()).join(""),
365+
).toBe(content);
366+
});
367+
});
368+
369+
it("should stop an unfinished interpolation at the closing tag", () => {
370+
const result = parse("<textarea>{{value</textarea><div>after</div>");
371+
expect(result.errors).toEqual([]);
372+
const element = result.rootNodes[0] as ast.Element;
373+
const text = element.children[0] as ast.Text;
374+
expect(text.value).toBe("{{value");
375+
expect(
376+
text.tokens.find((token) => token.type === TokenType.INTERPOLATION)
377+
?.parts,
378+
).toEqual(["{{", "value"]);
379+
expect(element.endSourceSpan?.toString()).toBe("</textarea>");
380+
expect((result.rootNodes[1] as ast.Element).name).toBe("div");
381+
});
382+
383+
it.each(["", "\\"])(
384+
"should keep the closing tag significant after %j in a quoted expression",
385+
(escape) => {
386+
const result = parse(
387+
`<textarea>{{ "${escape}</textarea>" }}<div>after</div>`,
388+
);
389+
expect(result.errors).toEqual([]);
390+
const element = result.rootNodes[0] as ast.Element;
391+
expect((element.children[0] as ast.Text).value).toBe(`{{ "${escape}`);
392+
expect(element.endSourceSpan?.toString()).toBe("</textarea>");
393+
expect((result.rootNodes[2] as ast.Element).name).toBe("div");
394+
},
395+
);
396+
397+
describe.each(["textarea", "title"])(
398+
"invalid entities in <%s>",
399+
(tagName) => {
400+
it.each([
401+
{ entity: "&bogus;", error: "Unknown entity" },
402+
{ entity: "&#x110000;", error: "Unknown entity" },
403+
{ entity: "&#x;", error: "Unknown entity" },
404+
{ entity: "&#x41", error: "Unable to parse entity" },
405+
{ entity: "\\&bogus;", error: "Unknown entity" },
406+
])("should report $entity as a parse error", ({ entity, error }) => {
407+
const result = parse(`<${tagName}>{{ "${entity}" }}</${tagName}>`);
408+
expect(result.errors[0]?.msg).toContain(error);
409+
});
410+
},
411+
);
412+
413+
it.each(["script", "style"])("should keep <%s> as raw text", (tagName) => {
414+
const content = '{{ "<b>&amp;" }}';
415+
const result = parse(`<${tagName}>${content}</${tagName}>`);
416+
expect(result.errors).toEqual([]);
417+
const element = result.rootNodes[0] as ast.Element;
418+
const text = element.children[0] as ast.Text;
419+
expect(text.value).toBe(content);
420+
expect(text.tokens.map((token) => token.type)).toEqual([
421+
TokenType.RAW_TEXT,
422+
]);
423+
});
424+
});

‎packages/compiler/src/ml_parser/lexer.ts‎

Lines changed: 55 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -738,6 +738,16 @@ class _Tokenizer {
738738

739739
private _consumeEntity(textTokenType: TokenType): void {
740740
this._beginToken(TokenType.ENCODED_ENTITY);
741+
const start = this._cursor.clone();
742+
const parts = this._readEntity();
743+
if (parts.length === 1) {
744+
this._beginToken(textTokenType, start);
745+
}
746+
this._endToken(parts);
747+
}
748+
749+
/** Consume an HTML entity without emitting a token. */
750+
private _readEntity(): [string] | [string, string] {
741751
const start = this._cursor.clone();
742752
this._cursor.advance();
743753
if (this._attemptCharCode(chars.$HASH)) {
@@ -758,7 +768,7 @@ class _Tokenizer {
758768
this._cursor.advance();
759769
try {
760770
const charCode = parseInt(strNum, isHex ? 16 : 10);
761-
this._endToken([String.fromCodePoint(charCode), this._cursor.getChars(start)]);
771+
return [String.fromCodePoint(charCode), this._cursor.getChars(start)];
762772
} catch {
763773
throw this._createError(
764774
_unknownEntityErrorMsg(this._cursor.getChars(start)),
@@ -769,25 +779,39 @@ class _Tokenizer {
769779
const nameStart = this._cursor.clone();
770780
this._attemptCharCodeUntilFn(isNamedEntityEnd);
771781
if (this._cursor.peek() != chars.$SEMICOLON) {
772-
// No semicolon was found so abort the encoded entity token that was in progress, and treat
773-
// this as a text token
774-
this._beginToken(textTokenType, start);
782+
// Without a semicolon, only the ampersand is consumed as plain text.
775783
this._cursor = nameStart;
776-
this._endToken(['&']);
784+
return ['&'];
777785
} else {
778786
const name = this._cursor.getChars(nameStart);
779787
this._cursor.advance();
780788
const char = Object.hasOwn(NAMED_ENTITIES, name) && NAMED_ENTITIES[name];
781789
if (!char) {
782790
throw this._createError(_unknownEntityErrorMsg(name), this._cursor.getSpan(start));
783791
}
784-
this._endToken([char, `&${name};`]);
792+
return [char, `&${name};`];
785793
}
786794
}
787795
}
788796

789797
private _consumeRawText(consumeEntities: boolean, endMarkerPredicate: () => boolean): void {
790-
this._beginToken(consumeEntities ? TokenType.ESCAPABLE_RAW_TEXT : TokenType.RAW_TEXT);
798+
if (consumeEntities) {
799+
const isRawTextEnd = () => {
800+
const start = this._cursor.clone();
801+
const foundEndMarker = endMarkerPredicate();
802+
this._cursor = start;
803+
return foundEndMarker;
804+
};
805+
this._consumeWithInterpolation(
806+
TokenType.ESCAPABLE_RAW_TEXT,
807+
TokenType.INTERPOLATION,
808+
isRawTextEnd,
809+
isRawTextEnd,
810+
);
811+
return;
812+
}
813+
814+
this._beginToken(TokenType.RAW_TEXT);
791815
const parts: string[] = [];
792816
while (true) {
793817
const tagCloseStart = this._cursor.clone();
@@ -796,14 +820,7 @@ class _Tokenizer {
796820
if (foundEndMarker) {
797821
break;
798822
}
799-
if (consumeEntities && this._cursor.peek() === chars.$AMPERSAND) {
800-
this._endToken([this._processCarriageReturns(parts.join(''))]);
801-
parts.length = 0;
802-
this._consumeEntity(TokenType.ESCAPABLE_RAW_TEXT);
803-
this._beginToken(TokenType.ESCAPABLE_RAW_TEXT);
804-
} else {
805-
parts.push(this._readChar());
806-
}
823+
parts.push(this._readChar());
807824
}
808825
this._endToken([this._processCarriageReturns(parts.join(''))]);
809826
}
@@ -1324,7 +1341,12 @@ class _Tokenizer {
13241341
if (this._attemptStr(INTERPOLATION.start)) {
13251342
this._endToken([this._processCarriageReturns(parts.join(''))], current);
13261343
parts.length = 0;
1327-
this._consumeInterpolation(interpolationTokenType, current, endInterpolation);
1344+
this._consumeInterpolation(
1345+
interpolationTokenType,
1346+
current,
1347+
endInterpolation,
1348+
textTokenType,
1349+
);
13281350
this._beginToken(textTokenType);
13291351
} else if (this._cursor.peek() === chars.$AMPERSAND) {
13301352
this._endToken([this._processCarriageReturns(parts.join(''))]);
@@ -1352,11 +1374,13 @@ class _Tokenizer {
13521374
* @param interpolationStart a cursor that points to the start of this interpolation.
13531375
* @param prematureEndPredicate a function that should return true if the next characters indicate
13541376
* an end to the interpolation before its normal closing marker.
1377+
* @param textTokenType the surrounding text type, which determines whether markup is literal.
13551378
*/
13561379
private _consumeInterpolation(
13571380
interpolationTokenType: TokenType,
13581381
interpolationStart: CharacterCursor,
13591382
prematureEndPredicate: (() => boolean) | null,
1383+
textTokenType: TokenType,
13601384
): void {
13611385
const parts: string[] = [];
13621386
this._beginToken(interpolationTokenType, interpolationStart);
@@ -1372,7 +1396,7 @@ class _Tokenizer {
13721396
) {
13731397
const current = this._cursor.clone();
13741398

1375-
if (this._isTagStart()) {
1399+
if (textTokenType !== TokenType.ESCAPABLE_RAW_TEXT && this._isTagStart()) {
13761400
// We are starting what looks like an HTML element in the middle of this interpolation.
13771401
// Reset the cursor to before the `<` character and end the interpolation token.
13781402
// (This is actually wrong but here for backward compatibility).
@@ -1396,8 +1420,22 @@ class _Tokenizer {
13961420
}
13971421

13981422
const char = this._cursor.peek();
1423+
if (textTokenType === TokenType.ESCAPABLE_RAW_TEXT && char === chars.$AMPERSAND) {
1424+
this._readEntity();
1425+
continue;
1426+
}
13991427
this._cursor.advance();
14001428
if (char === chars.$BACKSLASH) {
1429+
if (textTokenType === TokenType.ESCAPABLE_RAW_TEXT) {
1430+
// String escapes cannot hide a closing tag or bypass HTML entity validation in RCDATA.
1431+
if (prematureEndPredicate?.()) {
1432+
continue;
1433+
}
1434+
if (this._cursor.peek() === chars.$AMPERSAND) {
1435+
this._readEntity();
1436+
continue;
1437+
}
1438+
}
14011439
// Skip the next character because it was escaped.
14021440
this._cursor.advance();
14031441
} else if (char === inQuote) {

‎packages/compiler/src/ml_parser/parser.ts‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -437,6 +437,7 @@ class _TreeBuilder {
437437
while (
438438
this._peek.type === TokenType.INTERPOLATION ||
439439
this._peek.type === TokenType.TEXT ||
440+
this._peek.type === TokenType.ESCAPABLE_RAW_TEXT ||
440441
this._peek.type === TokenType.ENCODED_ENTITY
441442
) {
442443
token = this._advance();

0 commit comments

Comments
 (0)