Source file src/html/template/js.go

     1  // Copyright 2011 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package template
     6  
     7  import (
     8  	"bytes"
     9  	"encoding/json"
    10  	"fmt"
    11  	"reflect"
    12  	"regexp"
    13  	"strings"
    14  	"unicode/utf8"
    15  )
    16  
    17  // jsWhitespace contains all of the JS whitespace characters, as defined
    18  // by the \s character class.
    19  // See https://developer.mozilla.org/en-US/docs/Web/JavaScript/Guide/Regular_expressions/Character_classes.
    20  const jsWhitespace = "\f\n\r\t\v\u0020\u00a0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000\ufeff"
    21  
    22  // nextJSCtx returns the context that determines whether a slash after the
    23  // given run of tokens starts a regular expression instead of a division
    24  // operator: / or /=.
    25  //
    26  // This assumes that the token run does not include any string tokens, comment
    27  // tokens, regular expression literal tokens, or division operators.
    28  //
    29  // This fails on some valid but nonsensical JavaScript programs like
    30  // "x = ++/foo/i" which is quite different than "x++/foo/i", but is not known to
    31  // fail on any known useful programs. It is based on the draft
    32  // JavaScript 2.0 lexical grammar and requires one token of lookbehind:
    33  // https://www.mozilla.org/js/language/js20-2000-07/rationale/syntax.html
    34  func nextJSCtx(s []byte, preceding jsCtx) jsCtx {
    35  	// Trim all JS whitespace characters
    36  	s = bytes.TrimRight(s, jsWhitespace)
    37  	if len(s) == 0 {
    38  		return preceding
    39  	}
    40  
    41  	// All cases below are in the single-byte UTF-8 group.
    42  	switch c, n := s[len(s)-1], len(s); c {
    43  	case '+', '-':
    44  		// ++ and -- are not regexp preceders, but + and - are whether
    45  		// they are used as infix or prefix operators.
    46  		start := n - 1
    47  		// Count the number of adjacent dashes or pluses.
    48  		for start > 0 && s[start-1] == c {
    49  			start--
    50  		}
    51  		if (n-start)&1 == 1 {
    52  			// Reached for trailing minus signs since "---" is the
    53  			// same as "-- -".
    54  			return jsCtxRegexp
    55  		}
    56  		return jsCtxDivOp
    57  	case '.':
    58  		// Handle "42."
    59  		if n != 1 && '0' <= s[n-2] && s[n-2] <= '9' {
    60  			return jsCtxDivOp
    61  		}
    62  		return jsCtxRegexp
    63  	// Suffixes for all punctuators from section 7.7 of the language spec
    64  	// that only end binary operators not handled above.
    65  	case ',', '<', '>', '=', '*', '%', '&', '|', '^', '?':
    66  		return jsCtxRegexp
    67  	// Suffixes for all punctuators from section 7.7 of the language spec
    68  	// that are prefix operators not handled above.
    69  	case '!', '~':
    70  		return jsCtxRegexp
    71  	// Matches all the punctuators from section 7.7 of the language spec
    72  	// that are open brackets not handled above.
    73  	case '(', '[':
    74  		return jsCtxRegexp
    75  	// Matches all the punctuators from section 7.7 of the language spec
    76  	// that precede expression starts.
    77  	case ':', ';', '{':
    78  		return jsCtxRegexp
    79  	// CAVEAT: the close punctuators ('}', ']', ')') precede div ops and
    80  	// are handled in the default except for '}' which can precede a
    81  	// division op as in
    82  	//    ({ valueOf: function () { return 42 } } / 2
    83  	// which is valid, but, in practice, developers don't divide object
    84  	// literals, so our heuristic works well for code like
    85  	//    function () { ... }  /foo/.test(x) && sideEffect();
    86  	// The ')' punctuator can precede a regular expression as in
    87  	//     if (b) /foo/.test(x) && ...
    88  	// but this is much less likely than
    89  	//     (a + b) / c
    90  	case '}':
    91  		return jsCtxRegexp
    92  	default:
    93  		// Look for an IdentifierName and see if it is a keyword that
    94  		// can precede a regular expression.
    95  		j := n
    96  		for j > 0 && isJSIdentPart(rune(s[j-1])) {
    97  			j--
    98  		}
    99  		// An IdentifierName after a property-access dot is a property name,
   100  		// which precedes a div op.
   101  		if regexpPrecederKeywords[string(s[j:])] &&
   102  			!bytes.HasSuffix(bytes.TrimRight(s[:j], jsWhitespace), []byte(".")) {
   103  			return jsCtxRegexp
   104  		}
   105  	}
   106  	// Otherwise is a punctuator not listed above, or
   107  	// a string which precedes a div op, or an identifier
   108  	// which precedes a div op.
   109  	return jsCtxDivOp
   110  }
   111  
   112  // regexpPrecederKeywords is a set of JS keywords that can precede a regular
   113  // expression in JS source. It deliberately treats the context-sensitive
   114  // keyword yield as a keyword.
   115  var regexpPrecederKeywords = map[string]bool{
   116  	"break":      true,
   117  	"case":       true,
   118  	"continue":   true,
   119  	"delete":     true,
   120  	"do":         true,
   121  	"else":       true,
   122  	"finally":    true,
   123  	"in":         true,
   124  	"instanceof": true,
   125  	"return":     true,
   126  	"throw":      true,
   127  	"try":        true,
   128  	"typeof":     true,
   129  	"void":       true,
   130  	"yield":      true,
   131  }
   132  
   133  var jsonMarshalType = reflect.TypeFor[json.Marshaler]()
   134  
   135  // indirectToJSONMarshaler returns the value, after dereferencing as many times
   136  // as necessary to reach the base type (or nil) or an implementation of json.Marshal.
   137  func indirectToJSONMarshaler(a any) any {
   138  	// text/template now supports passing untyped nil as a func call
   139  	// argument, so we must support it. Otherwise we'd panic below, as one
   140  	// cannot call the Type or Interface methods on an invalid
   141  	// reflect.Value. See golang.org/issue/18716.
   142  	if a == nil {
   143  		return nil
   144  	}
   145  
   146  	v := reflect.ValueOf(a)
   147  	for !v.Type().Implements(jsonMarshalType) && v.Kind() == reflect.Pointer && !v.IsNil() {
   148  		v = v.Elem()
   149  	}
   150  	return v.Interface()
   151  }
   152  
   153  var scriptTagRe = regexp.MustCompile("(?i)<(/?)script")
   154  
   155  // jsValEscaper escapes its inputs to a JS Expression (section 11.14) that has
   156  // neither side-effects nor free variables outside (NaN, Infinity).
   157  func jsValEscaper(args ...any) string {
   158  	var a any
   159  	if len(args) == 1 {
   160  		a = indirectToJSONMarshaler(args[0])
   161  		switch t := a.(type) {
   162  		case JS:
   163  			return string(t)
   164  		case JSStr:
   165  			// TODO: normalize quotes.
   166  			return `"` + string(t) + `"`
   167  		case json.Marshaler:
   168  			// Do not treat as a Stringer.
   169  		case fmt.Stringer:
   170  			a = t.String()
   171  		}
   172  	} else {
   173  		for i, arg := range args {
   174  			args[i] = indirectToJSONMarshaler(arg)
   175  		}
   176  		a = fmt.Sprint(args...)
   177  	}
   178  	// TODO: detect cycles before calling Marshal which loops infinitely on
   179  	// cyclic data. This may be an unacceptable DoS risk.
   180  	b, err := json.Marshal(a)
   181  	if err != nil {
   182  		// While the standard JSON marshaler does not include user controlled
   183  		// information in the error message, if a type has a MarshalJSON method,
   184  		// the content of the error message is not guaranteed. Since we insert
   185  		// the error into the template, as part of a comment, we attempt to
   186  		// prevent the error from either terminating the comment, or the script
   187  		// block itself.
   188  		//
   189  		// In particular we:
   190  		//   * replace "*/" comment end tokens with "* /", which does not
   191  		//     terminate the comment
   192  		//   * replace "<script" and "</script" with "\x3Cscript" and "\x3C/script"
   193  		//     (case insensitively), and "<!--" with "\x3C!--", which prevents
   194  		//     confusing script block termination semantics
   195  		//
   196  		// We also put a space before the comment so that if it is flush against
   197  		// a division operator it is not turned into a line comment:
   198  		//     x/{{y}}
   199  		// turning into
   200  		//     x//* error marshaling y:
   201  		//          second line of error message */null
   202  		errStr := err.Error()
   203  		errStr = string(scriptTagRe.ReplaceAll([]byte(errStr), []byte(`\x3C${1}script`)))
   204  		errStr = strings.ReplaceAll(errStr, "*/", "* /")
   205  		errStr = strings.ReplaceAll(errStr, "<!--", `\x3C!--`)
   206  		return fmt.Sprintf(" /* %s */null ", errStr)
   207  	}
   208  
   209  	// TODO: maybe post-process output to prevent it from containing
   210  	// "<!--", "-->", "<![CDATA[", "]]>", or "</script"
   211  	// in case custom marshalers produce output containing those.
   212  	// Note: Do not use \x escaping to save bytes because it is not JSON compatible and this escaper
   213  	// supports ld+json content-type.
   214  	if len(b) == 0 {
   215  		// In, `x=y/{{.}}*z` a json.Marshaler that produces "" should
   216  		// not cause the output `x=y/*z`.
   217  		return " null "
   218  	}
   219  	first, _ := utf8.DecodeRune(b)
   220  	last, _ := utf8.DecodeLastRune(b)
   221  	var buf strings.Builder
   222  	// Prevent IdentifierNames and NumericLiterals from running into
   223  	// keywords: in, instanceof, typeof, void
   224  	pad := isJSIdentPart(first) || isJSIdentPart(last)
   225  	if pad {
   226  		buf.WriteByte(' ')
   227  	}
   228  	written := 0
   229  	// Make sure that json.Marshal escapes codepoints U+2028 & U+2029
   230  	// so it falls within the subset of JSON which is valid JS.
   231  	for i := 0; i < len(b); {
   232  		rune, n := utf8.DecodeRune(b[i:])
   233  		repl := ""
   234  		if rune == 0x2028 {
   235  			repl = `\u2028`
   236  		} else if rune == 0x2029 {
   237  			repl = `\u2029`
   238  		}
   239  		if repl != "" {
   240  			buf.Write(b[written:i])
   241  			buf.WriteString(repl)
   242  			written = i + n
   243  		}
   244  		i += n
   245  	}
   246  	if buf.Len() != 0 {
   247  		buf.Write(b[written:])
   248  		if pad {
   249  			buf.WriteByte(' ')
   250  		}
   251  		return buf.String()
   252  	}
   253  	return string(b)
   254  }
   255  
   256  // jsStrEscaper produces a string that can be included between quotes in
   257  // JavaScript source, in JavaScript embedded in an HTML5 <script> element,
   258  // or in an HTML5 event handler attribute such as onclick.
   259  func jsStrEscaper(args ...any) string {
   260  	s, t := stringify(args...)
   261  	if t == contentTypeJSStr {
   262  		return replace(s, jsStrNormReplacementTable)
   263  	}
   264  	return replace(s, jsStrReplacementTable)
   265  }
   266  
   267  func jsTmplLitEscaper(args ...any) string {
   268  	s, _ := stringify(args...)
   269  	return replace(s, jsBqStrReplacementTable)
   270  }
   271  
   272  // jsRegexpEscaper behaves like jsStrEscaper but escapes regular expression
   273  // specials so the result is treated literally when included in a regular
   274  // expression literal. /foo{{.X}}bar/ matches the string "foo" followed by
   275  // the literal text of {{.X}} followed by the string "bar".
   276  func jsRegexpEscaper(args ...any) string {
   277  	s, _ := stringify(args...)
   278  	s = replace(s, jsRegexpReplacementTable)
   279  	if s == "" {
   280  		// /{{.X}}/ should not produce a line comment when .X == "".
   281  		return "(?:)"
   282  	}
   283  	return s
   284  }
   285  
   286  // replace replaces each rune r of s with replacementTable[r], provided that
   287  // r < len(replacementTable). If replacementTable[r] is the empty string then
   288  // no replacement is made.
   289  // It also replaces runes U+2028 and U+2029 with the raw strings `\u2028` and
   290  // `\u2029`.
   291  func replace(s string, replacementTable []string) string {
   292  	var b strings.Builder
   293  	r, w, written := rune(0), 0, 0
   294  	for i := 0; i < len(s); i += w {
   295  		// See comment in htmlEscaper.
   296  		r, w = utf8.DecodeRuneInString(s[i:])
   297  		var repl string
   298  		switch {
   299  		case int(r) < len(lowUnicodeReplacementTable):
   300  			repl = lowUnicodeReplacementTable[r]
   301  		case int(r) < len(replacementTable) && replacementTable[r] != "":
   302  			repl = replacementTable[r]
   303  		case r == '\u2028':
   304  			repl = `\u2028`
   305  		case r == '\u2029':
   306  			repl = `\u2029`
   307  		default:
   308  			continue
   309  		}
   310  		if written == 0 {
   311  			b.Grow(len(s))
   312  		}
   313  		b.WriteString(s[written:i])
   314  		b.WriteString(repl)
   315  		written = i + w
   316  	}
   317  	if written == 0 {
   318  		return s
   319  	}
   320  	b.WriteString(s[written:])
   321  	return b.String()
   322  }
   323  
   324  var lowUnicodeReplacementTable = []string{
   325  	0: `\u0000`, 1: `\u0001`, 2: `\u0002`, 3: `\u0003`, 4: `\u0004`, 5: `\u0005`, 6: `\u0006`,
   326  	'\a': `\u0007`,
   327  	'\b': `\u0008`,
   328  	'\t': `\t`,
   329  	'\n': `\n`,
   330  	'\v': `\u000b`, // "\v" == "v" on IE 6.
   331  	'\f': `\f`,
   332  	'\r': `\r`,
   333  	0xe:  `\u000e`, 0xf: `\u000f`, 0x10: `\u0010`, 0x11: `\u0011`, 0x12: `\u0012`, 0x13: `\u0013`,
   334  	0x14: `\u0014`, 0x15: `\u0015`, 0x16: `\u0016`, 0x17: `\u0017`, 0x18: `\u0018`, 0x19: `\u0019`,
   335  	0x1a: `\u001a`, 0x1b: `\u001b`, 0x1c: `\u001c`, 0x1d: `\u001d`, 0x1e: `\u001e`, 0x1f: `\u001f`,
   336  }
   337  
   338  var jsStrReplacementTable = []string{
   339  	0:    `\u0000`,
   340  	'\t': `\t`,
   341  	'\n': `\n`,
   342  	'\v': `\u000b`, // "\v" == "v" on IE 6.
   343  	'\f': `\f`,
   344  	'\r': `\r`,
   345  	// Encode HTML specials as hex so the output can be embedded
   346  	// in HTML attributes without further encoding.
   347  	'"':  `\u0022`,
   348  	'`':  `\u0060`,
   349  	'&':  `\u0026`,
   350  	'\'': `\u0027`,
   351  	'+':  `\u002b`,
   352  	'/':  `\/`,
   353  	'<':  `\u003c`,
   354  	'>':  `\u003e`,
   355  	'\\': `\\`,
   356  }
   357  
   358  // jsBqStrReplacementTable is like jsStrReplacementTable except it also contains
   359  // the special characters for JS template literals: $, {, and }.
   360  var jsBqStrReplacementTable = []string{
   361  	0:    `\u0000`,
   362  	'\t': `\t`,
   363  	'\n': `\n`,
   364  	'\v': `\u000b`, // "\v" == "v" on IE 6.
   365  	'\f': `\f`,
   366  	'\r': `\r`,
   367  	// Encode HTML specials as hex so the output can be embedded
   368  	// in HTML attributes without further encoding.
   369  	'"':  `\u0022`,
   370  	'`':  `\u0060`,
   371  	'&':  `\u0026`,
   372  	'\'': `\u0027`,
   373  	'+':  `\u002b`,
   374  	'/':  `\/`,
   375  	'<':  `\u003c`,
   376  	'>':  `\u003e`,
   377  	'\\': `\\`,
   378  	'$':  `\u0024`,
   379  	'{':  `\u007b`,
   380  	'}':  `\u007d`,
   381  }
   382  
   383  // jsStrNormReplacementTable is like jsStrReplacementTable but does not
   384  // overencode existing escapes since this table has no entry for `\`.
   385  var jsStrNormReplacementTable = []string{
   386  	0:    `\u0000`,
   387  	'\t': `\t`,
   388  	'\n': `\n`,
   389  	'\v': `\u000b`, // "\v" == "v" on IE 6.
   390  	'\f': `\f`,
   391  	'\r': `\r`,
   392  	// Encode HTML specials as hex so the output can be embedded
   393  	// in HTML attributes without further encoding.
   394  	'"':  `\u0022`,
   395  	'&':  `\u0026`,
   396  	'\'': `\u0027`,
   397  	'`':  `\u0060`,
   398  	'+':  `\u002b`,
   399  	'/':  `\/`,
   400  	'<':  `\u003c`,
   401  	'>':  `\u003e`,
   402  }
   403  var jsRegexpReplacementTable = []string{
   404  	0:    `\u0000`,
   405  	'\t': `\t`,
   406  	'\n': `\n`,
   407  	'\v': `\u000b`, // "\v" == "v" on IE 6.
   408  	'\f': `\f`,
   409  	'\r': `\r`,
   410  	// Encode HTML specials as hex so the output can be embedded
   411  	// in HTML attributes without further encoding.
   412  	'"':  `\u0022`,
   413  	'$':  `\$`,
   414  	'&':  `\u0026`,
   415  	'\'': `\u0027`,
   416  	'(':  `\(`,
   417  	')':  `\)`,
   418  	'*':  `\*`,
   419  	'+':  `\u002b`,
   420  	'-':  `\-`,
   421  	'.':  `\.`,
   422  	'/':  `\/`,
   423  	'<':  `\u003c`,
   424  	'>':  `\u003e`,
   425  	'?':  `\?`,
   426  	'[':  `\[`,
   427  	'\\': `\\`,
   428  	']':  `\]`,
   429  	'^':  `\^`,
   430  	'{':  `\{`,
   431  	'|':  `\|`,
   432  	'}':  `\}`,
   433  }
   434  
   435  // isJSIdentPart reports whether the given rune is a JS identifier part.
   436  // It does not handle all the non-Latin letters, joiners, and combining marks,
   437  // but it does handle every codepoint that can occur in a numeric literal or
   438  // a keyword.
   439  func isJSIdentPart(r rune) bool {
   440  	switch {
   441  	case r == '$':
   442  		return true
   443  	case '0' <= r && r <= '9':
   444  		return true
   445  	case 'A' <= r && r <= 'Z':
   446  		return true
   447  	case r == '_':
   448  		return true
   449  	case 'a' <= r && r <= 'z':
   450  		return true
   451  	}
   452  	return false
   453  }
   454  
   455  // isJSType reports whether the given MIME type should be considered JavaScript.
   456  //
   457  // It is used to determine whether a script tag with a type attribute is a javascript container.
   458  func isJSType(mimeType string) bool {
   459  	// per
   460  	//   https://www.w3.org/TR/html5/scripting-1.html#attr-script-type
   461  	//   https://tools.ietf.org/html/rfc7231#section-3.1.1
   462  	//   https://tools.ietf.org/html/rfc4329#section-3
   463  	//   https://www.ietf.org/rfc/rfc4627.txt
   464  	// discard parameters
   465  	mimeType, _, _ = strings.Cut(mimeType, ";")
   466  	mimeType = strings.ToLower(mimeType)
   467  	mimeType = strings.TrimSpace(mimeType)
   468  	switch mimeType {
   469  	case
   470  		"",
   471  		"application/ecmascript",
   472  		"application/javascript",
   473  		"application/json",
   474  		"application/ld+json",
   475  		"application/x-ecmascript",
   476  		"application/x-javascript",
   477  		"module",
   478  		"text/ecmascript",
   479  		"text/javascript",
   480  		"text/javascript1.0",
   481  		"text/javascript1.1",
   482  		"text/javascript1.2",
   483  		"text/javascript1.3",
   484  		"text/javascript1.4",
   485  		"text/javascript1.5",
   486  		"text/jscript",
   487  		"text/livescript",
   488  		"text/x-ecmascript",
   489  		"text/x-javascript":
   490  		return true
   491  	default:
   492  		return false
   493  	}
   494  }
   495  

View as plain text