Source file src/encoding/json/jsontext/value.go

     1  // Copyright 2020 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  //go:build goexperiment.jsonv2
     6  
     7  package jsontext
     8  
     9  import (
    10  	"bytes"
    11  	"errors"
    12  	"io"
    13  	"slices"
    14  	"sync"
    15  
    16  	"encoding/json/internal/jsonflags"
    17  	"encoding/json/internal/jsonwire"
    18  )
    19  
    20  // AppendFloat appends src to dst as a JSON number per RFC 8259, section 6.
    21  //
    22  // Except for -0, which is formatted as -0 instead of 0,
    23  // the output is identical to ECMA-262, 10th edition, section 7.1.12.1
    24  // and (for 64-bit precision) identical to RFC 8785, section 3.2.2.3.
    25  // The values NaN, +Inf, and -Inf will be represented as a JSON string
    26  // with the values "NaN", "Infinity", and "-Infinity".
    27  //
    28  // Note that most JSON libraries and standards assume that JSON numbers
    29  // are 64-bit floating-point numbers. As such, prefer using 64 bits
    30  // of precision unless the recipient can know from other context
    31  // that the encoded number uses 32 bits of precision.
    32  func AppendFloat(dst []byte, src float64, bits int) []byte {
    33  	if bits != 32 && bits != 64 {
    34  		panic("illegal AppendFloat bit size")
    35  	}
    36  	return jsonwire.AppendFloat(dst, src, bits)
    37  }
    38  
    39  // AppendFormat formats the JSON value in src and appends it to dst
    40  // according to the specified options.
    41  // See [Value.Format] for more details about the formatting behavior.
    42  //
    43  // The dst and src may overlap.
    44  // If an error is reported, then the entirety of src is appended to dst.
    45  func AppendFormat[Bytes ~[]byte | ~string](dst []byte, src Bytes, opts ...Options) ([]byte, error) {
    46  	e := getBufferedEncoder(opts...)
    47  	defer putBufferedEncoder(e)
    48  	e.s.Flags.Set(jsonflags.OmitTopLevelNewline | 1)
    49  	if err := e.s.WriteValue(Value(src)); err != nil {
    50  		return append(dst, src...), err
    51  	}
    52  	return append(dst, e.s.Buf...), nil
    53  }
    54  
    55  // NOTE: Value is analogous to v1 json.RawMessage.
    56  
    57  // Value represents a single raw JSON value, which may be one of the following:
    58  //   - a JSON literal (i.e., null, true, or false)
    59  //   - a JSON string (e.g., "hello, world!")
    60  //   - a JSON number (e.g., 123.456)
    61  //   - an entire JSON object (e.g., {"fizz":"buzz"} )
    62  //   - an entire JSON array (e.g., [1,2,3] )
    63  //
    64  // Value can represent entire array or object values, while [Token] cannot.
    65  // Value may contain leading and/or trailing whitespace.
    66  type Value []byte
    67  
    68  // Clone returns a copy of v.
    69  func (v Value) Clone() Value {
    70  	return bytes.Clone(v)
    71  }
    72  
    73  // String returns the string formatting of v.
    74  func (v Value) String() string {
    75  	if v == nil {
    76  		return "null"
    77  	}
    78  	return string(v)
    79  }
    80  
    81  // IsValid reports whether the raw JSON value is syntactically valid
    82  // according to the specified options.
    83  //
    84  // By default (if no options are specified), it validates according to RFC 7493.
    85  // It verifies whether the input is properly encoded as UTF-8,
    86  // that escape sequences within strings decode to valid Unicode codepoints, and
    87  // that all names in each object are unique.
    88  // It does not verify whether numbers are representable within the limits
    89  // of any common numeric type (e.g., float64, int64, or uint64).
    90  //
    91  // Relevant options include:
    92  //   - [AllowDuplicateNames]
    93  //   - [AllowInvalidUTF8]
    94  //
    95  // All other options are ignored.
    96  func (v Value) IsValid(opts ...Options) bool {
    97  	// TODO: Document support for [WithByteLimit] and [WithDepthLimit].
    98  	d := getBufferedDecoder(v, opts...)
    99  	defer putBufferedDecoder(d)
   100  	_, errVal := d.ReadValue()
   101  	_, errEOF := d.ReadToken()
   102  	return errVal == nil && errEOF == io.EOF
   103  }
   104  
   105  // Format formats the raw JSON value in place.
   106  //
   107  // By default (if no options are specified), it validates according to RFC 7493
   108  // and produces the minimal JSON representation, where
   109  // all whitespace is elided and JSON strings use the shortest encoding.
   110  //
   111  // Relevant options include:
   112  //   - [AllowDuplicateNames]
   113  //   - [AllowInvalidUTF8]
   114  //   - [EscapeForHTML]
   115  //   - [EscapeForJS]
   116  //   - [PreserveRawStrings]
   117  //   - [CanonicalizeRawInts]
   118  //   - [CanonicalizeRawFloats]
   119  //   - [ReorderRawObjects]
   120  //   - [SpaceAfterColon]
   121  //   - [SpaceAfterComma]
   122  //   - [Multiline]
   123  //   - [WithIndent]
   124  //   - [WithIndentPrefix]
   125  //
   126  // All other options are ignored.
   127  //
   128  // It is guaranteed to succeed if the value is valid according to the same options.
   129  // If the value is already formatted, then the buffer is not mutated.
   130  func (v *Value) Format(opts ...Options) error {
   131  	// TODO: Document support for [WithByteLimit] and [WithDepthLimit].
   132  	return v.format(opts, nil)
   133  }
   134  
   135  // format accepts two []Options to avoid the allocation of appending them together.
   136  // It is equivalent to v.Format(append(opts1, opts2...)...).
   137  func (v *Value) format(opts1, opts2 []Options) error {
   138  	e := getBufferedEncoder(opts1...)
   139  	defer putBufferedEncoder(e)
   140  	e.s.Join(opts2...)
   141  	e.s.Flags.Set(jsonflags.OmitTopLevelNewline | 1)
   142  	if err := e.s.WriteValue(*v); err != nil {
   143  		return err
   144  	}
   145  	if !bytes.Equal(*v, e.s.Buf) {
   146  		*v = append((*v)[:0], e.s.Buf...)
   147  	}
   148  	return nil
   149  }
   150  
   151  // Compact removes all whitespace from the raw JSON value.
   152  //
   153  // It does not reformat JSON strings or numbers to use any other representation.
   154  // To maximize the set of JSON values that can be formatted,
   155  // it permits values with duplicate names and invalid UTF-8.
   156  //
   157  // Compact is equivalent to calling [Value.Format] with the following options:
   158  //   - [AllowDuplicateNames](true)
   159  //   - [AllowInvalidUTF8](true)
   160  //   - [PreserveRawStrings](true)
   161  //
   162  // Any options specified by the caller are applied after the initial set
   163  // and may deliberately override prior options.
   164  func (v *Value) Compact(opts ...Options) error {
   165  	return v.format([]Options{
   166  		AllowDuplicateNames(true),
   167  		AllowInvalidUTF8(true),
   168  		PreserveRawStrings(true),
   169  	}, opts)
   170  }
   171  
   172  // Indent reformats the whitespace in the raw JSON value so that each element
   173  // in a JSON object or array begins on an indented line according to the nesting.
   174  //
   175  // It does not reformat JSON strings or numbers to use any other representation.
   176  // To maximize the set of JSON values that can be formatted,
   177  // it permits values with duplicate names and invalid UTF-8.
   178  //
   179  // Indent is equivalent to calling [Value.Format] with the following options:
   180  //   - [AllowDuplicateNames](true)
   181  //   - [AllowInvalidUTF8](true)
   182  //   - [PreserveRawStrings](true)
   183  //   - [Multiline](true)
   184  //
   185  // Any options specified by the caller are applied after the initial set
   186  // and may deliberately override prior options.
   187  func (v *Value) Indent(opts ...Options) error {
   188  	return v.format([]Options{
   189  		AllowDuplicateNames(true),
   190  		AllowInvalidUTF8(true),
   191  		PreserveRawStrings(true),
   192  		Multiline(true),
   193  	}, opts)
   194  }
   195  
   196  // Canonicalize canonicalizes the raw JSON value according to the
   197  // JSON Canonicalization Scheme (JCS) as defined by RFC 8785.
   198  // Canonicalization produces a JSON value with the same meaning as the original,
   199  // but is stable in the sense that calling Canonicalize on a canonicalized
   200  // value does nothing.
   201  //
   202  // JSON strings are formatted to use their minimal representation,
   203  // JSON numbers are formatted as double precision numbers according
   204  // to some stable serialization algorithm.
   205  // JSON object members are sorted in ascending order by name.
   206  // All whitespace is removed.
   207  //
   208  // Canonicalize is equivalent to calling [Value.Format] with the following options:
   209  //   - [CanonicalizeRawInts](true)
   210  //   - [CanonicalizeRawFloats](true)
   211  //   - [ReorderRawObjects](true)
   212  //
   213  // Any options specified by the caller are applied after the initial set
   214  // and may deliberately override prior options.
   215  //
   216  // Note that JCS treats all JSON numbers as IEEE 754 double precision numbers.
   217  // Any numbers with precision beyond what is representable by that form
   218  // will lose their precision when canonicalized. For example, integer values
   219  // beyond ±2⁵³ will lose their precision. To preserve the original representation
   220  // of JSON integers, additionally set [CanonicalizeRawInts] to false:
   221  //
   222  //	v.Canonicalize(jsontext.CanonicalizeRawInts(false))
   223  func (v *Value) Canonicalize(opts ...Options) error {
   224  	return v.format([]Options{
   225  		CanonicalizeRawInts(true),
   226  		CanonicalizeRawFloats(true),
   227  		ReorderRawObjects(true),
   228  	}, opts)
   229  }
   230  
   231  // MarshalJSON returns v as the JSON encoding of v.
   232  // It performs no validation.
   233  // If v is nil, then this returns a JSON null.
   234  func (v Value) MarshalJSON() ([]byte, error) {
   235  	// NOTE: This matches the behavior of v1 json.RawMessage.MarshalJSON.
   236  	if v == nil {
   237  		return []byte("null"), nil
   238  	}
   239  	return v, nil
   240  }
   241  
   242  // UnmarshalJSON sets v as the JSON encoding of b.
   243  // It stores a copy of the provided raw JSON input without any validation.
   244  func (v *Value) UnmarshalJSON(b []byte) error {
   245  	// NOTE: This matches the behavior of v1 json.RawMessage.UnmarshalJSON.
   246  	if v == nil {
   247  		return errors.New("jsontext.Value: UnmarshalJSON on nil pointer")
   248  	}
   249  	*v = append((*v)[:0], b...)
   250  	return nil
   251  }
   252  
   253  // Kind returns the starting token kind.
   254  // For a valid value, this will never include [KindEndObject] or [KindEndArray].
   255  func (v Value) Kind() Kind {
   256  	if v := v[jsonwire.ConsumeWhitespace(v):]; len(v) > 0 {
   257  		return Kind(v[0]).normalize()
   258  	}
   259  	return invalidKind
   260  }
   261  
   262  const commaAndWhitespace = ", \n\r\t"
   263  
   264  type objectMember struct {
   265  	// name is the unquoted name.
   266  	name []byte // e.g., "name"
   267  	// buffer is the entirety of the raw JSON object member
   268  	// starting from right after the previous member (or opening '{')
   269  	// until right after the member value.
   270  	buffer []byte // e.g., `, \n\r\t"name": "value"`
   271  }
   272  
   273  func (x objectMember) Compare(y objectMember) int {
   274  	if c := jsonwire.CompareUTF16(x.name, y.name); c != 0 {
   275  		return c
   276  	}
   277  	// With [AllowDuplicateNames] or [AllowInvalidUTF8],
   278  	// names could be identical, so also sort using the member value.
   279  	return jsonwire.CompareUTF16(
   280  		bytes.TrimLeft(x.buffer, commaAndWhitespace),
   281  		bytes.TrimLeft(y.buffer, commaAndWhitespace))
   282  }
   283  
   284  var objectMemberPool = sync.Pool{New: func() any { return new([]objectMember) }}
   285  
   286  func getObjectMembers() *[]objectMember {
   287  	ns := objectMemberPool.Get().(*[]objectMember)
   288  	*ns = (*ns)[:0]
   289  	return ns
   290  }
   291  func putObjectMembers(ns *[]objectMember) {
   292  	if cap(*ns) < 1<<10 {
   293  		clear(*ns) // avoid pinning name and buffer
   294  		objectMemberPool.Put(ns)
   295  	}
   296  }
   297  
   298  // mustReorderObjects reorders in-place all object members in a JSON value,
   299  // which must be valid otherwise it panics.
   300  func mustReorderObjects(b []byte) {
   301  	// Obtain a buffered encoder just to use its internal buffer as
   302  	// a scratch buffer for reordering object members.
   303  	e2 := getBufferedEncoder()
   304  	defer putBufferedEncoder(e2)
   305  
   306  	// Disable unnecessary checks to syntactically parse the JSON value.
   307  	d := getBufferedDecoder(b)
   308  	defer putBufferedDecoder(d)
   309  	d.s.Flags.Set(jsonflags.AllowDuplicateNames | jsonflags.AllowInvalidUTF8 | 1)
   310  	mustReorderObjectsFromDecoder(d, &e2.s.Buf) // per RFC 8785, section 3.2.3
   311  }
   312  
   313  // mustReorderObjectsFromDecoder recursively reorders all object members in place
   314  // according to the ordering specified in RFC 8785, section 3.2.3.
   315  //
   316  // Pre-conditions:
   317  //   - The value is valid (i.e., no decoder errors should ever occur).
   318  //   - Initial call is provided a Decoder reading from the start of v.
   319  //
   320  // Post-conditions:
   321  //   - Exactly one JSON value is read from the Decoder.
   322  //   - All fully-parsed JSON objects are reordered by directly moving
   323  //     the members in the value buffer.
   324  //
   325  // The runtime is approximately O(n·log(n)) + O(m·log(m)),
   326  // where n is len(v) and m is the total number of object members.
   327  func mustReorderObjectsFromDecoder(d *Decoder, scratch *[]byte) {
   328  	switch tok, err := d.ReadToken(); tok.Kind() {
   329  	case '{':
   330  		// Iterate and collect the name and offsets for every object member.
   331  		members := getObjectMembers()
   332  		defer putObjectMembers(members)
   333  		var prevMember objectMember
   334  		isSorted := true
   335  
   336  		beforeBody := d.InputOffset() // offset after '{'
   337  		for d.PeekKind() != '}' {
   338  			beforeName := d.InputOffset()
   339  			var flags jsonwire.ValueFlags
   340  			name, _ := d.s.ReadValue(&flags)
   341  			name = jsonwire.UnquoteMayCopy(name, flags.IsVerbatim())
   342  			mustReorderObjectsFromDecoder(d, scratch)
   343  			afterValue := d.InputOffset()
   344  
   345  			currMember := objectMember{name, d.s.buf[beforeName:afterValue]}
   346  			if isSorted && len(*members) > 0 {
   347  				isSorted = objectMember.Compare(prevMember, currMember) < 0
   348  			}
   349  			*members = append(*members, currMember)
   350  			prevMember = currMember
   351  		}
   352  		afterBody := d.InputOffset() // offset before '}'
   353  		d.ReadToken()
   354  
   355  		// Sort the members; return early if it's already sorted.
   356  		if isSorted {
   357  			return
   358  		}
   359  		firstBufferBeforeSorting := (*members)[0].buffer
   360  		slices.SortFunc(*members, objectMember.Compare)
   361  		firstBufferAfterSorting := (*members)[0].buffer
   362  
   363  		// Append the reordered members to a new buffer,
   364  		// then copy the reordered members back over the original members.
   365  		// Avoid swapping in place since each member may be a different size
   366  		// where moving a member over a smaller member may corrupt the data
   367  		// for subsequent members before they have been moved.
   368  		//
   369  		// The following invariant must hold:
   370  		//	sum([m.after-m.before for m in members]) == afterBody-beforeBody
   371  		commaAndWhitespacePrefix := func(b []byte) []byte {
   372  			return b[:len(b)-len(bytes.TrimLeft(b, commaAndWhitespace))]
   373  		}
   374  		sorted := (*scratch)[:0]
   375  		for i, member := range *members {
   376  			switch {
   377  			case i == 0 && &member.buffer[0] != &firstBufferBeforeSorting[0]:
   378  				// First member after sorting is not the first member before sorting,
   379  				// so use the prefix of the first member before sorting.
   380  				sorted = append(sorted, commaAndWhitespacePrefix(firstBufferBeforeSorting)...)
   381  				sorted = append(sorted, bytes.TrimLeft(member.buffer, commaAndWhitespace)...)
   382  			case i != 0 && &member.buffer[0] == &firstBufferBeforeSorting[0]:
   383  				// Later member after sorting is the first member before sorting,
   384  				// so use the prefix of the first member after sorting.
   385  				sorted = append(sorted, commaAndWhitespacePrefix(firstBufferAfterSorting)...)
   386  				sorted = append(sorted, bytes.TrimLeft(member.buffer, commaAndWhitespace)...)
   387  			default:
   388  				sorted = append(sorted, member.buffer...)
   389  			}
   390  		}
   391  		if int(afterBody-beforeBody) != len(sorted) {
   392  			panic("BUG: length invariant violated")
   393  		}
   394  		copy(d.s.buf[beforeBody:afterBody], sorted)
   395  
   396  		// Update scratch buffer to the largest amount ever used.
   397  		if len(sorted) > len(*scratch) {
   398  			*scratch = sorted
   399  		}
   400  	case '[':
   401  		for d.PeekKind() != ']' {
   402  			mustReorderObjectsFromDecoder(d, scratch)
   403  		}
   404  		d.ReadToken()
   405  	default:
   406  		if err != nil {
   407  			panic("BUG: " + err.Error())
   408  		}
   409  	}
   410  }
   411  

View as plain text