Source file src/internal/strconv/atoi.go

     1  // Copyright 2009 The Go Authors. All rights reserved.
     2  // Use of this source code is governed by a BSD-style
     3  // license that can be found in the LICENSE file.
     4  
     5  package strconv
     6  
     7  // lower(c) is a lower-case letter if and only if
     8  // c is either that lower-case letter or the equivalent upper-case letter.
     9  // Instead of writing c == 'x' || c == 'X' one can write lower(c) == 'x'.
    10  // Note that lower of non-letters can produce other non-letters.
    11  func lower(c byte) byte {
    12  	return c | ('x' - 'X')
    13  }
    14  
    15  type Error int
    16  
    17  const (
    18  	_ Error = iota
    19  	ErrRange
    20  	ErrSyntax
    21  	ErrBase
    22  	ErrBitSize
    23  )
    24  
    25  func (e Error) Error() string {
    26  	switch e {
    27  	case ErrRange:
    28  		return "value out of range"
    29  	case ErrSyntax:
    30  		return "invalid syntax"
    31  	case ErrBase:
    32  		return "invalid base"
    33  	case ErrBitSize:
    34  		return "invalid bit size"
    35  	}
    36  	return "unknown error"
    37  }
    38  
    39  const intSize = 32 << (^uint(0) >> 63)
    40  
    41  // IntSize is the size in bits of an int or uint value.
    42  const IntSize = intSize
    43  
    44  // ParseUint is like [ParseInt] but for unsigned numbers.
    45  //
    46  // A sign prefix is not permitted.
    47  func ParseUint(s string, base int, bitSize int) (uint64, error) {
    48  	const fnParseUint = "ParseUint"
    49  
    50  	if s == "" {
    51  		return 0, ErrSyntax
    52  	}
    53  
    54  	base0 := base == 0
    55  
    56  	s0 := s
    57  	switch {
    58  	case 2 <= base && base <= 36:
    59  		// valid base; nothing to do
    60  
    61  	case base == 0:
    62  		// Look for octal, hex prefix.
    63  		base = 10
    64  		if s[0] == '0' {
    65  			switch {
    66  			case len(s) >= 3 && lower(s[1]) == 'b':
    67  				base = 2
    68  				s = s[2:]
    69  			case len(s) >= 3 && lower(s[1]) == 'o':
    70  				base = 8
    71  				s = s[2:]
    72  			case len(s) >= 3 && lower(s[1]) == 'x':
    73  				base = 16
    74  				s = s[2:]
    75  			default:
    76  				base = 8
    77  				s = s[1:]
    78  			}
    79  		}
    80  
    81  	default:
    82  		return 0, ErrBase
    83  	}
    84  
    85  	if bitSize == 0 {
    86  		bitSize = IntSize
    87  	} else if bitSize < 0 || bitSize > 64 {
    88  		return 0, ErrBitSize
    89  	}
    90  
    91  	// Cutoff is the smallest number such that cutoff*base > maxUint64.
    92  	// Use compile-time constants for common cases.
    93  	const maxUint64 = 1<<64 - 1
    94  	var cutoff uint64
    95  	switch base {
    96  	case 10:
    97  		cutoff = maxUint64/10 + 1
    98  	case 16:
    99  		cutoff = maxUint64/16 + 1
   100  	default:
   101  		cutoff = maxUint64/uint64(base) + 1
   102  	}
   103  
   104  	maxVal := uint64(1)<<uint(bitSize) - 1
   105  
   106  	underscores := false
   107  	var n uint64
   108  	for _, c := range []byte(s) {
   109  		var d byte
   110  		switch {
   111  		case c == '_' && base0:
   112  			underscores = true
   113  			continue
   114  		case '0' <= c && c <= '9':
   115  			d = c - '0'
   116  		case 'a' <= lower(c) && lower(c) <= 'z':
   117  			d = lower(c) - 'a' + 10
   118  		default:
   119  			return 0, ErrSyntax
   120  		}
   121  
   122  		if d >= byte(base) {
   123  			return 0, ErrSyntax
   124  		}
   125  
   126  		if n >= cutoff {
   127  			// n*base overflows
   128  			return maxVal, ErrRange
   129  		}
   130  		n *= uint64(base)
   131  
   132  		n1 := n + uint64(d)
   133  		if n1 < n || n1 > maxVal {
   134  			// n+d overflows
   135  			return maxVal, ErrRange
   136  		}
   137  		n = n1
   138  	}
   139  
   140  	if underscores && !underscoreOK(s0) {
   141  		return 0, ErrSyntax
   142  	}
   143  
   144  	return n, nil
   145  }
   146  
   147  // ParseInt interprets a string s in the given base (0, 2 to 36) and
   148  // bit size (0 to 64) and returns the corresponding value i.
   149  //
   150  // The string may begin with a leading sign: "+" or "-".
   151  //
   152  // If the base argument is 0, the true base is implied by the string's
   153  // prefix following the sign (if present): 2 for "0b", 8 for "0" or "0o",
   154  // 16 for "0x", and 10 otherwise. Also, for argument base 0 only,
   155  // underscore characters are permitted as defined by the Go syntax for
   156  // [integer literals].
   157  //
   158  // The bitSize argument specifies the integer type
   159  // that the result must fit into. Bit sizes 0, 8, 16, 32, and 64
   160  // correspond to int, int8, int16, int32, and int64.
   161  // If bitSize is below 0 or above 64, an error is returned.
   162  //
   163  // The errors that ParseInt returns have concrete type [*NumError]
   164  // and include err.Num = s. If s is empty or contains invalid
   165  // digits, err.Err = [ErrSyntax] and the returned value is 0;
   166  // if the value corresponding to s cannot be represented by a
   167  // signed integer of the given size, err.Err = [ErrRange] and the
   168  // returned value is the maximum magnitude integer of the
   169  // appropriate bitSize and sign.
   170  //
   171  // [integer literals]: https://go.dev/ref/spec#Integer_literals
   172  func ParseInt(s string, base int, bitSize int) (i int64, err error) {
   173  	const fnParseInt = "ParseInt"
   174  
   175  	if s == "" {
   176  		return 0, ErrSyntax
   177  	}
   178  
   179  	// Pick off leading sign.
   180  	neg := false
   181  	switch s[0] {
   182  	case '+':
   183  		s = s[1:]
   184  	case '-':
   185  		s = s[1:]
   186  		neg = true
   187  	}
   188  
   189  	// Convert unsigned and check range.
   190  	var un uint64
   191  	un, err = ParseUint(s, base, bitSize)
   192  	if err != nil && err != ErrRange {
   193  		return 0, err
   194  	}
   195  
   196  	if bitSize == 0 {
   197  		bitSize = IntSize
   198  	}
   199  
   200  	cutoff := uint64(1 << uint(bitSize-1))
   201  	if !neg && un >= cutoff {
   202  		return int64(cutoff - 1), ErrRange
   203  	}
   204  	if neg && un > cutoff {
   205  		return -int64(cutoff), ErrRange
   206  	}
   207  	n := int64(un)
   208  	if neg {
   209  		n = -n
   210  	}
   211  	return n, nil
   212  }
   213  
   214  // Atoi is equivalent to ParseInt(s, 10, 0), converted to type int.
   215  func Atoi(s string) (int, error) {
   216  	const fnAtoi = "Atoi"
   217  
   218  	sLen := len(s)
   219  	if intSize == 32 && (0 < sLen && sLen < 10) ||
   220  		intSize == 64 && (0 < sLen && sLen < 19) {
   221  		// Fast path for small integers that fit int type.
   222  		s0 := s
   223  		if s[0] == '-' || s[0] == '+' {
   224  			s = s[1:]
   225  			if len(s) < 1 {
   226  				return 0, ErrSyntax
   227  			}
   228  		}
   229  
   230  		n := 0
   231  		for _, ch := range []byte(s) {
   232  			ch -= '0'
   233  			if ch > 9 {
   234  				return 0, ErrSyntax
   235  			}
   236  			n = n*10 + int(ch)
   237  		}
   238  		if s0[0] == '-' {
   239  			n = -n
   240  		}
   241  		return n, nil
   242  	}
   243  
   244  	// Slow path for invalid, big, or underscored integers.
   245  	i64, err := ParseInt(s, 10, 0)
   246  	return int(i64), err
   247  }
   248  
   249  // underscoreOK reports whether the underscores in s are allowed.
   250  // Checking them in this one function lets all the parsers skip over them simply.
   251  // Underscore must appear only between digits or between a base prefix and a digit.
   252  func underscoreOK(s string) bool {
   253  	// saw tracks the last character (class) we saw:
   254  	// ^ for beginning of number,
   255  	// 0 for a digit or base prefix,
   256  	// _ for an underscore,
   257  	// ! for none of the above.
   258  	saw := '^'
   259  	i := 0
   260  
   261  	// Optional sign.
   262  	if len(s) >= 1 && (s[0] == '-' || s[0] == '+') {
   263  		s = s[1:]
   264  	}
   265  
   266  	// Optional base prefix.
   267  	hex := false
   268  	if len(s) >= 2 && s[0] == '0' && (lower(s[1]) == 'b' || lower(s[1]) == 'o' || lower(s[1]) == 'x') {
   269  		i = 2
   270  		saw = '0' // base prefix counts as a digit for "underscore as digit separator"
   271  		hex = lower(s[1]) == 'x'
   272  	}
   273  
   274  	// Number proper.
   275  	for ; i < len(s); i++ {
   276  		// Digits are always okay.
   277  		if '0' <= s[i] && s[i] <= '9' || hex && 'a' <= lower(s[i]) && lower(s[i]) <= 'f' {
   278  			saw = '0'
   279  			continue
   280  		}
   281  		// Underscore must follow digit.
   282  		if s[i] == '_' {
   283  			if saw != '0' {
   284  				return false
   285  			}
   286  			saw = '_'
   287  			continue
   288  		}
   289  		// Underscore must also be followed by digit.
   290  		if saw == '_' {
   291  			return false
   292  		}
   293  		// Saw non-digit, non-underscore.
   294  		saw = '!'
   295  	}
   296  	return saw != '_'
   297  }
   298  

View as plain text