all repos

clerk @ 17c786e0f3f26481a635695160d516a7039a3bef

missing tooling for ledger/hledger
2 files changed, 17 insertions(+), 3 deletions(-)
parser,lexxer: add fast paths
Author: Oleksandr Smirnov olexsmir@gmail.com
Committed at: 2026-09-03 13:51:47 +0300
Authored at: 2026-09-03 13:16:28 +0300
Change ID: nrnwowuvppkmlmktvwmruyznlpxpvvkq
Parent: 68a2727
M journal/lexer/lexer.go
···
        1
        1
         package lexer

      
        2
        2
         

      
        3
        3
         import (

      
        4
        
        -	"strings"

      
        5
        4
         	"unicode"

      
        6
        5
         	"unicode/utf8"

      
        7
        6
         	"unsafe"

      ···
        556
        555
         

      
        557
        556
         func (l *Lexer) lexNumber() token.Token {

      
        558
        557
         	s := l.save()

      
        
        558
        +	isDecimal := false

      
        559
        559
         	for {

      
        560
        560
         		if l.isDigit() || l.ch == '.' || l.ch == ',' || l.ch == '_' || l.ch == '\'' {

      
        
        561
        +			if l.ch == '.' || l.ch == ',' {

      
        
        562
        +				isDecimal = true

      
        
        563
        +			}

      
        561
        564
         			l.advance()

      
        562
        565
         		} else if l.ch == ' ' && (l.peek() >= '0' && l.peek() <= '9') {

      
        
        566
        +			isDecimal = true

      
        563
        567
         			l.advance()

      
        564
        568
         		} else if l.ch == 'e' || l.ch == 'E' {

      
        565
        569
         			// exponent: consume only when E[sign]digit, so `10E ` stays an integer

      ···
        570
        574
         			if p >= len(l.input) || l.input[p] < '0' || l.input[p] > '9' {

      
        571
        575
         				break

      
        572
        576
         			}

      
        
        577
        +			isDecimal = true

      
        573
        578
         			l.advance() // e/E

      
        574
        579
         			if l.ch == '+' || l.ch == '-' {

      
        575
        580
         				l.advance()

      ···
        583
        588
         	}

      
        584
        589
         	lit := l.lit(s)

      
        585
        590
         	kind := token.INT

      
        586
        
        -	if strings.ContainsAny(lit, "., eE") {

      
        
        591
        +	if isDecimal {

      
        587
        592
         		kind = token.DECIMAL

      
        588
        593
         	}

      
        589
        594
         	return token.Token{Type: kind, Literal: lit, Span: l.span(s)}

      ···
        704
        709
         }

      
        705
        710
         

      
        706
        711
         func (l *Lexer) peek() rune {

      
        
        712
        +	if l.rpos < len(l.input) && l.input[l.rpos] < utf8.RuneSelf {

      
        
        713
        +		return rune(l.input[l.rpos])

      
        
        714
        +	}

      
        707
        715
         	r, _ := utf8.DecodeRune(l.input[l.rpos:])

      
        708
        716
         	return r

      
        709
        717
         }

      ···
        721
        729
         		(l.ch >= 'A' && l.ch <= 'Z')

      
        722
        730
         }

      
        723
        731
         

      
        724
        
        -func (l *Lexer) isTwoSpaces() bool { return l.ch == ' ' && l.peek() == ' ' }

      
        
        732
        +func (l *Lexer) isTwoSpaces() bool {

      
        
        733
        +	return l.ch == ' ' && l.rpos < len(l.input) && l.input[l.rpos] == ' '

      
        
        734
        +}

      
        725
        735
         

      
        726
        736
         func (l *Lexer) isDateSep() bool { return l.ch == '-' || l.ch == '/' || l.ch == '.' }

      
        727
        737
         

      
M journal/parser/parser.go
···
        1399
        1399
         }

      
        1400
        1400
         

      
        1401
        1401
         func normalizeLiteral(lit string, thousands, decimal byte) string {

      
        
        1402
        +	// fast path: no separators to strip and the decimal mark is already '.'

      
        
        1403
        +	if thousands == 0 && (decimal == 0 || decimal == '.') {

      
        
        1404
        +		return lit

      
        
        1405
        +	}

      
        1402
        1406
         	var b strings.Builder

      
        1403
        1407
         	for _, ch := range []byte(lit) {

      
        1404
        1408
         		if thousands != 0 && ch == thousands {