commit 1ed68340553763ddf6da850c89fa18e6602bf664
parent d01d3ae2675406feaead139d3e77b1310827ca83
Author: weetat <wtenison@protonmail.com>
Date: Sat, 1 Mar 2025 07:07:22 +0000
Rework fmd grammar and shorten token names
Commands now take either a word or a {block}, with block bodies split
out as their own rule. Token types are renamed to tWord, tKeyword and
so on, and the parser is updated to match.
Diffstat:
5 files changed, 200 insertions(+), 109 deletions(-)
diff --git a/src/fmd/fmd.ebnf b/src/fmd/fmd.ebnf
@@ -1,15 +1,17 @@
-<fmd> ::= (<command> | <block>)*
+# This grammar is a work-in-progress and represents the ideal form
+# of the lanuage. It may not be fully implemented and is subject
+# to change as I learn.
-<command> ::= <keyword> ":" <word>|
- <keyword> ":" "{" <block> "}" |
- <keyword> "(" <block> ")" ":" <word> |
- <keyword> "(" <block> ")" ":" "{" <block> "}" |
+<fmd> ::= (<command> | <block> | <body>)*
+<command> ::= (<keyword>)+ ":" ( <word> | <block> )
<keyword> ::= "b" | "i" | "h" [1-6] | "@" | "#" | "col" | "img" | "_"
-<block> ::= (<word> | <space>)+
-<word> ::= (<letter> | <number> | <symbol>)+
+<block> ::= "{" <body> "}"
+
+<body> := (<word> | <space>)+
+<word> ::= (<letter> | <number> | <symbol>)+
<space> ::= " "
<letter> ::= [a-z] | [A-Z]
<number> ::= [0-9]+ | (<number> "." <number>)
diff --git a/src/fmd/fmd_test.go b/src/fmd/fmd_test.go
@@ -7,8 +7,10 @@ import (
func TestFMD(t *testing.T) {
_path := ("/home/weetat/projects/miki/data/fmd/home.fmd")
- // WriteFMDtoHTMLFile(_path)
t.Log("Path:", _path)
+
+ // WriteFMDtoHTMLFile(_path)
+
_input, _err := os.ReadFile(_path)
if _err != nil {
t.Error("_ERROR\tcannot open file:", _err.Error())
@@ -22,6 +24,7 @@ func TestFMD(t *testing.T) {
_p := MakeParser(_tokens)
_ast := _p.Exec()
+ // t.Log("AST:", _ast)
- t.Log(interpret(_ast))
+ t.Log("HTML:\n", interpret(_ast))
}
diff --git a/src/fmd/interpreter.go b/src/fmd/interpreter.go
@@ -13,11 +13,11 @@ func interpret(n node) string {
var _res strings.Builder
switch n.nType {
- case nodeFMD:
+ case nRoot:
for _, _child := range n.children {
_res.WriteString(interpret(_child))
}
- case nodeCommand:
+ case nCommand:
switch n.value {
case "b":
return "<b>" + interpret(n.children[0]) + "</b>"
@@ -49,25 +49,29 @@ func interpret(n node) string {
default:
return "<div>" + interpret(n.children[0]) + "</div>"
}
- case nodeBlock:
- _res.WriteString("<div class=\"__block\">")
+ case nBlock:
+ _res.WriteString("<div class=\"__b\">")
for _, _child := range n.children {
_res.WriteString(interpret(_child))
}
_res.WriteString("</div>")
- case nodeSpacing:
- switch n.value {
- case "\n":
- return "<br>"
- default:
- return n.value
+ case nBody:
+ for _, _child := range n.children {
+ _res.WriteString(interpret(_child))
}
- case nodeWord:
+ case nSpacing:
+ return n.value
+ case nNewline:
+ return "<br>\n"
+ case nWord:
if slices.Contains(invalidHTML, n.value) {
return ""
}
return n.value
+ default:
+ return n.value
}
+
return _res.String()
}
diff --git a/src/fmd/lexer.go b/src/fmd/lexer.go
@@ -10,22 +10,26 @@ https://www.youtube.com/watch?v=HxaD_trXwRE */
type tokenType int
const (
+
// WORDS
- tokenWord tokenType = iota // 0. (letters, numbers, symbols)+
- tokenKeyword // 1. (valid keywords)
+ tWord tokenType = iota // 0. (letters, numbers, symbols)+
+ tKeyword // 1. (valid keywords)
// SPACING
- tokenSpace // 2. " "
- tokenNewline // 3. \n
- tokenTab // 4. \t
+ tSpace // 2. " "
+ tNewline // 3. \n
+ tTab // 4. \t
// OPERATORS
- tokenEscape // 5. \
- tokenColon // 6. :
- tokenLBrace // 7. {
- tokenRBrace // 8. }
- tokenLParen // 9. (
- tokenRParen // 10. )
+ tEscape // 5. \
+ tColon // 6. :
+ tLBrace // 7. {
+ tRBrace // 8. }
+ tLParen // 9. (
+ tRParen // 10. )
+
+ tError // 11
+ tEmpty // 12
)
type token struct {
@@ -53,44 +57,44 @@ type lexer struct {
}
func MakeLexer(input []byte) *lexer {
- return &lexer{input: input, pos: 0}
+ return &lexer{input: input, pos: 0, tokens: []token{}}
}
func (l *lexer) Exec() []token {
+
for l.pos < len(l.input) {
_currentSymbol := l.input[l.pos]
switch {
case _currentSymbol == ' ':
- l.tokens = append(l.tokens, token{tokenSpace, " "})
+ l.tokens = append(l.tokens, token{tSpace, " "})
l.pos++
case _currentSymbol == '\n':
- l.tokens = append(l.tokens, token{tokenNewline, "\n"})
+ l.tokens = append(l.tokens, token{tNewline, "\n"})
l.pos++
case _currentSymbol == '\t':
- l.tokens = append(l.tokens, token{tokenTab, "\t"})
+ l.tokens = append(l.tokens, token{tTab, "\t"})
l.pos++
case _currentSymbol == '\\':
- l.tokens = append(l.tokens, token{tokenEscape, "\\"})
+ l.tokens = append(l.tokens, token{tEscape, "\\"})
l.pos++
case _currentSymbol == ':':
- l.tokens = append(l.tokens, token{tokenColon, ":"})
+ l.tokens = append(l.tokens, token{tColon, ":"})
l.pos++
case _currentSymbol == '{':
- l.tokens = append(l.tokens, token{tokenLBrace, "{"})
+ l.tokens = append(l.tokens, token{tLBrace, "{"})
l.pos++
case _currentSymbol == '}':
- l.tokens = append(l.tokens, token{tokenRBrace, "}"})
+ l.tokens = append(l.tokens, token{tRBrace, "}"})
l.pos++
case _currentSymbol == '(':
- l.tokens = append(l.tokens, token{tokenLParen, "("})
+ l.tokens = append(l.tokens, token{tLParen, "("})
l.pos++
case _currentSymbol == ')':
- l.tokens = append(l.tokens, token{tokenRParen, ")"})
+ l.tokens = append(l.tokens, token{tRParen, ")"})
l.pos++
default:
{ // NOTE: inlined check until good reason to abstract arrives.
- var _tokenString string
_start := l.pos
var _symbols []string
@@ -98,15 +102,19 @@ func (l *lexer) Exec() []token {
_symbols = append(_symbols, operatorSymbols...)
for l.pos < len(l.input) && !slices.Contains(_symbols, string(l.input[l.pos])) {
- l.pos++
+ l.pos += 1
}
- _tokenString = string(l.input[_start:l.pos])
+ _tokenStr := string(l.input[_start:l.pos])
- if slices.Contains(keywords, _tokenString) {
- l.tokens = append(l.tokens, token{tokenKeyword, _tokenString})
- } else if _tokenString != "" {
- l.tokens = append(l.tokens, token{tokenWord, _tokenString})
+ if slices.Contains(keywords, _tokenStr) {
+ if _start == 0 || l.tokens[len(l.tokens)-1].tType != tEscape {
+ l.tokens = append(l.tokens, token{tKeyword, _tokenStr})
+ } else {
+ l.tokens = append(l.tokens, token{tWord, _tokenStr})
+ }
+ } else if _tokenStr != "" {
+ l.tokens = append(l.tokens, token{tWord, _tokenStr})
}
}
}
diff --git a/src/fmd/parser.go b/src/fmd/parser.go
@@ -2,129 +2,203 @@ package fmd
import (
"fmt"
+ "log"
)
type nodeType string
const (
- nodeFMD nodeType = "fmd"
+ nRoot nodeType = "root"
- nodeBlock nodeType = "block"
- nodeCommand nodeType = "command"
- nodeWord nodeType = "literal"
- nodeOperator nodeType = "operator"
- nodeSpacing nodeType = "spacing"
+ nCommand nodeType = "command"
+ nBlock nodeType = "block"
+ nBody nodeType = "body"
+
+ nWord nodeType = "word"
+ nOperator nodeType = "operator"
+ nSpacing nodeType = "spacing"
+ nNewline nodeType = "newline"
+
+ nError nodeType = "error"
)
type node struct {
nType nodeType
value string
children []node
- prevNode *node
}
type parser struct {
- tokens []token
- pos int
- prevNode *node
+ tokens []token
+ pos int
+ depth int
+ errorNodes []node
}
func MakeParser(tokens []token) *parser {
+ if tokens == nil {
+ log.Printf("_DEBUG:MakeParser:token array is empty")
+ _tokens := []token{{tType: tError, Value: "_ERROR:MakeParser:empty token string"}}
+ return &parser{tokens: _tokens}
+ }
return &parser{tokens: tokens, pos: 0}
}
func (p *parser) Exec() node {
- _fmd := p.makeNode(nodeFMD, "_fmd_")
+ _fmd := p.makeNode(nRoot, token{tType: tEmpty, Value: "_fmd_"})
for p.pos < len(p.tokens) {
- if p.isNextCommand() { // keyword:?
+ if p.isNextCommand() {
_fmd.children = append(_fmd.children, p.parseCommand())
- } else { // block
+ } else if p.peekToken(0).tType == tLBrace {
_fmd.children = append(_fmd.children, p.parseBlock())
+ } else {
+ _fmd.children = append(_fmd.children, p.parseBody())
}
}
+
return _fmd
}
func (p *parser) parseCommand() node {
- _keyword := p.consume(tokenKeyword).Value
- _command := p.makeNode(nodeCommand, _keyword)
- p.consume(tokenColon)
+ // TODO: Added function for parseing multiple keywords; parseKeyword() (e.g. bi@:... )
+ _command := p.makeNode(nCommand, p.consume(tKeyword))
- // keyword:{block}
- if p.peekNextN(tokenLBrace, 0) {
- p.consume(tokenLBrace)
+ p.consume(tColon)
+ if p.isNextCommand() {
+ _command.children = []node{p.parseCommand()}
+ } else if p.isTokenType(tLBrace, 0) {
_command.children = []node{p.parseBlock()}
-
- p.consume(tokenRBrace)
- } else { // keyword:"literal"
- _command.children = []node{p.parseWord()}
+ } else {
+ _command.children = []node{p.makeNode(nWord, p.consume(tWord))}
}
return _command
}
func (p *parser) parseBlock() node {
- _block := p.makeNode(nodeBlock, "")
- for p.pos < len(p.tokens) && p.current().tType != tokenRBrace {
+ p.depth += 1
+
+ _block := p.makeNode(nBlock, token{tType: tEmpty, Value: ""})
+
+ p.consume(tLBrace)
+
+ for p.pos < len(p.tokens) && p.peekToken(0).tType != tRBrace {
switch {
- case p.peekNextN(tokenWord, 0):
- _block.children = append(_block.children, p.parseWord())
- case p.peekNextN(tokenSpace, 0):
- _block.children = append(_block.children, p.makeNode(nodeSpacing, " "))
- p.pos++
- case p.peekNextN(tokenTab, 0):
- _block.children = append(_block.children, p.makeNode(nodeSpacing, "\t"))
- p.pos++
- case p.peekNextN(tokenNewline, 0):
- _block.children = append(_block.children, p.makeNode(nodeSpacing, "\n"))
- p.pos++
case p.isNextCommand():
_block.children = append(_block.children, p.parseCommand())
+ default:
+ _block.children = append(_block.children, p.parseBody())
}
}
+ p.depth -= 1
+
+ p.consume(tRBrace)
+
return _block
}
-func (p *parser) parseWord() node {
- _wordVal := p.consume(tokenWord).Value
- _word := p.makeNode(nodeWord, _wordVal)
- return _word
-}
+func (p *parser) parseBody() node {
+ _body := p.makeNode(nBody, token{tType: tEmpty, Value: ""})
-func (p *parser) isNextCommand() bool {
- return p.peekNextN(tokenKeyword, 0) &&
- p.peekNextN(tokenColon, 1) &&
- p.peekNextN(tokenWord, 2) ||
- p.peekNextN(tokenLBrace, 2)
+ for p.pos < len(p.tokens) {
+
+ if p.peekToken(0).tType == tRBrace && p.depth > 0 {
+ break
+ }
+
+ _pToken := p.peekToken(0)
+
+ switch _pToken.tType {
+ case tWord:
+ _body.children = append(_body.children, p.makeNode(nWord, p.consume(tWord)))
+ case tSpace:
+ _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tSpace)))
+ case tTab:
+ _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tTab)))
+ case tNewline:
+ _body.children = append(_body.children, p.makeNode(nNewline, p.consume(tNewline)))
+ case tEscape:
+ p.consume(tEscape)
+ case tColon:
+ p.consume(tColon)
+ case tLBrace:
+ _body.children = append(_body.children, p.parseBlock())
+ case tRBrace:
+ _body.children = append(_body.children, p.errorNode("no matching LBrace"))
+ default:
+ if p.isNextCommand() {
+ _body.children = append(_body.children, p.parseCommand())
+ } else {
+ _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, Value: "unknown token;"}))
+ p.pos += 1
+ log.Printf("INFO:block:error:%s", _body.children[len(_body.children)-1])
+ }
+ }
+ }
+
+ return _body
}
-func (p *parser) makeNode(nType nodeType, value string) node {
- _n := node{nType: nType, value: value, prevNode: p.prevNode}
- p.prevNode = &_n
+//// UTILITY FUNCTIONS
+
+func (p *parser) makeNode(nType nodeType, iToken token) node {
+
+ if iToken.tType == tError {
+ return p.errorNode(iToken.Value)
+ }
+
+ _n := node{nType: nType, value: iToken.Value}
return _n
}
-func (p *parser) peekNextN(t tokenType, n int) bool {
- _res := p.pos+n < len(p.tokens) && p.tokens[p.pos+n].tType == t
- return _res
+func (p *parser) errorNode(message string) node {
+
+ _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, message)
+ p.errorNodes = append(p.errorNodes, node{nType: nError, value: _errorValue})
+
+ return p.errorNodes[len(p.errorNodes)-1]
+}
+
+func (p *parser) consume(expectedTType tokenType) token {
+
+ _token := p.tokens[p.pos]
+
+ if !p.isTokenType(expectedTType, 0) {
+ _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, "invalid token type")
+ _token = token{tError, _errorValue}
+ }
+
+ p.pos += 1
+
+ // log.Printf("INFO:consume:%v", _token.Value)
+ return _token
}
-func (p *parser) consume(t tokenType) token {
- if p.peekNextN(t, 0) {
- token := p.tokens[p.pos]
- p.pos++
- return token
+func (p *parser) peekToken(numOfTokensAhead int) token {
+
+ if p.pos+numOfTokensAhead < len(p.tokens) {
+ return p.tokens[p.pos]
}
- // TODO: Error out gracefully and send empty AST.
- panic(fmt.Sprintf("__PANIC\tfmd:parser:Invalid token:%v", t))
+
+ return token{tError, "_ERROR:peekToken:attempting to peek beyond array size"}
+}
+
+func (p *parser) isTokenType(expectedTType tokenType, indexOffset int) bool {
+ return p.pos+indexOffset > -1 &&
+ p.pos+indexOffset < len(p.tokens) &&
+ p.tokens[p.pos+indexOffset].tType == expectedTType
}
-func (p *parser) current() token {
- return p.tokens[p.pos]
+func (p *parser) isNextCommand() bool {
+ return !p.isTokenType(tEscape, -1) &&
+ p.isTokenType(tKeyword, 0) &&
+ p.isTokenType(tColon, 1) &&
+ p.isTokenType(tWord, 2) ||
+ p.isTokenType(tLBrace, 2)
}