miki

Miki: personal wiki with the fmd markup format
git clone https://wtenison.com/repo/miki.git
Log | Files | Refs | README | LICENSE

commit 1ed68340553763ddf6da850c89fa18e6602bf664
parent d01d3ae2675406feaead139d3e77b1310827ca83
Author: weetat <wtenison@protonmail.com>
Date:   Sat,  1 Mar 2025 07:07:22 +0000

Rework fmd grammar and shorten token names

Commands now take either a word or a {block}, with block bodies split
out as their own rule. Token types are renamed to tWord, tKeyword and
so on, and the parser is updated to match.

Diffstat:
Msrc/fmd/fmd.ebnf | 16+++++++++-------
Msrc/fmd/fmd_test.go | 7+++++--
Msrc/fmd/interpreter.go | 26+++++++++++++++-----------
Msrc/fmd/lexer.go | 64++++++++++++++++++++++++++++++++++++----------------------------
Msrc/fmd/parser.go | 196++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------------------
5 files changed, 200 insertions(+), 109 deletions(-)

diff --git a/src/fmd/fmd.ebnf b/src/fmd/fmd.ebnf @@ -1,15 +1,17 @@ -<fmd> ::= (<command> | <block>)* +# This grammar is a work-in-progress and represents the ideal form +# of the lanuage. It may not be fully implemented and is subject +# to change as I learn. -<command> ::= <keyword> ":" <word>| - <keyword> ":" "{" <block> "}" | - <keyword> "(" <block> ")" ":" <word> | - <keyword> "(" <block> ")" ":" "{" <block> "}" | +<fmd> ::= (<command> | <block> | <body>)* +<command> ::= (<keyword>)+ ":" ( <word> | <block> ) <keyword> ::= "b" | "i" | "h" [1-6] | "@" | "#" | "col" | "img" | "_" -<block> ::= (<word> | <space>)+ -<word> ::= (<letter> | <number> | <symbol>)+ +<block> ::= "{" <body> "}" + +<body> := (<word> | <space>)+ +<word> ::= (<letter> | <number> | <symbol>)+ <space> ::= " " <letter> ::= [a-z] | [A-Z] <number> ::= [0-9]+ | (<number> "." <number>) diff --git a/src/fmd/fmd_test.go b/src/fmd/fmd_test.go @@ -7,8 +7,10 @@ import ( func TestFMD(t *testing.T) { _path := ("/home/weetat/projects/miki/data/fmd/home.fmd") - // WriteFMDtoHTMLFile(_path) t.Log("Path:", _path) + + // WriteFMDtoHTMLFile(_path) + _input, _err := os.ReadFile(_path) if _err != nil { t.Error("_ERROR\tcannot open file:", _err.Error()) @@ -22,6 +24,7 @@ func TestFMD(t *testing.T) { _p := MakeParser(_tokens) _ast := _p.Exec() + // t.Log("AST:", _ast) - t.Log(interpret(_ast)) + t.Log("HTML:\n", interpret(_ast)) } diff --git a/src/fmd/interpreter.go b/src/fmd/interpreter.go @@ -13,11 +13,11 @@ func interpret(n node) string { var _res strings.Builder switch n.nType { - case nodeFMD: + case nRoot: for _, _child := range n.children { _res.WriteString(interpret(_child)) } - case nodeCommand: + case nCommand: switch n.value { case "b": return "<b>" + interpret(n.children[0]) + "</b>" @@ -49,25 +49,29 @@ func interpret(n node) string { default: return "<div>" + interpret(n.children[0]) + "</div>" } - case nodeBlock: - _res.WriteString("<div class=\"__block\">") + case nBlock: + _res.WriteString("<div class=\"__b\">") for _, _child := range n.children { _res.WriteString(interpret(_child)) } _res.WriteString("</div>") - case nodeSpacing: - switch n.value { - case "\n": - return "<br>" - default: - return n.value + case nBody: + for _, _child := range n.children { + _res.WriteString(interpret(_child)) } - case nodeWord: + case nSpacing: + return n.value + case nNewline: + return "<br>\n" + case nWord: if slices.Contains(invalidHTML, n.value) { return "" } return n.value + default: + return n.value } + return _res.String() } diff --git a/src/fmd/lexer.go b/src/fmd/lexer.go @@ -10,22 +10,26 @@ https://www.youtube.com/watch?v=HxaD_trXwRE */ type tokenType int const ( + // WORDS - tokenWord tokenType = iota // 0. (letters, numbers, symbols)+ - tokenKeyword // 1. (valid keywords) + tWord tokenType = iota // 0. (letters, numbers, symbols)+ + tKeyword // 1. (valid keywords) // SPACING - tokenSpace // 2. " " - tokenNewline // 3. \n - tokenTab // 4. \t + tSpace // 2. " " + tNewline // 3. \n + tTab // 4. \t // OPERATORS - tokenEscape // 5. \ - tokenColon // 6. : - tokenLBrace // 7. { - tokenRBrace // 8. } - tokenLParen // 9. ( - tokenRParen // 10. ) + tEscape // 5. \ + tColon // 6. : + tLBrace // 7. { + tRBrace // 8. } + tLParen // 9. ( + tRParen // 10. ) + + tError // 11 + tEmpty // 12 ) type token struct { @@ -53,44 +57,44 @@ type lexer struct { } func MakeLexer(input []byte) *lexer { - return &lexer{input: input, pos: 0} + return &lexer{input: input, pos: 0, tokens: []token{}} } func (l *lexer) Exec() []token { + for l.pos < len(l.input) { _currentSymbol := l.input[l.pos] switch { case _currentSymbol == ' ': - l.tokens = append(l.tokens, token{tokenSpace, " "}) + l.tokens = append(l.tokens, token{tSpace, " "}) l.pos++ case _currentSymbol == '\n': - l.tokens = append(l.tokens, token{tokenNewline, "\n"}) + l.tokens = append(l.tokens, token{tNewline, "\n"}) l.pos++ case _currentSymbol == '\t': - l.tokens = append(l.tokens, token{tokenTab, "\t"}) + l.tokens = append(l.tokens, token{tTab, "\t"}) l.pos++ case _currentSymbol == '\\': - l.tokens = append(l.tokens, token{tokenEscape, "\\"}) + l.tokens = append(l.tokens, token{tEscape, "\\"}) l.pos++ case _currentSymbol == ':': - l.tokens = append(l.tokens, token{tokenColon, ":"}) + l.tokens = append(l.tokens, token{tColon, ":"}) l.pos++ case _currentSymbol == '{': - l.tokens = append(l.tokens, token{tokenLBrace, "{"}) + l.tokens = append(l.tokens, token{tLBrace, "{"}) l.pos++ case _currentSymbol == '}': - l.tokens = append(l.tokens, token{tokenRBrace, "}"}) + l.tokens = append(l.tokens, token{tRBrace, "}"}) l.pos++ case _currentSymbol == '(': - l.tokens = append(l.tokens, token{tokenLParen, "("}) + l.tokens = append(l.tokens, token{tLParen, "("}) l.pos++ case _currentSymbol == ')': - l.tokens = append(l.tokens, token{tokenRParen, ")"}) + l.tokens = append(l.tokens, token{tRParen, ")"}) l.pos++ default: { // NOTE: inlined check until good reason to abstract arrives. - var _tokenString string _start := l.pos var _symbols []string @@ -98,15 +102,19 @@ func (l *lexer) Exec() []token { _symbols = append(_symbols, operatorSymbols...) for l.pos < len(l.input) && !slices.Contains(_symbols, string(l.input[l.pos])) { - l.pos++ + l.pos += 1 } - _tokenString = string(l.input[_start:l.pos]) + _tokenStr := string(l.input[_start:l.pos]) - if slices.Contains(keywords, _tokenString) { - l.tokens = append(l.tokens, token{tokenKeyword, _tokenString}) - } else if _tokenString != "" { - l.tokens = append(l.tokens, token{tokenWord, _tokenString}) + if slices.Contains(keywords, _tokenStr) { + if _start == 0 || l.tokens[len(l.tokens)-1].tType != tEscape { + l.tokens = append(l.tokens, token{tKeyword, _tokenStr}) + } else { + l.tokens = append(l.tokens, token{tWord, _tokenStr}) + } + } else if _tokenStr != "" { + l.tokens = append(l.tokens, token{tWord, _tokenStr}) } } } diff --git a/src/fmd/parser.go b/src/fmd/parser.go @@ -2,129 +2,203 @@ package fmd import ( "fmt" + "log" ) type nodeType string const ( - nodeFMD nodeType = "fmd" + nRoot nodeType = "root" - nodeBlock nodeType = "block" - nodeCommand nodeType = "command" - nodeWord nodeType = "literal" - nodeOperator nodeType = "operator" - nodeSpacing nodeType = "spacing" + nCommand nodeType = "command" + nBlock nodeType = "block" + nBody nodeType = "body" + + nWord nodeType = "word" + nOperator nodeType = "operator" + nSpacing nodeType = "spacing" + nNewline nodeType = "newline" + + nError nodeType = "error" ) type node struct { nType nodeType value string children []node - prevNode *node } type parser struct { - tokens []token - pos int - prevNode *node + tokens []token + pos int + depth int + errorNodes []node } func MakeParser(tokens []token) *parser { + if tokens == nil { + log.Printf("_DEBUG:MakeParser:token array is empty") + _tokens := []token{{tType: tError, Value: "_ERROR:MakeParser:empty token string"}} + return &parser{tokens: _tokens} + } return &parser{tokens: tokens, pos: 0} } func (p *parser) Exec() node { - _fmd := p.makeNode(nodeFMD, "_fmd_") + _fmd := p.makeNode(nRoot, token{tType: tEmpty, Value: "_fmd_"}) for p.pos < len(p.tokens) { - if p.isNextCommand() { // keyword:? + if p.isNextCommand() { _fmd.children = append(_fmd.children, p.parseCommand()) - } else { // block + } else if p.peekToken(0).tType == tLBrace { _fmd.children = append(_fmd.children, p.parseBlock()) + } else { + _fmd.children = append(_fmd.children, p.parseBody()) } } + return _fmd } func (p *parser) parseCommand() node { - _keyword := p.consume(tokenKeyword).Value - _command := p.makeNode(nodeCommand, _keyword) - p.consume(tokenColon) + // TODO: Added function for parseing multiple keywords; parseKeyword() (e.g. bi@:... ) + _command := p.makeNode(nCommand, p.consume(tKeyword)) - // keyword:{block} - if p.peekNextN(tokenLBrace, 0) { - p.consume(tokenLBrace) + p.consume(tColon) + if p.isNextCommand() { + _command.children = []node{p.parseCommand()} + } else if p.isTokenType(tLBrace, 0) { _command.children = []node{p.parseBlock()} - - p.consume(tokenRBrace) - } else { // keyword:"literal" - _command.children = []node{p.parseWord()} + } else { + _command.children = []node{p.makeNode(nWord, p.consume(tWord))} } return _command } func (p *parser) parseBlock() node { - _block := p.makeNode(nodeBlock, "") - for p.pos < len(p.tokens) && p.current().tType != tokenRBrace { + p.depth += 1 + + _block := p.makeNode(nBlock, token{tType: tEmpty, Value: ""}) + + p.consume(tLBrace) + + for p.pos < len(p.tokens) && p.peekToken(0).tType != tRBrace { switch { - case p.peekNextN(tokenWord, 0): - _block.children = append(_block.children, p.parseWord()) - case p.peekNextN(tokenSpace, 0): - _block.children = append(_block.children, p.makeNode(nodeSpacing, " ")) - p.pos++ - case p.peekNextN(tokenTab, 0): - _block.children = append(_block.children, p.makeNode(nodeSpacing, "\t")) - p.pos++ - case p.peekNextN(tokenNewline, 0): - _block.children = append(_block.children, p.makeNode(nodeSpacing, "\n")) - p.pos++ case p.isNextCommand(): _block.children = append(_block.children, p.parseCommand()) + default: + _block.children = append(_block.children, p.parseBody()) } } + p.depth -= 1 + + p.consume(tRBrace) + return _block } -func (p *parser) parseWord() node { - _wordVal := p.consume(tokenWord).Value - _word := p.makeNode(nodeWord, _wordVal) - return _word -} +func (p *parser) parseBody() node { + _body := p.makeNode(nBody, token{tType: tEmpty, Value: ""}) -func (p *parser) isNextCommand() bool { - return p.peekNextN(tokenKeyword, 0) && - p.peekNextN(tokenColon, 1) && - p.peekNextN(tokenWord, 2) || - p.peekNextN(tokenLBrace, 2) + for p.pos < len(p.tokens) { + + if p.peekToken(0).tType == tRBrace && p.depth > 0 { + break + } + + _pToken := p.peekToken(0) + + switch _pToken.tType { + case tWord: + _body.children = append(_body.children, p.makeNode(nWord, p.consume(tWord))) + case tSpace: + _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tSpace))) + case tTab: + _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tTab))) + case tNewline: + _body.children = append(_body.children, p.makeNode(nNewline, p.consume(tNewline))) + case tEscape: + p.consume(tEscape) + case tColon: + p.consume(tColon) + case tLBrace: + _body.children = append(_body.children, p.parseBlock()) + case tRBrace: + _body.children = append(_body.children, p.errorNode("no matching LBrace")) + default: + if p.isNextCommand() { + _body.children = append(_body.children, p.parseCommand()) + } else { + _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, Value: "unknown token;"})) + p.pos += 1 + log.Printf("INFO:block:error:%s", _body.children[len(_body.children)-1]) + } + } + } + + return _body } -func (p *parser) makeNode(nType nodeType, value string) node { - _n := node{nType: nType, value: value, prevNode: p.prevNode} - p.prevNode = &_n +//// UTILITY FUNCTIONS + +func (p *parser) makeNode(nType nodeType, iToken token) node { + + if iToken.tType == tError { + return p.errorNode(iToken.Value) + } + + _n := node{nType: nType, value: iToken.Value} return _n } -func (p *parser) peekNextN(t tokenType, n int) bool { - _res := p.pos+n < len(p.tokens) && p.tokens[p.pos+n].tType == t - return _res +func (p *parser) errorNode(message string) node { + + _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, message) + p.errorNodes = append(p.errorNodes, node{nType: nError, value: _errorValue}) + + return p.errorNodes[len(p.errorNodes)-1] +} + +func (p *parser) consume(expectedTType tokenType) token { + + _token := p.tokens[p.pos] + + if !p.isTokenType(expectedTType, 0) { + _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, "invalid token type") + _token = token{tError, _errorValue} + } + + p.pos += 1 + + // log.Printf("INFO:consume:%v", _token.Value) + return _token } -func (p *parser) consume(t tokenType) token { - if p.peekNextN(t, 0) { - token := p.tokens[p.pos] - p.pos++ - return token +func (p *parser) peekToken(numOfTokensAhead int) token { + + if p.pos+numOfTokensAhead < len(p.tokens) { + return p.tokens[p.pos] } - // TODO: Error out gracefully and send empty AST. - panic(fmt.Sprintf("__PANIC\tfmd:parser:Invalid token:%v", t)) + + return token{tError, "_ERROR:peekToken:attempting to peek beyond array size"} +} + +func (p *parser) isTokenType(expectedTType tokenType, indexOffset int) bool { + return p.pos+indexOffset > -1 && + p.pos+indexOffset < len(p.tokens) && + p.tokens[p.pos+indexOffset].tType == expectedTType } -func (p *parser) current() token { - return p.tokens[p.pos] +func (p *parser) isNextCommand() bool { + return !p.isTokenType(tEscape, -1) && + p.isTokenType(tKeyword, 0) && + p.isTokenType(tColon, 1) && + p.isTokenType(tWord, 2) || + p.isTokenType(tLBrace, 2) }