miki

Miki: personal wiki with the fmd markup format
git clone https://wtenison.com/repo/miki.git
Log | Files | Refs | README | LICENSE

commit 46cb98c0bbecb439ad744f61b53eed9c70455e33
parent c25ceb5bdd7937504c28ba68e24090b38efdbd51
Author: weetat <wtenison@protonmail.com>
Date:   Sun,  2 Mar 2025 01:25:51 +0000

Support escaping commands and backslashes in fmd

Diffstat:
Msrc/fmd/fmd.ebnf | 9+++++++--
Msrc/fmd/fmd_test.go | 30++++++++++++++++--------------
Msrc/fmd/interpreter.go | 2+-
Msrc/fmd/lexer.go | 46++++++++++++++++++++++++++++++----------------
Msrc/fmd/parser.go | 99++++++++++++++++++++++++++++++++++++++++++++++++-------------------------------
5 files changed, 114 insertions(+), 72 deletions(-)

diff --git a/src/fmd/fmd.ebnf b/src/fmd/fmd.ebnf @@ -4,15 +4,20 @@ <fmd> ::= (<command> | <block> | <body>)* -<command> ::= (<keyword>)+ ":" ( <word> | <block> ) +<command> ::= (<command_id>)+ ":" <word_or_block> +<command_id> ::= <keyword> | <keyword> "(" <params> ")" <keyword> ::= "b" | "i" | "h" [1-6] | "@" | "#" | "col" | "img" | "_" +<params> ::= ("id=" | "class=" | "ref=") <word_or_block> -<block> ::= "{" <body> "}" +<word_or_block> ::= <word> | <block> +<block> ::= "{" <body> "}" <body> := (<word> | <space>)+ <word> ::= (<letter> | <number> | <symbol>)+ + <space> ::= " " <letter> ::= [a-z] | [A-Z] <number> ::= [0-9]+ | (<number> "." <number>) + <symbol> ::= "." | "," | ";" | "-" | "!" | "#" | "$" | "%" | "^" | "&" | "*" | "(" | ")" \ No newline at end of file diff --git a/src/fmd/fmd_test.go b/src/fmd/fmd_test.go @@ -1,29 +1,31 @@ package fmd import ( + "os" "testing" ) func TestFMD(t *testing.T) { - _path := ("/home/weetat/projects/miki/data/README.fmd") + _path := ("/home/weetat/projects/miki/data/fmd/home.fmd") t.Log("Path:", _path) - WriteFMDToHTMLFile(_path, "/home/weetat/projects/miki/README.md") + // WriteFMDToHTMLFile(_path, "/home/weetat/projects/miki/README.md") - // _input, _err := os.ReadFile(_path) - // if _err != nil { - // t.Error("_ERROR\tcannot open file:", _err.Error()) - // return - // } + _input, _err := os.ReadFile(_path) + if _err != nil { + t.Error("_ERROR\tcannot open file:", _err.Error()) + return + } - // _l := MakeLexer(_input) - // _tokens := _l.Exec() + _l := MakeLexer(_input) + _tokens := _l.Exec() - // t.Log("Tokens:", _tokens) + for _, _t := range _tokens { + t.Log("Tokens:", _t) + } - // _p := MakeParser(_tokens) - // _ast := _p.Exec() - // t.Log("AST:", _ast) + _p := MakeParser(_tokens) + _ast := _p.Exec() - // t.Log("HTML:\n", interpret(_ast)) + t.Log("HTML:\n", interpret(_ast)) } diff --git a/src/fmd/interpreter.go b/src/fmd/interpreter.go @@ -62,7 +62,7 @@ func interpret(n node) string { case nSpacing: return n.value case nNewline: - if n.pNode.nType != nNewline { + if n.nParent.nType != nNewline { return "<br>\n" } else { return "\n" diff --git a/src/fmd/lexer.go b/src/fmd/lexer.go @@ -1,6 +1,7 @@ package fmd import ( + "log" "slices" ) @@ -28,17 +29,18 @@ const ( tLParen // 9. ( tRParen // 10. ) - tError // 11 - tEmpty // 12 + // UTILS + tError // 11. error + tEmpty // 12. nil ) type token struct { tType tokenType - Value string + value string } var operatorSymbols = []string{"\\", ":", "{", "}", "(", ")"} -var spacingSymbols = []string{" ", "\n", "\t", ","} +var spacingSymbols = []string{" ", "\n", "\t"} var keywords = []string{ "b", // bold "i", // italic @@ -51,9 +53,9 @@ var keywords = []string{ } type lexer struct { - input []byte // text input to scan - pos int // the start position for scanning - tokens []token + input []byte // text input to scan + pos int // the start position for scanning + tokens []token // the output token stream } func MakeLexer(input []byte) *lexer { @@ -61,6 +63,9 @@ func MakeLexer(input []byte) *lexer { } func (l *lexer) Exec() []token { + var _symbols []string + _symbols = append(_symbols, spacingSymbols...) + _symbols = append(_symbols, operatorSymbols...) for l.pos < len(l.input) { _currentSymbol := l.input[l.pos] @@ -76,7 +81,11 @@ func (l *lexer) Exec() []token { l.tokens = append(l.tokens, token{tTab, "\t"}) l.pos++ case _currentSymbol == '\\': - l.tokens = append(l.tokens, token{tEscape, "\\"}) + if l.peekTokenType(-1) == tEscape { + l.tokens = append(l.tokens, token{tWord, "\\"}) + } else { + l.tokens = append(l.tokens, token{tEscape, "\\"}) + } l.pos++ case _currentSymbol == ':': l.tokens = append(l.tokens, token{tColon, ":"}) @@ -93,17 +102,10 @@ func (l *lexer) Exec() []token { case _currentSymbol == ')': l.tokens = append(l.tokens, token{tRParen, ")"}) l.pos++ - case _currentSymbol == ',': - l.tokens = append(l.tokens, token{tWord, ","}) - l.pos++ default: { // NOTE: inlined check until good reason to abstract arrives. _start := l.pos - var _symbols []string - _symbols = append(_symbols, spacingSymbols...) - _symbols = append(_symbols, operatorSymbols...) - for l.pos < len(l.input) && !slices.Contains(_symbols, string(l.input[l.pos])) { l.pos += 1 } @@ -111,7 +113,7 @@ func (l *lexer) Exec() []token { _tokenStr := string(l.input[_start:l.pos]) if slices.Contains(keywords, _tokenStr) { - if _start == 0 || l.tokens[len(l.tokens)-1].tType != tEscape { + if _start == 0 || l.peekTokenType(-1) != tEscape { l.tokens = append(l.tokens, token{tKeyword, _tokenStr}) } else { l.tokens = append(l.tokens, token{tWord, _tokenStr}) @@ -126,3 +128,15 @@ func (l *lexer) Exec() []token { return l.tokens } + +func (l *lexer) peekTokenType(posOffset int) tokenType { + + _index := len(l.tokens) + posOffset + + if posOffset < 0 && _index > -1 { + return l.tokens[_index].tType + } + + log.Printf("_ERROR:peekTokenType %t, %t at %d", l.pos+posOffset > -1, l.pos+posOffset < len(l.tokens), l.pos+posOffset) + return tError +} diff --git a/src/fmd/parser.go b/src/fmd/parser.go @@ -10,9 +10,10 @@ type nodeType string const ( nRoot nodeType = "root" - nCommand nodeType = "command" - nBlock nodeType = "block" - nBody nodeType = "body" + nCommand nodeType = "command" + nParameter nodeType = "parameter" + nBlock nodeType = "block" + nBody nodeType = "body" nWord nodeType = "word" nOperator nodeType = "operator" @@ -20,34 +21,37 @@ const ( nNewline nodeType = "newline" nError nodeType = "error" + nEmpty nodeType = "nil" ) type node struct { nType nodeType value string - pNode *node + nParent *node children []node } type parser struct { - tokens []token - pos int - depth int - pNode *node - errorNodes []node + tokens []token // input tokens + nErrors []node // output error nodes + + pos int // current 'head' position + depth int // nest depth + nPrev *node // previous node reference } func MakeParser(tokens []token) *parser { if tokens == nil { log.Printf("_DEBUG:MakeParser:token array is empty") - _tokens := []token{{tType: tError, Value: "_ERROR:MakeParser:empty token string"}} + _tokens := []token{{tType: tError, value: "_ERROR:MakeParser:empty token string"}} return &parser{tokens: _tokens} } - return &parser{tokens: tokens, pos: 0} + + return &parser{tokens: tokens} } func (p *parser) Exec() node { - _fmd := p.makeNode(nRoot, token{tType: tEmpty, Value: "_fmd_"}) + _fmd := p.makeNode(nRoot, token{tType: tEmpty, value: "_fmd-root_"}) for p.pos < len(p.tokens) { if p.isNextCommand() { @@ -64,27 +68,42 @@ func (p *parser) Exec() node { func (p *parser) parseCommand() node { - // TODO: Added function for parseing multiple keywords; parseKeyword() (e.g. bi@:... ) _command := p.makeNode(nCommand, p.consume(tKeyword)) p.consume(tColon) if p.isNextCommand() { - _command.children = []node{p.parseCommand()} + _command.children = append(_command.children, p.parseCommand()) + } else if p.isTokenType(tLParen, 0) { + _command.children = append(_command.children, p.parseParam()) } else if p.isTokenType(tLBrace, 0) { - _command.children = []node{p.parseBlock()} + _command.children = append(_command.nParent.children, p.parseBlock()) } else { - _command.children = []node{p.makeNode(nWord, p.consume(tWord))} + _command.children = append(_command.children, p.makeNode(nWord, p.consume(tWord))) } return _command } +func (p *parser) parseParam() node { + var _n node + + p.consume(tLParen) + + for p.pos < len(p.tokens) && p.peekToken(0).tType != tRParen { + + } + + p.consume(tRParen) + + return _n +} + func (p *parser) parseBlock() node { p.depth += 1 - _block := p.makeNode(nBlock, token{tType: tEmpty, Value: ""}) + _block := p.makeNode(nBlock, token{tType: tEmpty, value: ""}) p.consume(tLBrace) @@ -106,7 +125,7 @@ func (p *parser) parseBlock() node { } func (p *parser) parseBody() node { - _body := p.makeNode(nBody, token{tType: tEmpty, Value: ""}) + _body := p.makeNode(nBody, token{tType: tEmpty, value: ""}) for p.pos < len(p.tokens) { @@ -141,7 +160,7 @@ func (p *parser) parseBody() node { if p.isNextCommand() { _body.children = append(_body.children, p.parseCommand()) } else { - _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, Value: "unknown token;"})) + _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, value: "unknown token;"})) p.pos += 1 log.Printf("INFO:block:error:%v", _body.children[len(_body.children)-1]) } @@ -154,24 +173,24 @@ func (p *parser) parseBody() node { //// UTILITY FUNCTIONS func (p *parser) makeNode(nType nodeType, iToken token) node { - _n := node{nType: nType, value: iToken.Value} + _n := node{nType: nType, value: iToken.value} if iToken.tType == tError { - _n = p.errorNode(iToken.Value) + _n = p.errorNode(iToken.value) } - _n.pNode = p.pNode - p.pNode = &_n + _n.nParent = p.nPrev + p.nPrev = &_n return _n } func (p *parser) errorNode(message string) node { - _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, message) - p.errorNodes = append(p.errorNodes, node{nType: nError, value: _errorValue}) + _errorValue := fmt.Sprintf("_ERROR:errorNode:on %s at pos %d:%s", p.tokens[p.pos].value, p.pos, message) + p.nErrors = append(p.nErrors, node{nType: nError, value: _errorValue}) - return p.errorNodes[len(p.errorNodes)-1] + return p.nErrors[len(p.nErrors)-1] } func (p *parser) consume(expectedTType tokenType) token { @@ -179,35 +198,37 @@ func (p *parser) consume(expectedTType tokenType) token { _token := p.tokens[p.pos] if !p.isTokenType(expectedTType, 0) { - _errorValue := fmt.Sprintf("_ERROR:on %s at pos %d:%s", p.tokens[p.pos].Value, p.pos, "invalid token type") + _errorValue := fmt.Sprintf("_ERROR:consume:on %s at pos %d:expected type %d got %d", p.tokens[p.pos].value, p.pos, expectedTType, _token.tType) _token = token{tError, _errorValue} } p.pos += 1 - // log.Printf("INFO:consume:%v", _token.Value) + log.Printf("INFO:consume %d,%s", _token.tType, _token.value) return _token } -func (p *parser) peekToken(numOfTokensAhead int) token { +func (p *parser) peekToken(posOffset int) token { - if p.pos+numOfTokensAhead < len(p.tokens) { + if p.pos+posOffset < len(p.tokens) { return p.tokens[p.pos] } return token{tError, "_ERROR:peekToken:attempting to peek beyond array size"} } -func (p *parser) isTokenType(expectedTType tokenType, indexOffset int) bool { - return p.pos+indexOffset > -1 && - p.pos+indexOffset < len(p.tokens) && - p.tokens[p.pos+indexOffset].tType == expectedTType +func (p *parser) isTokenType(expectedTType tokenType, posOffset int) bool { + return p.pos+posOffset > -1 && + p.pos+posOffset < len(p.tokens) && + p.tokens[p.pos+posOffset].tType == expectedTType } func (p *parser) isNextCommand() bool { - return !p.isTokenType(tEscape, -1) && - p.isTokenType(tKeyword, 0) && - p.isTokenType(tColon, 1) && - p.isTokenType(tWord, 2) || - p.isTokenType(tLBrace, 2) + if p.isTokenType(tEscape, 0) { + p.consume(tEscape) + return false + } + + return p.isTokenType(tKeyword, 0) && p.isTokenType(tColon, 1) + }