commit c7dd8655b9e6a17c63ba4f49867d47ff6c034129
parent d7a479f32ada12f0aa1a61bb6e4e94b3dc6d59de
Author: weetat <wtenison@protonmail.com>
Date: Mon, 3 Mar 2025 17:59:13 +0000
Moved 'fmd' to own project/repo.
Diffstat:
6 files changed, 0 insertions(+), 585 deletions(-)
diff --git a/src/fmd/fmd.ebnf b/src/fmd/fmd.ebnf
@@ -1,23 +0,0 @@
-# This grammar is a work-in-progress and represents the ideal form
-# of the lanuage. It may not be fully implemented and is subject
-# to change as I learn.
-
-<fmd> ::= (<command> | <block> | <body>)*
-
-<command> ::= (<command_id>)+ ":" <word_or_block>
-<command_id> ::= <keyword> | <keyword> "(" <params> ")"
-<keyword> ::= "b" | "i" | "h" [1-6] | "@" | "#" | "col" | "img" | "_"
-<params> ::= ("id=" | "class=" | "ref=") <word_or_block>
-
-<word_or_block> ::= <word> | <block>
-
-<block> ::= "{" <body> "}"
-<body> := (<word> | <space>)+
-
-<word> ::= (<letter> | <number> | <symbol>)+
-
-<space> ::= " "
-<letter> ::= [a-z] | [A-Z]
-<number> ::= [0-9]+ | (<number> "." <number>)
-
-<symbol> ::= "." | "," | ";" | "-" | "!" | "#" | "$" | "%" | "^" | "&" | "*" | "(" | ")"
-\ No newline at end of file
diff --git a/src/fmd/fmd.go b/src/fmd/fmd.go
@@ -1,65 +0,0 @@
-package fmd
-
-import (
- "bytes"
- "log"
- "os"
- "path/filepath"
-)
-
-func ReadFMDStreamtoHTMLString(inputStreamFMD []byte) string {
-
- _l := MakeLexer(inputStreamFMD)
-
- _tokens := _l.Exec()
-
- _p := MakeParser(_tokens)
-
- _ast := _p.Exec()
-
- return interpret(_ast)
-}
-
-func WriteFMDToHTMLFile(inputFilePath string, outputFilePath string) {
- _fileName := filepath.Base(inputFilePath)
- if filepath.Ext(_fileName) != ".fmd" {
- log.Printf("WriteFMDToHTMLFile:filepath.Ext:%s:invalid file extension", _fileName)
- return
- }
-
- _fileBytes, _err := os.ReadFile(inputFilePath)
- if _err != nil {
- log.Println("_ERROR\tfmd:WriteFMDToHTMLFile:ReadFile:", inputFilePath, ":", _err)
- }
-
- _l := MakeLexer(_fileBytes)
-
- _tokens := _l.Exec()
-
- _p := MakeParser(_tokens)
-
- _ast := _p.Exec()
-
- _outputBytes := bytes.NewBufferString(interpret(_ast))
-
- if _err = os.WriteFile(outputFilePath, _outputBytes.Bytes(), 0666); _err != nil {
- log.Println("_ERROR\tfmd:WriteFMDToHTMLFile:os.WriteFile:", inputFilePath, ":", _err)
- }
-}
-
-func WritFMDToHTMLString(inputFilePath string) string {
- _fileBytes, _err := os.ReadFile(inputFilePath)
- if _err != nil {
- log.Println("_ERROR\tfmd:WriteFMDToHTMLString:ReadFile:", inputFilePath, ":", _err)
- }
-
- _l := MakeLexer(_fileBytes)
-
- _tokens := _l.Exec()
-
- _p := MakeParser(_tokens)
-
- _ast := _p.Exec()
-
- return interpret(_ast)
-}
diff --git a/src/fmd/fmd_test.go b/src/fmd/fmd_test.go
@@ -1,31 +0,0 @@
-package fmd
-
-import (
- "os"
- "testing"
-)
-
-func TestFMD(t *testing.T) {
- _path := ("/home/weetat/projects/miki/data/README.fmd")
- t.Log("Path:", _path)
-
- // WriteFMDToHTMLFile(_path, "/home/weetat/projects/miki/README.md")
-
- _input, _err := os.ReadFile(_path)
- if _err != nil {
- t.Error("_ERROR\tcannot open file:", _err.Error())
- return
- }
-
- _l := MakeLexer(_input)
- _tokens := _l.Exec()
-
- // for _, _t := range _tokens {
- // t.Log("Tokens:", _t)
- // }
-
- _p := MakeParser(_tokens)
- _ast := _p.Exec()
-
- t.Log("HTML:\n", interpret(_ast))
-}
diff --git a/src/fmd/interpreter.go b/src/fmd/interpreter.go
@@ -1,84 +0,0 @@
-package fmd
-
-import (
- "slices"
- "strings"
-)
-
-// TODO: Implment better solution to sanitization.
-// TODO: ALternative; move interpreter to client side via Web Assembly.
-// NOTE: Solution; move sanitization to separate package and let the interpreter work fast, then clean?
-var invalidHTML = []string{"<script>", "</script>", "<object>", "</object>", "<embed", "<link>", "</link>"}
-
-func interpret(n node) string {
- var _res strings.Builder
-
- switch n.nType {
- case nRoot:
- for _, _child := range n.children {
- _res.WriteString(interpret(_child))
- }
- case nCommand:
- switch n.value {
- case "b":
- return "<b>" + interpret(n.children[0]) + "</b>"
- case "i":
- return "<i>" + interpret(n.children[0]) + "</i>"
- case "h1":
- return "<h1>" + interpret(n.children[0]) + "</h1>"
- case "h2":
- return "<h2>" + interpret(n.children[0]) + "</h2>"
- case "h3":
- return "<h3>" + interpret(n.children[0]) + "</h3>"
- case "h4":
- return "<h4>" + interpret(n.children[0]) + "</h4>"
- case "h5":
- return "<h5>" + interpret(n.children[0]) + "</h5>"
- case "h6":
- return "<h6>" + interpret(n.children[0]) + "</h6>"
- case "code":
- return "<code>" + interpret(n.children[0]) + "</code>"
- case "@":
- return "<a>" + interpret(n.children[0]) + "</a>"
- case "img":
- {
- _child := interpret(n.children[0])
- _src := strings.Split(_child, "/")
- _alt := _src[len(_src)-1]
- return "<img src=\"" + _child + "\" alt=\"" + _alt + "\">" + "</img>"
- }
- case "col":
- return "<div class=\"__mcol\">" + interpret(n.children[0]) + "</div>"
- default:
- return "<div>" + interpret(n.children[0]) + "</div>"
- }
- case nBlock:
- _res.WriteString("<div class=\"__b\">")
- for _, _child := range n.children {
- _res.WriteString(interpret(_child))
- }
- _res.WriteString("</div>")
- case nBody:
- for _, _child := range n.children {
- _res.WriteString(interpret(_child))
- }
- case nSpacing:
- return n.value
- case nNewline:
- if n.nPrev.nType != nNewline {
- return "<br>\n"
- } else {
- return "\n"
- }
- case nWord:
- if slices.Contains(invalidHTML, n.value) {
- return ""
- }
-
- return n.value
- default:
- return n.value
- }
-
- return _res.String()
-}
diff --git a/src/fmd/lexer.go b/src/fmd/lexer.go
@@ -1,146 +0,0 @@
-package fmd
-
-import (
- "slices"
-)
-
-/* Reference source: Lexical Scanning in Go - Rob Pike,
-https://www.youtube.com/watch?v=HxaD_trXwRE */
-
-type tokenType int
-
-const (
-
- // WORDS
- tWord tokenType = iota // 0. (letters, numbers, symbols)+
- tKeyword // 1. (valid keywords)
-
- // SPACING
- tSpace // 2. " "
- tNewline // 3. \n
- tTab // 4. \t
-
- // OPERATORS
- tEscape // 5. \
- tColon // 6. :
- tLBrace // 7. {
- tRBrace // 8. }
- tLParen // 9. (
- tRParen // 10. )
-
- // UTILS
- tError // 11. error
- tEmpty // 12. nil
-)
-
-type token struct {
- tType tokenType
- value string
-}
-
-var operatorSymbols = []string{"\\", ":", "{", "}", "(", ")"}
-var spacingSymbols = []string{" ", "\n", "\t"}
-var keywords = []string{
- "b", // bold
- "i", // italic
- "h1", "h2", "h3", "h4", "h5", "h6", // header
- "code", // code element
- "@", // link
- "#", // tag
- "col", // column container
- "img", // image
- "_", // empty container (<div>)
-}
-
-type lexer struct {
- input []byte // text input to scan
- pos int // the start position for scanning
- tokens []token // the output token stream
-}
-
-func MakeLexer(input []byte) *lexer {
- return &lexer{input: input, pos: 0, tokens: []token{}}
-}
-
-func (l *lexer) Exec() []token {
- var _symbols []string
- _symbols = append(_symbols, spacingSymbols...)
- _symbols = append(_symbols, operatorSymbols...)
-
- for l.pos < len(l.input) {
- _currentSymbol := l.input[l.pos]
-
- switch {
- case _currentSymbol == ' ':
- l.tokens = append(l.tokens, token{tSpace, " "})
- l.pos++
- case _currentSymbol == '\n':
- l.tokens = append(l.tokens, token{tNewline, "\n"})
- l.pos++
- case _currentSymbol == '\t':
- l.tokens = append(l.tokens, token{tTab, "\t"})
- l.pos++
- case _currentSymbol == '\\':
- if l.pos != 0 && l.peekTokenType(-1) == tEscape {
- l.tokens = append(l.tokens, token{tWord, "\\"})
- } else {
- l.tokens = append(l.tokens, token{tEscape, "\\"})
- }
- l.pos++
- case _currentSymbol == ':':
- l.tokens = append(l.tokens, token{tColon, ":"})
- l.pos++
- case _currentSymbol == '{':
- l.tokens = append(l.tokens, token{tLBrace, "{"})
- l.pos++
- case _currentSymbol == '}':
- l.tokens = append(l.tokens, token{tRBrace, "}"})
- l.pos++
- case _currentSymbol == '(':
- l.tokens = append(l.tokens, token{tLParen, "("})
- l.pos++
- case _currentSymbol == ')':
- l.tokens = append(l.tokens, token{tRParen, ")"})
- l.pos++
- default:
- { // NOTE: inlined check until good reason to abstract arrives.
- _start := l.pos
-
- for l.pos < len(l.input) && !slices.Contains(_symbols, string(l.input[l.pos])) {
- l.pos += 1
- }
-
- _tokenStr := string(l.input[_start:l.pos])
-
- if slices.Contains(keywords, _tokenStr) && l.peekInput(1) == ':' {
- if _start == 0 || l.peekTokenType(-1) != tEscape {
- l.tokens = append(l.tokens, token{tKeyword, _tokenStr})
- } else {
- l.tokens = append(l.tokens, token{tWord, _tokenStr})
- l.tokens = append(l.tokens, token{tWord, ":"})
- l.pos += 1
- }
- } else if _tokenStr != "" {
- l.tokens = append(l.tokens, token{tWord, _tokenStr})
- }
- }
- }
- }
-
- return l.tokens
-}
-
-func (l *lexer) peekInput(posOffset int) byte {
- if l.pos+posOffset > -1 && l.pos+posOffset < len(l.input) {
- return l.input[l.pos]
- }
- panic("__PANIC:peekInput:check out of bounds")
-}
-
-func (l *lexer) peekTokenType(posOffset int) tokenType {
- _index := len(l.tokens) + posOffset
- if posOffset < 0 && _index > -1 {
- return l.tokens[_index].tType
- }
- panic("__PANIC:peekTokenType:check out of bounds")
-}
diff --git a/src/fmd/parser.go b/src/fmd/parser.go
@@ -1,235 +0,0 @@
-package fmd
-
-import (
- "fmt"
- "log"
-)
-
-type nodeType string
-
-const (
- nRoot nodeType = "root"
-
- nCommand nodeType = "command"
- nParameter nodeType = "parameter"
- nBlock nodeType = "block"
- nBody nodeType = "body"
-
- nWord nodeType = "word"
- nOperator nodeType = "operator"
- nSpacing nodeType = "spacing"
- nNewline nodeType = "newline"
-
- nError nodeType = "error"
- nEmpty nodeType = "nil"
-)
-
-type node struct {
- nType nodeType
- value string
- nPrev *node
- children []node
-}
-
-type parser struct {
- tokens []token // input tokens
- nErrors []node // output error nodes
-
- pos int // current 'head' position
- depth int // nest depth
- nPrev *node // previous node reference
-}
-
-func MakeParser(tokens []token) *parser {
- if tokens == nil {
- log.Printf("_DEBUG:MakeParser:token array is empty")
- _tokens := []token{{tType: tError, value: "_ERROR:MakeParser:empty token string"}}
- return &parser{tokens: _tokens}
- }
-
- return &parser{tokens: tokens}
-}
-
-func (p *parser) Exec() node {
- _fmd := p.makeNode(nRoot, token{tType: tEmpty, value: "_fmd-root_"})
-
- for p.pos < len(p.tokens) {
- if p.isNextCommand() {
- _fmd.children = append(_fmd.children, p.parseCommand())
- } else if p.peekToken(0).tType == tLBrace {
- _fmd.children = append(_fmd.children, p.parseBlock())
- } else {
- _fmd.children = append(_fmd.children, p.parseBody())
- }
- }
-
- return _fmd
-}
-
-func (p *parser) parseCommand() node {
-
- _command := p.makeNode(nCommand, p.consume(tKeyword))
-
- p.consume(tColon)
-
- if p.isNextCommand() {
- _command.children = append(_command.children, p.parseCommand())
- } else if p.isTokenType(tLParen, 0) {
- _command.children = append(_command.children, p.parseParam())
- } else if p.isTokenType(tLBrace, 0) {
- _command.children = append(_command.nPrev.children, p.parseBlock())
- } else {
- _command.children = append(_command.children, p.makeNode(nWord, p.consume(tWord)))
- }
-
- return _command
-}
-
-func (p *parser) parseParam() node {
- var _n node
-
- p.consume(tLParen)
-
- // TODO: Implement parameter parsing.
- for p.pos < len(p.tokens) && p.peekToken(0).tType != tRParen {
- p.pos += 1
- }
-
- p.consume(tRParen)
-
- return _n
-}
-
-func (p *parser) parseBlock() node {
-
- p.depth += 1
-
- _block := p.makeNode(nBlock, token{tType: tEmpty, value: ""})
-
- p.consume(tLBrace)
-
- for p.pos < len(p.tokens) && p.peekToken(0).tType != tRBrace {
-
- switch {
- case p.isNextCommand():
- _block.children = append(_block.children, p.parseCommand())
- default:
- _block.children = append(_block.children, p.parseBody())
- }
- }
-
- p.depth -= 1
-
- p.consume(tRBrace)
-
- return _block
-}
-
-func (p *parser) parseBody() node {
- _body := p.makeNode(nBody, token{tType: tEmpty, value: ""})
-
- for p.pos < len(p.tokens) {
-
- if p.peekToken(0).tType == tRBrace && p.depth > 0 {
- break
- }
-
- _pToken := p.peekToken(0)
-
- switch _pToken.tType {
- case tWord:
- _body.children = append(_body.children, p.makeNode(nWord, p.consume(tWord)))
- case tSpace:
- _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tSpace)))
- case tTab:
- _body.children = append(_body.children, p.makeNode(nSpacing, p.consume(tTab)))
- case tNewline:
- _body.children = append(_body.children, p.makeNode(nNewline, p.consume(tNewline)))
- case tLParen:
- _body.children = append(_body.children, p.makeNode(nWord, p.consume(tLParen)))
- case tRParen:
- _body.children = append(_body.children, p.makeNode(nWord, p.consume(tRParen)))
- case tEscape:
- p.consume(tEscape)
- case tColon:
- p.consume(tColon)
- case tLBrace:
- _body.children = append(_body.children, p.parseBlock())
- case tRBrace:
- _body.children = append(_body.children, p.errorNode("no matching LBrace"))
- default:
- if p.isNextCommand() {
- _body.children = append(_body.children, p.parseCommand())
- } else {
- _body.children = append(_body.children, p.makeNode(nError, token{tType: tError, value: "unknown token;"}))
- p.pos += 1
- log.Printf("INFO:block:error:%v", _body.children[len(_body.children)-1])
- }
- }
- }
-
- return _body
-}
-
-//// UTILITY FUNCTIONS
-
-func (p *parser) makeNode(nType nodeType, iToken token) node {
- _n := node{nType: nType, value: iToken.value}
-
- if iToken.tType == tError {
- _n = p.errorNode(iToken.value)
- }
-
- _n.nPrev = p.nPrev
- p.nPrev = &_n
-
- return _n
-}
-
-func (p *parser) errorNode(message string) node {
-
- _errorValue := fmt.Sprintf("_ERROR:errorNode:on %s at pos %d:%s", p.tokens[p.pos].value, p.pos, message)
- p.nErrors = append(p.nErrors, node{nType: nError, value: _errorValue})
-
- return p.nErrors[len(p.nErrors)-1]
-}
-
-func (p *parser) consume(expectedTType tokenType) token {
-
- _token := p.tokens[p.pos]
-
- if !p.isTokenType(expectedTType, 0) {
- _errorValue := fmt.Sprintf("_ERROR:consume:on %s at pos %d:expected type %d got %d", p.tokens[p.pos].value, p.pos, expectedTType, _token.tType)
- _token = token{tError, _errorValue}
- }
-
- p.pos += 1
-
- log.Printf("INFO:consume %d,%s", _token.tType, _token.value)
- return _token
-}
-
-func (p *parser) peekToken(posOffset int) token {
-
- if p.pos+posOffset < len(p.tokens) {
- return p.tokens[p.pos]
- }
-
- return token{tError, "_ERROR:peekToken:attempting to peek beyond array size"}
-}
-
-func (p *parser) isTokenType(expectedTType tokenType, posOffset int) bool {
- return p.pos+posOffset > -1 &&
- p.pos+posOffset < len(p.tokens) &&
- p.tokens[p.pos+posOffset].tType == expectedTType
-}
-
-func (p *parser) isNextCommand() bool {
- if p.isTokenType(tEscape, 0) {
- p.consume(tEscape)
- return false
- }
-
- return p.isTokenType(tKeyword, 0) && p.isTokenType(tColon, 1)
-
-}