slweb

Једноставни генератор статичких веб страна
Дневник | Датотеке | Референце | ПРОЧИТАЈМЕ | ЛИЦЕНЦА

чување 2ac06532ee52ee9986d19addbb4d9d96491e50e0
родитељ 0bfc85afbb6942543638118c36f25c0f36d7a919
Аутор: Страхиња Радић <contact@strahinja.org>
Датум:   Fri, 11 Dec 2020 16:14:16 +0100

Continued WIP migration to C

Signed-off-by: Страхиња Радић <contact@strahinja.org>

Diffstat:
MREADME | 30++----------------------------
Mdefs.h | 19+++++++++++++++++++
Mindex.html | 72+++++++++++++++++++++++++++++++++---------------------------------------
Mindex.slw | 3++-
Mslweb.c | 358+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----
измењених датотека: 5, додавања: 399(+), брисања: 83(-)

diff --git a/README b/README @@ -1,35 +1,9 @@ - **IMPORTANT NOTE** - ================== - -Version v0.2.8-beta is the last version of slweb using sed. I decided to switch -to C because of the limitation of sed(1) when used to parse the entire file as a -single string (the -z option), as listed in - - https://www.gnu.org/software/sed/manual/sed.html#Limitations - - > For those who want to write portable sed scripts, be aware that some - > implementations have been known to limit line lengths (for the pattern - > and hold spaces) to be no more than 4000 bytes. The POSIX standard - > specifies that conforming sed implementations shall support at least - > 8192 byte line lengths. GNU sed has no built-in limit on line length; - > as long as it can malloc() more (virtual) memory, you can feed or - > construct lines as long as you like. - -When processed through slweb, some of the test pages using the cyrillic text in -UTF-8 would be rendered incorrectly, even when using GNU sed. I suspect this is -the result of the mentioned limitation. - - The next version of slweb is going to be written in C from scratch and -labeled v0.3.0. - - -- Strahinya - slweb ===== -Slweb is a static website generator which aims at being simplistic. It uses -sed(1) to transform a custom Markdown-like syntax into HTML. +Slweb is a static website generator which aims at being simplistic. It +transforms custom Markdown-like syntax into HTML. Install diff --git a/defs.h b/defs.h @@ -40,6 +40,9 @@ typedef enum TRUE = 1 } BOOL; +typedef unsigned char UBYTE; +typedef unsigned short USHORT; + typedef enum { CMD_NONE, @@ -49,5 +52,21 @@ typedef enum CMD_VERSION } Command; +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Wunused-const-variable" +const UBYTE MAX_HEADING_LEVEL = 4; + +const USHORT ST_NONE = 0; +const USHORT ST_YAML = 1; +const USHORT ST_YAML_VAR = 1 << 1; +const USHORT ST_YAML_VAL = 1 << 2; +const USHORT ST_PARA_OPEN = 1 << 3; +const USHORT ST_TAG = 1 << 4; +const USHORT ST_HEADING = 1 << 5; +const USHORT ST_PRE = 1 << 6; +const USHORT ST_CODE = 1 << 7; +const USHORT ST_BLOCKQUOTE = 1 << 8; +#pragma GCC diagnostic pop + #endif /* __DEFS_H */ diff --git a/index.html b/index.html @@ -1,28 +1,23 @@ -site-name: slweb - Simple static website generator. -site-desc: Simple static website generator -stylesheet: index.css ---- -{main} -# slweb +<main> -**Slweb** is a static website generator which aims at being simplistic. It uses -**sed**(1) to transform a custom Markdown-like syntax into HTML. +<h1>slweb</h1> -## Install +<p>**Slweb** is a static website generator which aims at being simplistic. It uses +**sed**(1) to transform a custom Markdown-like syntax into HTML.</p> -``` +<h2>Install</h2> +<pre> $ git clone https://github.com/Strahinja/slweb.git $ cd slweb $ make && sudo make install -``` - -## Examples +</pre> -See the `[examples/](https://github.com/Strahinja/slweb/tree/master/examples)` directory in this repository. +<h2>Examples</h2> -Given the file `index.slw` in the current directory: +<p>See the [<code>examples/</code>](https://github.com/Strahinja/slweb/tree/master/examples) directory in this repository.</p> -``` +<p>Given the file <code>index.slw</code> in the current directory:</p> +<pre> site-name: Test website site-desc: My first website in slweb --- @@ -33,17 +28,15 @@ site-desc: My first website in slweb This is an \_example_ of a statically generated HTML. \ \{/main} -``` - -after using the command: +</pre> -``` +<p>after using the command:</p> +<pre> $ slweb index.slw > index.html -``` +</pre> -file `index.html` contains: - -``` +<p>file <code>index.html</code> contains:</p> +<pre> &lt;!DOCTYPE html&gt; &lt;html lang="en"&gt; &lt;head&gt; @@ -63,27 +56,28 @@ file `index.html` contains: &lt;/main&gt; &lt;/body&gt; &lt;/html&gt; -``` +</pre> -## License +<h2>License</h2> -slweb - Simple static website generator. -Copyright (C) 2020 Страхиња Радић +<p>slweb - Simple static website generator.<br /> +Copyright (C) 2020 Страхиња Радић</p> -This program is free software: you can redistribute it and/or modify it under +<p>This program is free software: you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free Software Foundation, either version 3 of the License, or (at your option) any later -version. +version. </p> -This program is distributed in the hope that it will be useful, but WITHOUT ANY +<p>This program is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A -PARTICULAR PURPOSE. See the GNU General Public License for more details. - -You should have received a copy of the GNU General Public License along with -this program. If not, see &lt;https://www.gnu.org/licenses/&gt;. - -{git-log} +PARTICULAR PURPOSE. See the GNU General Public License for more details. </p> -{made-by} -{/main} +<p>You should have received a copy of the GNU General Public License along with +this program. If not, see &lt;https://www.gnu.org/licenses/&gt;. </p> +<git-log> +<div id="made-by"> +Generated by <a href="https://github.com/Strahinja/slweb">slweb</a> +© 2020 Strahinya Radich. +</div><!--made-by--> +</main> diff --git a/index.slw b/index.slw @@ -1,3 +1,4 @@ +--- site-name: slweb - Simple static website generator. site-desc: Simple static website generator stylesheet: index.css @@ -18,7 +19,7 @@ $ make && sudo make install ## Examples -See the `[examples/](https://github.com/Strahinja/slweb/tree/master/examples)` directory in this repository. +See the [`examples/`](https://github.com/Strahinja/slweb/tree/master/examples) directory in this repository. Given the file `index.slw` in the current directory: diff --git a/slweb.c b/slweb.c @@ -18,7 +18,13 @@ */ #include "defs.h" +#include <asm-generic/errno-base.h> +#include <stdio.h> #include <stdlib.h> +#include <unistr.h> + +static size_t lineno = 0; +static size_t colno = 1; int version() @@ -47,6 +53,18 @@ error(int code, uint8_t* fmt, ...) return code; } +int +warning(int code, uint8_t* fmt, ...) +{ + uint8_t buf[BUFSIZE]; + va_list args; + va_start(args, fmt); + u8_vsnprintf(buf, sizeof(buf), (const char*)fmt, args); + va_end(args); + fprintf(stderr, "Warning: %s", buf); + return code; +} + char* substr(char* src, int start, int finish) { @@ -57,6 +75,8 @@ substr(char* src, int start, int finish) if (substr_len < 0) substr_len = 0; char* result = (char*) calloc(substr_len+1, sizeof(char)); + if (!result) + exit(error(ENOMEM, (uint8_t*)"Memory allocation failed (out of memory?)\n")); char* presult = result; for (int i = start; i < finish && *(src+i) != '\0'; i++) @@ -82,13 +102,70 @@ set_basedir(char* arg, char** basedir) *basedir = (char*) calloc(strlen(arg)+1, sizeof(char)); if (!*basedir) - return 1; + return error(ENOMEM, (uint8_t*)"Memory allocation failed (out of memory?)\n"); strcpy(*basedir, arg); return 0; } +uint8_t* +init_string(uint8_t** str) +{ + if (!str || !*str) + return NULL; + + *str[0] = '\0'; + + return *str; +} + +int +finish_and_print_token(uint8_t** token, uint8_t** ptoken, FILE* output) +{ + if (!token || !*token || !ptoken || !*ptoken) + return 1; + + *ptoken[0] = '\0'; + fprintf(output, "%s", *token); +} + +int +process_heading(uint8_t* token, FILE* output, UBYTE heading_level) +{ + if (!token || u8_strlen(token) < 1) + warning(1, (uint8_t*)"Empty heading\n"); + + fprintf(output, "<h%d>%s</h%d>", + heading_level, + token ? (char*)token : "", + heading_level); + + return 0; +} + +int +process_tag(uint8_t* token, FILE* output, BOOL end_tag) +{ + if (!token || u8_strlen(token) < 1) + return warning(1, (uint8_t*)"No tag\n"); + + if (!strcmp(token, "made-by")) + { + fprintf(output, "<div id=\"made-by\">\n" + "Generated by <a href=\"https://github.com/Strahinja/slweb\">" + "slweb</a>\n" + "© 2020 Strahinya Radich.\n" + "</div><!--made-by-->\n"); + } + else + fprintf(output, "<%s%s>", + end_tag ? (const char*)"/" : (const char*)"", + token); + + return 0; +} + int main(int argc, char** argv) { @@ -99,6 +176,8 @@ main(int argc, char** argv) char* basedir = NULL; basedir = (char*) calloc(2, sizeof(char)); + if (!basedir) + return error(ENOMEM, (uint8_t*)"Memory allocation failed (out of memory?)\n"); basedir[0] = '.'; while ((arg = *++argv)) @@ -133,7 +212,7 @@ main(int argc, char** argv) } else { - error(1, (uint8_t*)"Invalid argument: --%s\n", arg); + error(EINVAL, (uint8_t*)"Invalid argument: --%s\n", arg); return usage(); } } @@ -154,7 +233,7 @@ main(int argc, char** argv) cmd = CMD_VERSION; break; default: - error(1, (uint8_t*)"Invalid argument: -%c\n", c); + error(EINVAL, (uint8_t*)"Invalid argument: -%c\n", c); return usage(); } } @@ -183,42 +262,291 @@ main(int argc, char** argv) if (cmd == CMD_VERSION) return version(); - /* - *printf("debug: basedir=%s, filename=%s, body_only=%s\n", - * basedir, filename ? filename : "(NULL)", body_only ? "TRUE" : "FALSE"); - */ - FILE* input = NULL; if (filename) { input = fopen(filename, "r"); if (!input) { - return error(ENOENT, (uint8_t*)"File not found: %s\n", filename); + return error(ENOENT, (uint8_t*)"No such file: %s\n", filename); } } else input = stdin; uint8_t* line = (uint8_t*) calloc(BUFSIZE, sizeof(uint8_t)); + uint8_t* pline = NULL; size_t line_len = 0; + uint8_t* token = (uint8_t*) calloc(BUFSIZE, sizeof(uint8_t)); + uint8_t* ptoken = NULL; + size_t token_len = 0; + USHORT state = ST_NONE; + UBYTE heading_level = 0; + BOOL end_tag = FALSE; + BOOL first_line_in_doc = TRUE; + BOOL previous_line_blank = FALSE; + BOOL print_newline = FALSE; + + if (!line) + return error(ENOMEM, (uint8_t*)"Memory allocation failed (out of memory?)\n"); do { - fgets((char*)line, sizeof(uint8_t) * BUFSIZE, input); + if (!fgets((char*)line, sizeof(char) * BUFSIZE, input)) + continue; + + uint8_t* eol = u8_strchr(line, (ucs4_t)'\n'); + if (eol) + *eol = (uint8_t)'\0'; + + lineno++; + colno = 1; + + /* + *fprintf(stderr, "=D: s:0x%X, hl:%d, et:%s, flid:%s, plb:%s \n", + * state, + * heading_level, + * end_tag ? "T" : "F", + * first_line_in_doc ? "T" : "F", + * previous_line_blank ? "T" : "F"); + */ + line_len = u8_strlen(line); + pline = line; + ptoken = init_string(&token); + + do + { + switch (*pline) + { + case '-': + if (colno == 1 + && !(state & ST_PRE) + && u8_strlen(pline) > 2 + && !strcmp(substr((char*)pline, 0, 3), "---")) + { + state ^= ST_YAML; + + if (state & ST_YAML) + first_line_in_doc = FALSE; + else + first_line_in_doc = TRUE; + + pline += 3; + colno += 3; + } + else + { + *ptoken++ = *pline++; + colno++; + } + break; + case '`': + if (colno == 1 + && u8_strlen(pline) > 2 + && !strcmp(substr((char*)pline, 0, 3), "```")) + { + state ^= ST_PRE; + + if (state & ST_PRE) + printf("<pre>"); + else + printf("</pre>"); + print_newline = TRUE; - if (!feof(input)) + pline += 3; + colno += 3; + } + else if (!(state & (ST_HEADING | ST_YAML | ST_PRE))) + { + state ^= ST_CODE; + + if (state & ST_CODE) + { + *ptoken = '\0'; + u8_strncat(ptoken, (uint8_t*)"<code>", strlen("<code>")); + ptoken += strlen("<code>"); + } + else + { + *ptoken = '\0'; + u8_strncat(ptoken, (uint8_t*)"</code>", strlen("</code>")); + ptoken += strlen("</code>"); + } + + pline++; + colno++; + } + else + { + *ptoken++ = *pline++; + colno++; + } + break; + case '#': + if (colno == 1 + && !(state & (ST_PRE | ST_YAML))) + { + state |= ST_HEADING; + heading_level = 1; + pline++; + colno++; + } + else if (state & ST_HEADING && *(pline-1) == '#') + { + if (heading_level < MAX_HEADING_LEVEL) + heading_level++; + pline++; + colno++; + } + else { + *ptoken++ = *pline++; + colno++; + } + break; + case ' ': + if (u8_strlen(pline) == 2) + { + if (*(pline+1) == ' ') + { + *ptoken = '\0'; + u8_strncat(ptoken, (uint8_t*)"<br />", strlen("<br />")); + ptoken += strlen("<br />"); + pline++; + colno++; + } + else + *ptoken++ = *pline; + } + else if (!(state & ST_HEADING + && *(pline-1) == '#')) + *ptoken++ = *pline; + pline++; + colno++; + break; + case '{': + if (state & (ST_PRE | ST_YAML | ST_HEADING)) + { + *ptoken++ = *pline++; + colno++; + break; + } + + state |= ST_TAG; + if (token[0]) + finish_and_print_token(&token, &ptoken, stdout); + ptoken = init_string(&token); + pline++; + colno++; + break; + + case '/': + if (state & (ST_PRE | ST_YAML | ST_HEADING)) + { + *ptoken++ = *pline++; + colno++; + break; + } + + if (state & ST_TAG) + { + if (*(pline-1) != '{') + return error(1, (uint8_t*)"Character '/' not allowed here"); + + end_tag = TRUE; + } + else + *ptoken++ = *pline; + + pline++; + colno++; + break; + + case '}': + if (state & (ST_PRE | ST_YAML | ST_HEADING)) + { + *ptoken++ = *pline++; + colno++; + break; + } + + state &= ~ST_TAG; + *ptoken = '\0'; + process_tag(token, stdout, end_tag); + first_line_in_doc = FALSE; + print_newline = TRUE; + ptoken = init_string(&token); + end_tag = FALSE; + + pline++; + colno++; + break; + + default: + *ptoken++ = *pline++; + colno++; + } + } + while (pline && *pline); + + if (!(state & ST_YAML)) { - uint8_t* eol = u8_strchr(line, (ucs4_t)'\n'); - if (eol) - *eol = (uint8_t)'\0'; + if (token[0]) + { + if (colno != 2 && !(state & ST_PRE)) + printf("\n"); + + if (state & ST_HEADING) + { + state &= ~ST_HEADING; + *ptoken = '\0'; + process_heading(token, stdout, heading_level); + first_line_in_doc = FALSE; + ptoken = init_string(&token); + heading_level = 0; + print_newline = TRUE; + } + else { + + if (first_line_in_doc || previous_line_blank) + { + printf("<p>"); + first_line_in_doc = FALSE; + state |= ST_PARA_OPEN; + } + finish_and_print_token(&token, &ptoken, stdout); + print_newline = state & (ST_PRE | ST_CODE); + } + previous_line_blank = FALSE; + } + else if (colno == 2) + { + if (!previous_line_blank && (state & ST_PARA_OPEN)) + { + printf("</p>\n"); + print_newline = FALSE; + state &= ~ST_PARA_OPEN; + } + previous_line_blank = TRUE; + } + else + previous_line_blank = FALSE; + + if (print_newline) + { + printf("\n"); + print_newline = FALSE; + } - printf("%s\n", line); } + + ptoken = init_string(&token); + } while (!feof(input)); + free(line); + return 0; }