feat: Added deflist markdown extension.

This commit is contained in:
Spencer Brower
2026-08-04 12:53:01 -04:00
parent 0e8d770ac8
commit 876f767548
8 changed files with 317 additions and 22 deletions
+3 -1
View File
@@ -79,6 +79,7 @@ thor/
| | `sectionate.odin` | `wrap_sections` — splits HTML at `<h2>` into `<section>` wrappers |
| | `highlight.odin` | Syntax highlighting via tree-sitter. Imports `../treesitter`. |
| | `heading_ids.odin` | `inject_heading_ids` — adds `id` attributes to `<h1>`-`<h6>` from heading text. Slug-based, deduplicated. |
| | `deflists.odin` | `convert_deflists` — pre-cmark pass. Scans for definition list patterns (`term\n\n: definition`) and converts to `<dl><dt><dd>` HTML blocks. Terms and definitions rendered through cmark individually for inline markdown. Consecutive pairs grouped into single `<dl>`. |
| | `toc.odin` | `generate_toc(html, allocator)` — page-level feature (not a pipeline extension). Scans `<h1>`-`<h6>` for IDs (after `inject_heading_ids`), builds nested `<ul>` with `<a href="#id">` links. Called from `load_page` when frontmatter `"toc": true`. Depends on `.HeadingIDs` being enabled. |
| `mustache/` | See [Mustache engine](#mustache-engine) below | Template engine |
| `bench/` | `bench.odin` + `templates/` | Standalone template rendering benchmark. Generates 500 posts + 100 comments, renders with indented partials + inheritance + pipes. `--dump <path>` for output validation, positional arg for iteration count (default 250). |
@@ -177,7 +178,7 @@ Config is split into three structs with a clear 5-step initialization flow:
**`Feature` enum** — `Drafts`, `Minify`, `Watch`. Checked with `.Minify in site.features`.
**`markdown.Extension` enum** (in the `markdown` package, not main) — `Emoji`, `Sidenotes`, `Alerts`, `Highlight`, `Sections`, `HeadingIDs`. Default is `md.DEFAULT_EXTENSIONS` (currently `.Emoji, .Sidenotes, .Alerts, .HeadingIDs`). Configurable via:
**`markdown.Extension` enum** (in the `markdown` package, not main) — `Emoji`, `Sidenotes`, `Alerts`, `Highlight`, `Sections`, `HeadingIDs`, `DefLists`. Default is `md.DEFAULT_EXTENSIONS` (currently `.Emoji, .Sidenotes, .Alerts, .HeadingIDs, .DefLists`). Configurable via:
- `thor.json`: `"markdown_extensions": { "emoji": true, "highlight": false, ... }`
- CLI: `-ext:highlight,sections` (enable) / `-no-ext:emoji` (disable). Comma-separated, case-insensitive.
@@ -258,6 +259,7 @@ Lives in the `markdown` package. Entry point: `md.process(body, ext, file_path)`
```
raw markdown
→ md.strip_definitions (if .Sidenotes — pre-cmark)
→ md.convert_deflists (if .DefLists — pre-cmark)
→ cmark markdown_to_html (Unsafe mode for HTML passthrough)
→ md.expand_emoji (if .Emoji — post-cmark)
→ md.inject_notes (if .Sidenotes — post-cmark)
+1
View File
@@ -118,6 +118,7 @@
- Checks all links on each page to make sure they are valid.
- [ ] Peruse [GitHub's](https://docs.github.com/en/get-started/writing-on-github/getting-started-with-writing-and-formatting-on-github/basic-writing-and-formatting-syntax#alerts)
docs for any juicy nuggets we may have missed.
- [ ] Avoid using `render_inline_md` if possible.
## Dates
- [ ] display an error when no part of the date appears in the output.
+1
View File
@@ -158,6 +158,7 @@
tree-sitter
gdb
perf
# IDE
unstable.helix
+197
View File
@@ -0,0 +1,197 @@
package markdown
import cm "vendor:commonmark"
import "core:strings"
// DefList_Entry represents a single term-definition pair in a definition list.
DefList_Entry :: struct {
term: string,
definition: string,
}
// convert_deflists scans markdown text for definition list patterns and
// converts them to <dl><dt><dd> HTML blocks before cmark processing.
//
// A definition line starts with optional whitespace followed by a colon and
// a space. The term is the nearest preceding non-blank line (immediately or
// within one blank line). Consecutive term+definition pairs are grouped into
// a single <dl> block.
//
// Terms and definitions are rendered through cmark individually so that
// inline markdown (code, links, emphasis) is processed.
convert_deflists :: proc(body: string, allocator := context.allocator) -> string {
lines := strings.split(body, "\n", allocator = context.temp_allocator)
sb := strings.builder_make(context.temp_allocator)
first := true
need_blank := false
i := 0
for i < len(lines) {
entries, matched, next := try_match_deflist(lines, i)
if matched {
html := render_deflist(entries)
if !first {
strings.write_string(&sb, "\n\n")
}
strings.write_string(&sb, html)
first = false
need_blank = true
i = next
continue
}
if need_blank {
strings.write_string(&sb, "\n\n")
need_blank = false
} else if !first {
strings.write_string(&sb, "\n")
}
strings.write_string(&sb, lines[i])
first = false
i += 1
}
return strings.clone(strings.to_string(sb), allocator)
}
// try_match_deflist attempts to match a definition list group starting at
// lines[start]. A group is one or more term+definition pairs. Returns the
// matched entries, whether a match was found, and the index past the group.
try_match_deflist :: proc(
lines: []string,
start: int,
) -> (
entries: [dynamic]DefList_Entry,
ok: bool,
end: int,
) {
entries = make([dynamic]DefList_Entry, 0, allocator = context.temp_allocator)
end = start
i := start
for i < len(lines) {
// A term must be non-blank and not itself a def line
if is_blank_line(lines[i]) || is_def_line(lines[i]) {
break
}
// Look for a def line: immediately after or with one blank line
def_idx := i + 1
if def_idx < len(lines) && is_blank_line(lines[def_idx]) {
def_idx += 1
}
if def_idx >= len(lines) || !is_def_line(lines[def_idx]) {
break
}
// Found a term + def pair
append(
&entries,
DefList_Entry {
term = strings.trim_space(lines[i]),
definition = def_content(lines[def_idx]),
},
)
i = def_idx + 1
// After a pair, check if another pair follows (optionally
// separated by one blank line). If so, continue the group.
// If not, break without consuming the blank line.
if i < len(lines) && is_blank_line(lines[i]) {
after_blank := i + 1
if after_blank < len(lines) &&
!is_blank_line(lines[after_blank]) &&
!is_def_line(lines[after_blank]) {
// Check whether a def follows this potential term
check_def := after_blank + 1
if check_def < len(lines) && is_blank_line(lines[check_def]) {
check_def += 1
}
if check_def < len(lines) && is_def_line(lines[check_def]) {
i = after_blank
continue
}
}
break
}
}
if len(entries) > 0 {
ok = true
end = i
}
return
}
// is_def_line returns true if the line is a definition line:
// optional leading whitespace, a colon, then whitespace or end-of-line.
is_def_line :: proc(line: string) -> bool {
trimmed := strings.trim_left(line, " \t")
if len(trimmed) < 1 || trimmed[0] != ':' {
return false
}
if len(trimmed) == 1 {
return true
}
return trimmed[1] == ' ' || trimmed[1] == '\t'
}
// def_content extracts the definition text from a definition line,
// stripping the leading colon and surrounding whitespace.
def_content :: proc(line: string) -> string {
trimmed := strings.trim_left(line, " \t")
content := trimmed[1:]
content = strings.trim_left(content, " \t")
return content
}
// is_blank_line returns true for empty or whitespace-only lines.
is_blank_line :: proc(line: string) -> bool {
return strings.trim_space(line) == ""
}
// render_deflist builds the <dl> HTML block from a list of entries.
// Each term and definition is rendered through cmark to process inline
// markdown. Result lives in context.temp_allocator.
render_deflist :: proc(entries: [dynamic]DefList_Entry) -> string {
sb := strings.builder_make(context.temp_allocator)
strings.write_string(&sb, "<dl>")
for entry in entries {
term_html := render_inline_md(entry.term)
def_html := render_inline_md(entry.definition)
strings.write_string(&sb, "<dt>")
strings.write_string(&sb, term_html)
strings.write_string(&sb, "</dt><dd>")
strings.write_string(&sb, def_html)
strings.write_string(&sb, "</dd>")
}
strings.write_string(&sb, "</dl>")
return strings.to_string(sb)
}
// render_inline_md renders a snippet of markdown through cmark and strips
// the surrounding <p> tags. Result lives in context.temp_allocator.
render_inline_md :: proc(text: string) -> string {
raw := cm.markdown_to_html_from_string(text, {.Unsafe})
defer cm.free_string(raw)
return strings.clone(strip_p_tags(raw), context.temp_allocator)
}
// strip_p_tags removes surrounding <p></p> if the HTML is a single paragraph.
strip_p_tags :: proc(html: string) -> string {
s := html
if len(s) > 0 && s[len(s) - 1] == '\n' {
s = s[:len(s) - 1]
}
if strings.has_prefix(s, "<p>") && strings.has_suffix(s, "</p>") {
return s[3:len(s) - 4]
}
return s
}
+104
View File
@@ -0,0 +1,104 @@
#+test
package markdown
import "core:strings"
import "core:testing"
@(test)
test_single_entry :: proc(t: ^testing.T) {
input := "term\n\n: definition"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>term</dt><dd>definition</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_single_entry_no_blank :: proc(t: ^testing.T) {
input := "term\n: definition"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>term</dt><dd>definition</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_indented_variant :: proc(t: ^testing.T) {
input := " term\n : definition"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>term</dt><dd>definition</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_multiple_entries :: proc(t: ^testing.T) {
input := "t1\n\n: d1\n\nt2\n\n: d2"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>t1</dt><dd>d1</dd><dt>t2</dt><dd>d2</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_mixed_indented_and_non_indented :: proc(t: ^testing.T) {
input := "t1\n\n: d1\n\n t2\n : d2\n\nt3\n\n: d3"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>t1</dt><dd>d1</dd><dt>t2</dt><dd>d2</dd><dt>t3</dt><dd>d3</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_inline_markdown_in_term :: proc(t: ^testing.T) {
input := "`code`\n\n: def"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt><code>code</code></dt><dd>def</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_inline_markdown_in_definition :: proc(t: ^testing.T) {
input := "term\n\n: see [Content](#content) here"
result := convert_deflists(input, context.temp_allocator)
testing.expect(t, strings.contains(result, `<a href="#content">Content</a>`))
testing.expect(t, strings.contains(result, "<dd>see "))
testing.expect(t, strings.contains(result, "</dd>"))
}
@(test)
test_regular_text_passes_through :: proc(t: ^testing.T) {
input := "This is: not a deflist"
result := convert_deflists(input, context.temp_allocator)
testing.expect_value(t, result, input)
}
@(test)
test_colon_inside_paragraph_no_false_positive :: proc(t: ^testing.T) {
input := "First paragraph.\n\nSecond paragraph."
result := convert_deflists(input, context.temp_allocator)
testing.expect_value(t, result, input)
}
@(test)
test_deflist_between_paragraphs :: proc(t: ^testing.T) {
input := "Before.\n\nterm\n\n: def\n\nAfter."
result := convert_deflists(input, context.temp_allocator)
testing.expect(t, strings.has_prefix(result, "Before."))
testing.expect(t, strings.contains(result, "<dl><dt>term</dt><dd>def</dd></dl>"))
testing.expect(t, strings.has_suffix(result, "After."))
}
@(test)
test_empty_body :: proc(t: ^testing.T) {
result := convert_deflists("", context.temp_allocator)
testing.expect_value(t, result, "")
}
@(test)
test_docs_md_pattern :: proc(t: ^testing.T) {
input := "content\n\n: `content` holds your pages.\n\n assets\n : `assets` contains files."
result := convert_deflists(input, context.temp_allocator)
testing.expect(t, strings.contains(result, "<dl>"))
testing.expect(t, strings.contains(result, "<dt>content</dt>"))
testing.expect(t, strings.contains(result, "<dt>assets</dt>"))
testing.expect(t, strings.contains(result, "<code>content</code>"))
testing.expect(t, strings.contains(result, "<code>assets</code>"))
testing.expect(t, strings.contains(result, "</dl>"))
}
+1 -17
View File
@@ -1,7 +1,5 @@
package markdown
import cm "vendor:commonmark"
import "core:fmt"
import "core:strings"
@@ -171,9 +169,7 @@ inject_notes :: proc(html: string, sn_defs, mn_defs: map[string]string) -> strin
}
// Render definition through cmark for markdown support
raw_html := cm.markdown_to_html_from_string(def_text, {.Unsafe})
defer cm.free_string(raw_html)
def_html := strip_p_tags(raw_html)
def_html := render_inline_md(def_text)
note: string
defer delete(note)
@@ -200,15 +196,3 @@ inject_notes :: proc(html: string, sn_defs, mn_defs: map[string]string) -> strin
return strings.to_string(parts)
}
// strip_p_tags removes surrounding <p></p> if the HTML is a single paragraph.
strip_p_tags :: proc(html: string) -> string {
s := html
if len(s) > 0 && s[len(s) - 1] == '\n' {
s = s[:len(s) - 1]
}
if strings.has_prefix(s, "<p>") && strings.has_suffix(s, "</p>") {
return s[3:len(s) - 4]
}
return s
}
+9 -1
View File
@@ -12,9 +12,10 @@ Extension :: enum {
Highlight,
Sections,
HeadingIDs,
DefLists,
}
DEFAULT_EXTENSIONS :: bit_set[Extension]{.Emoji, .Sidenotes, .Alerts, .HeadingIDs}
DEFAULT_EXTENSIONS :: bit_set[Extension]{.Emoji, .Sidenotes, .Alerts, .HeadingIDs, .DefLists}
// Caller is responsible for freeing string
process :: proc(
@@ -29,6 +30,9 @@ process :: proc(
if .Sidenotes in ext {
clean_body, side_notes, margin_notes = strip_definitions(body)
}
if .DefLists in ext {
clean_body = convert_deflists(clean_body, allocator)
}
original_html := cm.markdown_to_html_from_string(clean_body, {.Unsafe})
html := strings.clone(original_html, allocator)
cm.free_string(original_html)
@@ -71,6 +75,8 @@ parse_extension_list :: proc(s: string) -> (result: bit_set[Extension]) {
result += {.Sections}
case "heading_ids":
result += {.HeadingIDs}
case "deflists":
result += {.DefLists}
}
}
return result
@@ -94,6 +100,8 @@ apply_extension_config :: proc(ext: ^bit_set[Extension], config: json.Object) {
if enabled {ext^ += {.Sections}} else {ext^ -= {.Sections}}
case "heading_ids":
if enabled {ext^ += {.HeadingIDs}} else {ext^ -= {.HeadingIDs}}
case "deflists":
if enabled {ext^ += {.DefLists}} else {ext^ -= {.DefLists}}
}
}
}
+1 -3
View File
@@ -82,9 +82,7 @@ TODO: Describe
TODO: Don't forget to highlight differences from Hugo.
## Templates[^tempmod]
[^tempmod]: Template modification is an "advanced" feature, and shoud probably be discussed later in the page. (or possibly in the guide.)
## Templates
Sites are built using one or more template files written in an extended version of [mustache](https://mustache.github.io) templates. The [mustache manual](https://mustache.github.io/mustache.5.html) has great explainations and a lot of examples if you want to know more, but I'll summarize them for you here.