feat: Added deflist markdown extension.

This commit is contained in:
Spencer Brower
2026-08-04 12:53:01 -04:00
parent 0e8d770ac8
commit 876f767548
8 changed files with 317 additions and 22 deletions
+3 -1
View File
@@ -79,6 +79,7 @@ thor/
| | `sectionate.odin` | `wrap_sections` — splits HTML at `<h2>` into `<section>` wrappers | | | `sectionate.odin` | `wrap_sections` — splits HTML at `<h2>` into `<section>` wrappers |
| | `highlight.odin` | Syntax highlighting via tree-sitter. Imports `../treesitter`. | | | `highlight.odin` | Syntax highlighting via tree-sitter. Imports `../treesitter`. |
| | `heading_ids.odin` | `inject_heading_ids` — adds `id` attributes to `<h1>`-`<h6>` from heading text. Slug-based, deduplicated. | | | `heading_ids.odin` | `inject_heading_ids` — adds `id` attributes to `<h1>`-`<h6>` from heading text. Slug-based, deduplicated. |
| | `deflists.odin` | `convert_deflists` — pre-cmark pass. Scans for definition list patterns (`term\n\n: definition`) and converts to `<dl><dt><dd>` HTML blocks. Terms and definitions rendered through cmark individually for inline markdown. Consecutive pairs grouped into single `<dl>`. |
| | `toc.odin` | `generate_toc(html, allocator)` — page-level feature (not a pipeline extension). Scans `<h1>`-`<h6>` for IDs (after `inject_heading_ids`), builds nested `<ul>` with `<a href="#id">` links. Called from `load_page` when frontmatter `"toc": true`. Depends on `.HeadingIDs` being enabled. | | | `toc.odin` | `generate_toc(html, allocator)` — page-level feature (not a pipeline extension). Scans `<h1>`-`<h6>` for IDs (after `inject_heading_ids`), builds nested `<ul>` with `<a href="#id">` links. Called from `load_page` when frontmatter `"toc": true`. Depends on `.HeadingIDs` being enabled. |
| `mustache/` | See [Mustache engine](#mustache-engine) below | Template engine | | `mustache/` | See [Mustache engine](#mustache-engine) below | Template engine |
| `bench/` | `bench.odin` + `templates/` | Standalone template rendering benchmark. Generates 500 posts + 100 comments, renders with indented partials + inheritance + pipes. `--dump <path>` for output validation, positional arg for iteration count (default 250). | | `bench/` | `bench.odin` + `templates/` | Standalone template rendering benchmark. Generates 500 posts + 100 comments, renders with indented partials + inheritance + pipes. `--dump <path>` for output validation, positional arg for iteration count (default 250). |
@@ -177,7 +178,7 @@ Config is split into three structs with a clear 5-step initialization flow:
**`Feature` enum** — `Drafts`, `Minify`, `Watch`. Checked with `.Minify in site.features`. **`Feature` enum** — `Drafts`, `Minify`, `Watch`. Checked with `.Minify in site.features`.
**`markdown.Extension` enum** (in the `markdown` package, not main) — `Emoji`, `Sidenotes`, `Alerts`, `Highlight`, `Sections`, `HeadingIDs`. Default is `md.DEFAULT_EXTENSIONS` (currently `.Emoji, .Sidenotes, .Alerts, .HeadingIDs`). Configurable via: **`markdown.Extension` enum** (in the `markdown` package, not main) — `Emoji`, `Sidenotes`, `Alerts`, `Highlight`, `Sections`, `HeadingIDs`, `DefLists`. Default is `md.DEFAULT_EXTENSIONS` (currently `.Emoji, .Sidenotes, .Alerts, .HeadingIDs, .DefLists`). Configurable via:
- `thor.json`: `"markdown_extensions": { "emoji": true, "highlight": false, ... }` - `thor.json`: `"markdown_extensions": { "emoji": true, "highlight": false, ... }`
- CLI: `-ext:highlight,sections` (enable) / `-no-ext:emoji` (disable). Comma-separated, case-insensitive. - CLI: `-ext:highlight,sections` (enable) / `-no-ext:emoji` (disable). Comma-separated, case-insensitive.
@@ -258,6 +259,7 @@ Lives in the `markdown` package. Entry point: `md.process(body, ext, file_path)`
``` ```
raw markdown raw markdown
→ md.strip_definitions (if .Sidenotes — pre-cmark) → md.strip_definitions (if .Sidenotes — pre-cmark)
→ md.convert_deflists (if .DefLists — pre-cmark)
→ cmark markdown_to_html (Unsafe mode for HTML passthrough) → cmark markdown_to_html (Unsafe mode for HTML passthrough)
→ md.expand_emoji (if .Emoji — post-cmark) → md.expand_emoji (if .Emoji — post-cmark)
→ md.inject_notes (if .Sidenotes — post-cmark) → md.inject_notes (if .Sidenotes — post-cmark)
+1
View File
@@ -118,6 +118,7 @@
- Checks all links on each page to make sure they are valid. - Checks all links on each page to make sure they are valid.
- [ ] Peruse [GitHub's](https://docs.github.com/en/get-started/writing-on-github/getting-started-with-writing-and-formatting-on-github/basic-writing-and-formatting-syntax#alerts) - [ ] Peruse [GitHub's](https://docs.github.com/en/get-started/writing-on-github/getting-started-with-writing-and-formatting-on-github/basic-writing-and-formatting-syntax#alerts)
docs for any juicy nuggets we may have missed. docs for any juicy nuggets we may have missed.
- [ ] Avoid using `render_inline_md` if possible.
## Dates ## Dates
- [ ] display an error when no part of the date appears in the output. - [ ] display an error when no part of the date appears in the output.
+1
View File
@@ -158,6 +158,7 @@
tree-sitter tree-sitter
gdb gdb
perf
# IDE # IDE
unstable.helix unstable.helix
+197
View File
@@ -0,0 +1,197 @@
package markdown
import cm "vendor:commonmark"
import "core:strings"
// DefList_Entry represents a single term-definition pair in a definition list.
DefList_Entry :: struct {
term: string,
definition: string,
}
// convert_deflists scans markdown text for definition list patterns and
// converts them to <dl><dt><dd> HTML blocks before cmark processing.
//
// A definition line starts with optional whitespace followed by a colon and
// a space. The term is the nearest preceding non-blank line (immediately or
// within one blank line). Consecutive term+definition pairs are grouped into
// a single <dl> block.
//
// Terms and definitions are rendered through cmark individually so that
// inline markdown (code, links, emphasis) is processed.
convert_deflists :: proc(body: string, allocator := context.allocator) -> string {
lines := strings.split(body, "\n", allocator = context.temp_allocator)
sb := strings.builder_make(context.temp_allocator)
first := true
need_blank := false
i := 0
for i < len(lines) {
entries, matched, next := try_match_deflist(lines, i)
if matched {
html := render_deflist(entries)
if !first {
strings.write_string(&sb, "\n\n")
}
strings.write_string(&sb, html)
first = false
need_blank = true
i = next
continue
}
if need_blank {
strings.write_string(&sb, "\n\n")
need_blank = false
} else if !first {
strings.write_string(&sb, "\n")
}
strings.write_string(&sb, lines[i])
first = false
i += 1
}
return strings.clone(strings.to_string(sb), allocator)
}
// try_match_deflist attempts to match a definition list group starting at
// lines[start]. A group is one or more term+definition pairs. Returns the
// matched entries, whether a match was found, and the index past the group.
try_match_deflist :: proc(
lines: []string,
start: int,
) -> (
entries: [dynamic]DefList_Entry,
ok: bool,
end: int,
) {
entries = make([dynamic]DefList_Entry, 0, allocator = context.temp_allocator)
end = start
i := start
for i < len(lines) {
// A term must be non-blank and not itself a def line
if is_blank_line(lines[i]) || is_def_line(lines[i]) {
break
}
// Look for a def line: immediately after or with one blank line
def_idx := i + 1
if def_idx < len(lines) && is_blank_line(lines[def_idx]) {
def_idx += 1
}
if def_idx >= len(lines) || !is_def_line(lines[def_idx]) {
break
}
// Found a term + def pair
append(
&entries,
DefList_Entry {
term = strings.trim_space(lines[i]),
definition = def_content(lines[def_idx]),
},
)
i = def_idx + 1
// After a pair, check if another pair follows (optionally
// separated by one blank line). If so, continue the group.
// If not, break without consuming the blank line.
if i < len(lines) && is_blank_line(lines[i]) {
after_blank := i + 1
if after_blank < len(lines) &&
!is_blank_line(lines[after_blank]) &&
!is_def_line(lines[after_blank]) {
// Check whether a def follows this potential term
check_def := after_blank + 1
if check_def < len(lines) && is_blank_line(lines[check_def]) {
check_def += 1
}
if check_def < len(lines) && is_def_line(lines[check_def]) {
i = after_blank
continue
}
}
break
}
}
if len(entries) > 0 {
ok = true
end = i
}
return
}
// is_def_line returns true if the line is a definition line:
// optional leading whitespace, a colon, then whitespace or end-of-line.
is_def_line :: proc(line: string) -> bool {
trimmed := strings.trim_left(line, " \t")
if len(trimmed) < 1 || trimmed[0] != ':' {
return false
}
if len(trimmed) == 1 {
return true
}
return trimmed[1] == ' ' || trimmed[1] == '\t'
}
// def_content extracts the definition text from a definition line,
// stripping the leading colon and surrounding whitespace.
def_content :: proc(line: string) -> string {
trimmed := strings.trim_left(line, " \t")
content := trimmed[1:]
content = strings.trim_left(content, " \t")
return content
}
// is_blank_line returns true for empty or whitespace-only lines.
is_blank_line :: proc(line: string) -> bool {
return strings.trim_space(line) == ""
}
// render_deflist builds the <dl> HTML block from a list of entries.
// Each term and definition is rendered through cmark to process inline
// markdown. Result lives in context.temp_allocator.
render_deflist :: proc(entries: [dynamic]DefList_Entry) -> string {
sb := strings.builder_make(context.temp_allocator)
strings.write_string(&sb, "<dl>")
for entry in entries {
term_html := render_inline_md(entry.term)
def_html := render_inline_md(entry.definition)
strings.write_string(&sb, "<dt>")
strings.write_string(&sb, term_html)
strings.write_string(&sb, "</dt><dd>")
strings.write_string(&sb, def_html)
strings.write_string(&sb, "</dd>")
}
strings.write_string(&sb, "</dl>")
return strings.to_string(sb)
}
// render_inline_md renders a snippet of markdown through cmark and strips
// the surrounding <p> tags. Result lives in context.temp_allocator.
render_inline_md :: proc(text: string) -> string {
raw := cm.markdown_to_html_from_string(text, {.Unsafe})
defer cm.free_string(raw)
return strings.clone(strip_p_tags(raw), context.temp_allocator)
}
// strip_p_tags removes surrounding <p></p> if the HTML is a single paragraph.
strip_p_tags :: proc(html: string) -> string {
s := html
if len(s) > 0 && s[len(s) - 1] == '\n' {
s = s[:len(s) - 1]
}
if strings.has_prefix(s, "<p>") && strings.has_suffix(s, "</p>") {
return s[3:len(s) - 4]
}
return s
}
+104
View File
@@ -0,0 +1,104 @@
#+test
package markdown
import "core:strings"
import "core:testing"
@(test)
test_single_entry :: proc(t: ^testing.T) {
input := "term\n\n: definition"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>term</dt><dd>definition</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_single_entry_no_blank :: proc(t: ^testing.T) {
input := "term\n: definition"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>term</dt><dd>definition</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_indented_variant :: proc(t: ^testing.T) {
input := " term\n : definition"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>term</dt><dd>definition</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_multiple_entries :: proc(t: ^testing.T) {
input := "t1\n\n: d1\n\nt2\n\n: d2"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>t1</dt><dd>d1</dd><dt>t2</dt><dd>d2</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_mixed_indented_and_non_indented :: proc(t: ^testing.T) {
input := "t1\n\n: d1\n\n t2\n : d2\n\nt3\n\n: d3"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt>t1</dt><dd>d1</dd><dt>t2</dt><dd>d2</dd><dt>t3</dt><dd>d3</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_inline_markdown_in_term :: proc(t: ^testing.T) {
input := "`code`\n\n: def"
result := convert_deflists(input, context.temp_allocator)
expected := "<dl><dt><code>code</code></dt><dd>def</dd></dl>"
testing.expect_value(t, result, expected)
}
@(test)
test_inline_markdown_in_definition :: proc(t: ^testing.T) {
input := "term\n\n: see [Content](#content) here"
result := convert_deflists(input, context.temp_allocator)
testing.expect(t, strings.contains(result, `<a href="#content">Content</a>`))
testing.expect(t, strings.contains(result, "<dd>see "))
testing.expect(t, strings.contains(result, "</dd>"))
}
@(test)
test_regular_text_passes_through :: proc(t: ^testing.T) {
input := "This is: not a deflist"
result := convert_deflists(input, context.temp_allocator)
testing.expect_value(t, result, input)
}
@(test)
test_colon_inside_paragraph_no_false_positive :: proc(t: ^testing.T) {
input := "First paragraph.\n\nSecond paragraph."
result := convert_deflists(input, context.temp_allocator)
testing.expect_value(t, result, input)
}
@(test)
test_deflist_between_paragraphs :: proc(t: ^testing.T) {
input := "Before.\n\nterm\n\n: def\n\nAfter."
result := convert_deflists(input, context.temp_allocator)
testing.expect(t, strings.has_prefix(result, "Before."))
testing.expect(t, strings.contains(result, "<dl><dt>term</dt><dd>def</dd></dl>"))
testing.expect(t, strings.has_suffix(result, "After."))
}
@(test)
test_empty_body :: proc(t: ^testing.T) {
result := convert_deflists("", context.temp_allocator)
testing.expect_value(t, result, "")
}
@(test)
test_docs_md_pattern :: proc(t: ^testing.T) {
input := "content\n\n: `content` holds your pages.\n\n assets\n : `assets` contains files."
result := convert_deflists(input, context.temp_allocator)
testing.expect(t, strings.contains(result, "<dl>"))
testing.expect(t, strings.contains(result, "<dt>content</dt>"))
testing.expect(t, strings.contains(result, "<dt>assets</dt>"))
testing.expect(t, strings.contains(result, "<code>content</code>"))
testing.expect(t, strings.contains(result, "<code>assets</code>"))
testing.expect(t, strings.contains(result, "</dl>"))
}
+1 -17
View File
@@ -1,7 +1,5 @@
package markdown package markdown
import cm "vendor:commonmark"
import "core:fmt" import "core:fmt"
import "core:strings" import "core:strings"
@@ -171,9 +169,7 @@ inject_notes :: proc(html: string, sn_defs, mn_defs: map[string]string) -> strin
} }
// Render definition through cmark for markdown support // Render definition through cmark for markdown support
raw_html := cm.markdown_to_html_from_string(def_text, {.Unsafe}) def_html := render_inline_md(def_text)
defer cm.free_string(raw_html)
def_html := strip_p_tags(raw_html)
note: string note: string
defer delete(note) defer delete(note)
@@ -200,15 +196,3 @@ inject_notes :: proc(html: string, sn_defs, mn_defs: map[string]string) -> strin
return strings.to_string(parts) return strings.to_string(parts)
} }
// strip_p_tags removes surrounding <p></p> if the HTML is a single paragraph.
strip_p_tags :: proc(html: string) -> string {
s := html
if len(s) > 0 && s[len(s) - 1] == '\n' {
s = s[:len(s) - 1]
}
if strings.has_prefix(s, "<p>") && strings.has_suffix(s, "</p>") {
return s[3:len(s) - 4]
}
return s
}
+9 -1
View File
@@ -12,9 +12,10 @@ Extension :: enum {
Highlight, Highlight,
Sections, Sections,
HeadingIDs, HeadingIDs,
DefLists,
} }
DEFAULT_EXTENSIONS :: bit_set[Extension]{.Emoji, .Sidenotes, .Alerts, .HeadingIDs} DEFAULT_EXTENSIONS :: bit_set[Extension]{.Emoji, .Sidenotes, .Alerts, .HeadingIDs, .DefLists}
// Caller is responsible for freeing string // Caller is responsible for freeing string
process :: proc( process :: proc(
@@ -29,6 +30,9 @@ process :: proc(
if .Sidenotes in ext { if .Sidenotes in ext {
clean_body, side_notes, margin_notes = strip_definitions(body) clean_body, side_notes, margin_notes = strip_definitions(body)
} }
if .DefLists in ext {
clean_body = convert_deflists(clean_body, allocator)
}
original_html := cm.markdown_to_html_from_string(clean_body, {.Unsafe}) original_html := cm.markdown_to_html_from_string(clean_body, {.Unsafe})
html := strings.clone(original_html, allocator) html := strings.clone(original_html, allocator)
cm.free_string(original_html) cm.free_string(original_html)
@@ -71,6 +75,8 @@ parse_extension_list :: proc(s: string) -> (result: bit_set[Extension]) {
result += {.Sections} result += {.Sections}
case "heading_ids": case "heading_ids":
result += {.HeadingIDs} result += {.HeadingIDs}
case "deflists":
result += {.DefLists}
} }
} }
return result return result
@@ -94,6 +100,8 @@ apply_extension_config :: proc(ext: ^bit_set[Extension], config: json.Object) {
if enabled {ext^ += {.Sections}} else {ext^ -= {.Sections}} if enabled {ext^ += {.Sections}} else {ext^ -= {.Sections}}
case "heading_ids": case "heading_ids":
if enabled {ext^ += {.HeadingIDs}} else {ext^ -= {.HeadingIDs}} if enabled {ext^ += {.HeadingIDs}} else {ext^ -= {.HeadingIDs}}
case "deflists":
if enabled {ext^ += {.DefLists}} else {ext^ -= {.DefLists}}
} }
} }
} }
+1 -3
View File
@@ -82,9 +82,7 @@ TODO: Describe
TODO: Don't forget to highlight differences from Hugo. TODO: Don't forget to highlight differences from Hugo.
## Templates[^tempmod] ## Templates
[^tempmod]: Template modification is an "advanced" feature, and shoud probably be discussed later in the page. (or possibly in the guide.)
Sites are built using one or more template files written in an extended version of [mustache](https://mustache.github.io) templates. The [mustache manual](https://mustache.github.io/mustache.5.html) has great explainations and a lot of examples if you want to know more, but I'll summarize them for you here. Sites are built using one or more template files written in an extended version of [mustache](https://mustache.github.io) templates. The [mustache manual](https://mustache.github.io/mustache.5.html) has great explainations and a lot of examples if you want to know more, but I'll summarize them for you here.