-
Notifications
You must be signed in to change notification settings - Fork 72
Expand file tree
/
Copy pathmarkdown_versions.py
More file actions
292 lines (218 loc) · 7.32 KB
/
Copy pathmarkdown_versions.py
File metadata and controls
292 lines (218 loc) · 7.32 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
from pathlib import Path
import html
import re
# Markdown captured while MkDocs processes each page.
_markdown_pages = {}
HTML_COMMENT_RE = re.compile(
r"<!--.*?-->",
re.DOTALL,
)
SCRIPT_STYLE_RE = re.compile(
r"<(script|style)\b[^>]*>.*?</\1\s*>",
re.IGNORECASE | re.DOTALL,
)
BLOCK_HTML_RE = re.compile(
r"</?(?:div|span|section|article|aside|nav|header|footer|main)"
r"\b[^>]*>",
re.IGNORECASE,
)
BR_RE = re.compile(
r"<br\s*/?>",
re.IGNORECASE,
)
ADMONITION_RE = re.compile(
r'^(?P<indent>[ \t]*)'
r'!!![ \t]+'
r'(?P<type>[\w-]+)'
r'(?:[ \t]+"(?P<title>[^"]*)")?'
r"[ \t]*$"
)
FONTAWESOME_RE = re.compile(
r":fontawesome-(?:brands|regular|solid)-([a-z0-9-]+):",
re.IGNORECASE,
)
ATTRIBUTE_LIST_RE = re.compile(
r"[ \t]*\{:[ \t]*[^{}\n]*\}",
)
EMPTY_LINK_PADDING_RE = re.compile(
r"\[[ \t]+([^]]*?)\]"
)
# Icons whose meaning should be retained in plain Markdown.
# All unlisted Font Awesome icons are treated as decorative.
ICON_REPLACEMENTS = {
"dollar-sign": "$",
"sterling-sign": "£",
"euro-sign": "€",
}
def on_config(config, **kwargs):
"""
Clear saved pages at the beginning of each build.
This matters when using `mkdocs serve`, where multiple builds can run in
the same Python process.
"""
_markdown_pages.clear()
return config
def _strip_front_matter(markdown):
"""Remove YAML front matter from the published Markdown."""
lines = markdown.splitlines()
if not lines or lines[0].strip() != "---":
return markdown
for index in range(1, len(lines)):
if lines[index].strip() == "---":
return "\n".join(lines[index + 1:])
# No closing delimiter was found, so leave the content unchanged.
return markdown
def _convert_admonitions(markdown):
"""
Convert MkDocs admonitions to ordinary Markdown blockquotes.
For example:
!!! warning "Important"
Do not run this in production.
becomes:
> **Warning — Important**
>
> Do not run this in production.
"""
lines = markdown.splitlines()
output = []
index = 0
while index < len(lines):
match = ADMONITION_RE.match(lines[index])
if not match:
output.append(lines[index])
index += 1
continue
base_indent = len(match.group("indent").expandtabs(4))
admonition_type = match.group("type").replace("-", " ").title()
title = match.group("title")
if title:
heading = f"**{admonition_type} — {title}**"
else:
heading = f"**{admonition_type}**"
output.append(f"> {heading}")
output.append(">")
index += 1
while index < len(lines):
line = lines[index]
if not line.strip():
output.append(">")
index += 1
continue
expanded = line.expandtabs(4)
indentation = len(expanded) - len(expanded.lstrip(" "))
# A non-indented line marks the end of the admonition.
if indentation <= base_indent:
break
# MkDocs admonition bodies normally have four additional spaces.
content_start = min(base_indent + 4, len(expanded))
content = expanded[content_start:]
output.append(f"> {content}" if content else ">")
index += 1
return "\n".join(output)
def _replace_fontawesome(match):
"""
Replace meaningful icons with text and discard decorative icons.
"""
icon_name = match.group(1).lower()
return ICON_REPLACEMENTS.get(icon_name, "")
def _clean_text_fragment(text):
"""
Remove presentation-specific constructs from Markdown text.
This function is only called for text outside fenced code blocks.
"""
text = HTML_COMMENT_RE.sub("", text)
text = SCRIPT_STYLE_RE.sub("", text)
# Preserve the intended line break before removing container HTML.
text = BR_RE.sub("\n", text)
text = BLOCK_HTML_RE.sub("", text)
# Remove MkDocs/Material presentation syntax.
text = ATTRIBUTE_LIST_RE.sub("", text)
text = FONTAWESOME_RE.sub(_replace_fontawesome, text)
# Removing an icon can leave whitespace at the beginning of link text:
#
# [ Download](/download/)
#
# Change that back to:
#
# [Download](/download/)
text = EMPTY_LINK_PADDING_RE.sub(r"[\1]", text)
return html.unescape(text)
def _clean_html_outside_code_fences(markdown):
"""
Clean HTML and Material syntax without modifying fenced code blocks.
Both backtick and tilde fences are recognised.
"""
output = []
text_buffer = []
in_fence = False
fence_character = None
fence_length = 0
def flush_text():
if not text_buffer:
return
text = "\n".join(text_buffer)
text = _clean_text_fragment(text)
output.extend(text.splitlines())
text_buffer.clear()
for line in markdown.splitlines():
stripped = line.lstrip()
fence_match = re.match(r"(`{3,}|~{3,})", stripped)
if fence_match:
marker = fence_match.group(1)
marker_character = marker[0]
if not in_fence:
flush_text()
in_fence = True
fence_character = marker_character
fence_length = len(marker)
output.append(line)
continue
if (
marker_character == fence_character
and len(marker) >= fence_length
):
output.append(line)
in_fence = False
fence_character = None
fence_length = 0
continue
if in_fence:
output.append(line)
else:
text_buffer.append(line)
flush_text()
return "\n".join(output)
def _normalise_blank_lines(markdown):
"""Remove excessive blank lines and ensure one final newline."""
markdown = re.sub(r"\n{4,}", "\n\n\n", markdown)
return markdown.strip() + "\n"
def _sanitise_markdown(markdown):
"""Create the clean Markdown representation of a page."""
markdown = _strip_front_matter(markdown)
markdown = _convert_admonitions(markdown)
markdown = _clean_html_outside_code_fences(markdown)
markdown = _normalise_blank_lines(markdown)
return markdown
def on_page_markdown(markdown, *, page, config, files, **kwargs):
"""
Capture a cleaned Markdown version of every generated page.
The original Markdown is returned so this hook does not change the normal
HTML output.
"""
_markdown_pages[page.url] = _sanitise_markdown(markdown)
return markdown
def on_post_build(*, config, **kwargs):
"""Write the cleaned Markdown files beside the generated HTML site."""
site_dir = Path(config["site_dir"])
for page_url, markdown in _markdown_pages.items():
url = page_url.rstrip("/")
# Support configurations where MkDocs produces page.html rather than
# directory-style page/index.html URLs.
if url.endswith(".html"):
url = url[:-5]
if url:
target = site_dir / f"{url}.md"
else:
target = site_dir / "index.md"
target.parent.mkdir(parents=True, exist_ok=True)
target.write_text(markdown, encoding="utf-8")