-
-
Notifications
You must be signed in to change notification settings - Fork 33
Expand file tree
/
Copy pathmdrender.py
More file actions
354 lines (298 loc) · 13.3 KB
/
Copy pathmdrender.py
File metadata and controls
354 lines (298 loc) · 13.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
"""Markdown -> Substack ProseMirror via markdown-it-py.
Implements Post.from_markdown() using a real CommonMark parser (markdown-it-py)
plus the standard footnote plugin, with a small renderer that walks the syntax
tree into Substack's node schema.
Node construction goes through ``substack.nodes`` so the (undocumented) schema
lives in exactly one place.
Footnotes: Substack numbers footnote anchors by their position in the document
and pairs them one-to-one, in order, with the footnote blocks at the end (it
ignores any explicit number and does not support one block serving several
anchors). So each reference is emitted as its own sequentially-numbered anchor,
and a matching footnote block is appended for each -- a definition referenced
more than once is duplicated, which mirrors how Substack's own editor behaves.
"""
from __future__ import annotations
import base64
import copy
import json
import re
from pathlib import Path
from typing import Dict, List, Optional
from urllib.parse import unquote, urlsplit
from markdown_it import MarkdownIt
from markdown_it.tree import SyntaxTreeNode
from mdit_py_plugins.container import container_plugin
from mdit_py_plugins.dollarmath import dollarmath_plugin
from mdit_py_plugins.footnote import footnote_plugin
from mdit_py_plugins.subscript import sub_plugin
from mdit_py_plugins.superscript import superscript_plugin
from substack import nodes
from substack.nodes import MarkType, NodeType
_MARK_FOR = {
"strong": {"type": MarkType.STRONG},
"em": {"type": MarkType.EM},
"s": {"type": MarkType.STRIKETHROUGH},
"sup": {"type": MarkType.SUPERSCRIPT},
"sub": {"type": MarkType.SUBSCRIPT},
}
def parse_node_marker(comment_content: str) -> dict | None:
"""
Parse a python-substack-node:v1 comment marker and return the parsed JSON dictionary.
If it is not a python-substack-node:v1 marker, returns None.
If it is an attempted marker but is corrupt/malformed, raises ValueError.
"""
clean = comment_content.strip()
match = re.match(
r"^<!--\s*python-substack-node:v1\s+([A-Za-z0-9_-]+=*)\s*-->$", clean
)
if not match:
if "python-substack-node:v1" in clean:
raise ValueError("Corrupt marker format")
return None
encoded = match.group(1)
try:
padding = len(encoded) % 4
if padding:
encoded += "=" * (4 - padding)
decoded_bytes = base64.urlsafe_b64decode(encoded.encode("ascii"))
except Exception as exc:
raise ValueError("Invalid URL-safe base64 in marker") from exc
try:
decoded_str = decoded_bytes.decode("utf-8")
except Exception as exc:
raise ValueError("Invalid UTF-8 in marker") from exc
try:
data = json.loads(decoded_str)
except Exception as exc:
raise ValueError("Invalid JSON in marker") from exc
if not isinstance(data, dict):
raise ValueError("Marker payload is not a top-level JSON object")
node_type = data.get("type")
if not isinstance(node_type, str) or not node_type:
raise ValueError("Marker payload has missing or empty 'type'")
return data
def parse_image_marker(comment_content: str) -> dict | None:
clean = comment_content.strip()
match = re.match(
r"^<!--\s*python-substack-image:v1\s+([A-Za-z0-9_-]+=*)\s*-->$", clean
)
if not match:
if "python-substack-image:v1" in clean:
raise ValueError("Corrupt image marker format")
return None
encoded = match.group(1)
encoded += "=" * (-len(encoded) % 4)
try:
attrs = json.loads(base64.urlsafe_b64decode(encoded.encode("ascii")))
except Exception as exc:
raise ValueError("Invalid image marker") from exc
if not isinstance(attrs, dict) or any(
key in {"src", "alt", "href", "isProcessing"} for key in attrs
):
raise ValueError("Invalid image marker attributes")
return attrs
def _make_parser() -> MarkdownIt:
return (
MarkdownIt("commonmark")
.use(footnote_plugin)
# Pandoc-style delimiters: no whitespace just inside the dollars and no
# digit just outside them, so paired currency amounts ("$5 ... $10")
# stay plain text instead of becoming math.
.use(dollarmath_plugin, allow_space=False, allow_digits=False)
.use(sub_plugin)
.use(superscript_plugin)
.use(container_plugin, name="pullquote")
.use(container_plugin, name="callout")
.enable("strikethrough")
)
def _coalesce(out_nodes: List[Dict]) -> List[Dict]:
"""Merge adjacent text nodes that carry identical marks (e.g. softbreaks)."""
merged: List[Dict] = []
for node in out_nodes:
if (
merged
and node.get("type") == NodeType.TEXT
and merged[-1].get("type") == NodeType.TEXT
and node.get("marks") == merged[-1].get("marks")
):
merged[-1]["text"] += node["text"]
else:
merged.append(node)
return merged
def _render_inline(node: SyntaxTreeNode, marks: List[Dict], ctx: Dict) -> List[Dict]:
"""Render an inline subtree into a flat list of text / anchor nodes."""
out: List[Dict] = []
for child in node.children:
t = child.type
if t == "text":
if child.content:
out.append(nodes.text(child.content, marks))
elif t == "code_inline":
out.append(nodes.text(child.content, marks + [nodes.code_mark()]))
elif t == "math_inline":
out.append(nodes.latex_inline(child.content.strip()))
elif t in _MARK_FOR:
out.extend(_render_inline(child, marks + [_MARK_FOR[t]], ctx))
elif t == "link":
href = child.attrs.get("href", "")
out.extend(_render_inline(child, marks + [nodes.link_mark(href)], ctx))
elif t in ("softbreak", "hardbreak"):
out.append(nodes.text(" ", marks))
elif t == "footnote_ref":
# Number anchors by document position and record which definition each
# one points to, so matching blocks can be emitted 1:1 afterwards.
ctx["order"].append(child.meta["id"])
out.append(nodes.footnote_anchor(len(ctx["order"])))
elif t == "image":
# Inline images are rare in this schema; fall back to alt text.
alt = child.attrs.get("alt") or "".join(
c.content for c in child.children if c.type == "text"
)
if alt:
out.append(nodes.text(alt, marks))
elif t == "html_inline":
marker_data = parse_node_marker(child.content)
if marker_data is not None:
out.append(marker_data)
return _coalesce(out)
def _only_image(inline: SyntaxTreeNode) -> Optional[SyntaxTreeNode]:
"""If an inline node is just an image (optionally wrapped in a link), return it."""
kids = [c for c in inline.children if c.type != "softbreak"]
if len(kids) == 1 and kids[0].type == "image":
return kids[0]
if len(kids) == 1 and kids[0].type == "link":
inner = [c for c in kids[0].children if c.type != "softbreak"]
if len(inner) == 1 and inner[0].type == "image":
img = inner[0]
img._link_href = kids[0].attrs.get("href") # type: ignore[attr-defined]
return img
return None
def _captioned_image(img: SyntaxTreeNode, api) -> Dict:
src = img.attrs.get("src", "")
if api is not None and (
urlsplit(src).scheme.lower() not in ("http", "https")
and not src.startswith("//")
):
# Markdown destinations encode spaces and non-ASCII characters as URLs.
path = Path(unquote(src)).expanduser()
if not path.is_file() and src.startswith("/"):
# Preserve legacy root-relative asset paths only when they resolve
# to a real file; an existing absolute path always takes precedence.
relative = Path(unquote(src[1:]))
if relative.is_file():
path = relative
if not path.is_file():
raise FileNotFoundError(f"Local image file not found: {path}")
try:
uploaded = api.get_image(str(path))
url = uploaded.get("url")
if not isinstance(url, str) or not url.startswith(("https://", "http://")):
raise ValueError("Upload did not return an HTTP image URL")
except Exception as exc:
raise ValueError(f"Failed to upload local image: {path}") from exc
src = url
# markdown-it stores the image alt text as the node's content, not in attrs.
alt = img.content or img.attrs.get("alt") or None
# Standard markdown image title `` maps to Substack's caption node.
title = img.attrs.get("title") or None
caption = [nodes.text(title)] if title else None
return nodes.captioned_image(
src,
alt=alt,
href=getattr(img, "_link_href", None),
caption=caption,
)
def _render_block(node: SyntaxTreeNode, api, ctx: Dict) -> List[Dict]:
"""Render a block-level node into zero or more Substack nodes."""
t = node.type
if t == "html_block":
marker_data = parse_node_marker(node.content)
if marker_data is not None:
return [marker_data]
return []
if t == "paragraph":
inline = node.children[0]
img = _only_image(inline)
if img is not None:
return [_captioned_image(img, api)]
return [nodes.paragraph(_render_inline(inline, [], ctx))]
if t == "heading":
level = int(node.tag[1])
return [nodes.heading(_render_inline(node.children[0], [], ctx), level=level)]
if t == "hr":
return [nodes.horizontal_rule()]
if t in ("fence", "code_block"):
return [
nodes.code_block(
node.content.rstrip("\n"), language=node.info.strip() or None
)
]
if t == "blockquote":
paras: List[Dict] = []
for child in node.children:
paras.extend(_render_block(child, api, ctx))
return [nodes.blockquote(paras)]
if t == "bullet_list":
return [nodes.bullet_list(_render_list_items(node, api, ctx))]
if t == "ordered_list":
return [nodes.ordered_list(_render_list_items(node, api, ctx))]
# "$$...$$ (label)" tokenizes as math_block_label; Substack has no equation
# labels, so it renders like an unlabeled block.
if t in ("math_block", "math_block_label"):
return [nodes.latex_block(node.content.strip())]
if t == "container_pullquote":
return [nodes.pullquote(_render_container_body(node, api, ctx))]
if t == "container_callout":
return [nodes.callout_block(_render_container_body(node, api, ctx))]
# footnote_block is handled separately in markdown_to_doc; ignore it here.
return []
def _render_container_body(node: SyntaxTreeNode, api, ctx: Dict) -> List[Dict]:
body: List[Dict] = []
for child in node.children:
body.extend(_render_block(child, api, ctx))
return body
def _render_list_items(list_node: SyntaxTreeNode, api, ctx: Dict) -> List[Dict]:
items = []
for li in list_node.children:
content: List[Dict] = []
for child in li.children:
content.extend(_render_block(child, api, ctx))
items.append({"type": NodeType.LIST_ITEM, "content": content})
return items
def _footnote_definitions(tree: SyntaxTreeNode, api) -> Dict[int, List[Dict]]:
"""Map each footnote id to its rendered block content."""
definitions: Dict[int, List[Dict]] = {}
for node in tree.children:
if node.type != "footnote_block":
continue
for fn in node.children:
# A footnote's own content should not register anchors of its own.
local_ctx = {"order": []}
content: List[Dict] = []
for child in fn.children:
content.extend(_render_block(child, api, local_ctx))
definitions[fn.meta["id"]] = content
return definitions
def markdown_to_doc(markdown_content: str, api=None) -> List[Dict]:
"""Convert Markdown into a list of Substack ProseMirror block nodes."""
tree = SyntaxTreeNode(_make_parser().parse(markdown_content))
definitions = _footnote_definitions(tree, api)
ctx: Dict = {"order": []}
out: List[Dict] = []
for node in tree.children:
if node.type == "footnote_block":
continue
if node.type == "html_block":
attrs = parse_image_marker(node.content)
if attrs is not None:
if not out or out[-1].get("type") != "captionedImage":
raise ValueError("Image marker must immediately follow an image")
out[-1]["content"][0]["attrs"].update(attrs)
out[-1]["content"][0]["attrs"]["isProcessing"] = False
continue
out.extend(_render_block(node, api, ctx))
# Emit one footnote block per reference, in anchor order, numbered to match.
for number, footnote_id in enumerate(ctx["order"], start=1):
content = copy.deepcopy(definitions.get(footnote_id, []))
out.append(nodes.footnote(number, content))
return out