Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Keep a rewritten script's original src in __wb_orig_src
Bundlers identify a chunk by the raw getAttribute("src") rather than the
.src property, and they expect the string the server sent. Rewriting an
absolute src to a relative one therefore breaks them silently: Next.js
with Turbopack strips a leading "/_next/" to derive a chunk's name, the
strip stops matching once the leading slash is gone, every chunk
registers under a name nothing is waiting for, and the page never
hydrates. Vite is reported to be affected the same way.

wombat cannot repair this on its own. Its getAttribute override
un-rewrites URLs that wombat rewrote, and this one was rewritten here,
at scrape time, so extractOriginalURL finds no prefix it knows and hands
the value straight back. What it does have is __wb_orig_src, which it
checks first for script elements (retrieveWBOSRC) and which only the
rewriter can fill in.

Nothing is added when the tag is not a script, when it has no src, when
rewriting left the src alone, or when the attribute is already there —
that last one being HTML rewritten twice, where the attribute already
present holds the true original and a second would both duplicate it and
win, since a browser takes the first.

Measured on a warc2zim capture of draculatheme.com: the replayed page is
900px before and 19851px after, against 19858px live.

Fixes openzim/warc2zim#473

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
  • Loading branch information
epheterson and claude committed Sep 12, 2026
commit 469aad43888d93828a33b8e0c875ba6ce9b3e6a4
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
### Fixed

- Carry a script's top-level `const`, `let` and `class` declarations across the wombat block, so other scripts on the page can still see them (#329)
- Keep a rewritten `<script>`'s original `src` in `__wb_orig_src`, so wombat can return it from `getAttribute("src")` and bundlers that identify chunks by it (Next.js/Turbopack, Vite) still work (openzim/warc2zim#473)

## [5.4.1] - 2026-07-31

Expand Down
92 changes: 71 additions & 21 deletions src/zimscraperlib/rewriting/html.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,57 @@ def get_attr_value_from(
return default


# The attribute wombat reads a script's pre-rewrite src back out of.
#
# wombat already overrides Element.prototype.getAttribute and, for a <script>,
# checks this attribute before anything else (retrieveWBOSRC). It exists for
# exactly this situation and nothing was filling it in.
WB_ORIG_SRC_ATTRIBUTE = "__wb_orig_src"


def get_script_original_src(
tag: str, attrs: AttrsList, rewritten_attrs: list[AttrNameAndValue]
) -> str | None:
"""The src a <script> had before rewriting, when rewriting changed it.

Bundlers identify their chunks by the raw ``getAttribute("src")``, not by
the ``.src`` property, and they expect the string the server sent. Rewriting
an absolute path to a relative one therefore breaks them silently: Next.js
with Turbopack strips a leading ``/_next/`` to get a chunk's name, the strip
no longer matches, every chunk registers under a name nothing is waiting
for, and the page never hydrates. Vite is reported to be affected the same
way.

wombat cannot repair this on its own. Its getAttribute override un-rewrites
URLs that WOMBAT rewrote, and this one was rewritten here, at scrape time —
so extractOriginalURL sees no prefix it knows and hands the value straight
back. What it does have is ``__wb_orig_src``, which it checks first for
script elements and which only the rewriter can fill in.

None when the tag is not a script, when it had no src, when rewriting left
the src alone, or when the tag already carries the attribute — there is
nothing to remember in those cases, and an attribute that merely repeats
the src is noise in every captured page.

The already-carries case is HTML being rewritten a second time. What is
recorded then is the src of the FIRST pass, which is already a rewritten
value; keeping the attribute that is there preserves the true original, and
appending a second one would both duplicate the attribute and let the stale
value win, since a browser takes the first.
"""
if tag != "script":
return None
if get_attr_value_from(attrs, WB_ORIG_SRC_ATTRIBUTE) is not None:
return None
original = get_attr_value_from(attrs, "src")
if not original:
return None
rewritten = get_attr_value_from(rewritten_attrs, "src")
if rewritten is None or rewritten == original:
return None
return original


def format_attr(name: str, value: str | None) -> str:
"""Format a given attribute name and value, properly escaping the value"""
if value is None:
Expand Down Expand Up @@ -195,28 +246,27 @@ def handle_starttag(self, tag: str, attrs: AttrsList, *, auto_close: bool = Fals
self.send(f"<{tag}")
if attrs:
self.send(" ")
self.send(
" ".join(
format_attr(*attr)
for attr in (
rules.do_attribute_rewrite(
tag=tag,
attr_name=attr_name,
attr_value=attr_value,
attrs=attrs,
js_rewriter=self.js_rewriter,
css_rewriter=self.css_rewriter,
url_rewriter=self.url_rewriter,
base_href=self.base_href,
notify_js_module=self.notify_js_module,
)
for attr_name, attr_value in attrs
if not rules.do_drop_attribute(
tag=tag, attr_name=attr_name, attr_value=attr_value, attrs=attrs
)
)
rewritten_attrs = [
rules.do_attribute_rewrite(
tag=tag,
attr_name=attr_name,
attr_value=attr_value,
attrs=attrs,
js_rewriter=self.js_rewriter,
css_rewriter=self.css_rewriter,
url_rewriter=self.url_rewriter,
base_href=self.base_href,
notify_js_module=self.notify_js_module,
)
)
for attr_name, attr_value in attrs
if not rules.do_drop_attribute(
tag=tag, attr_name=attr_name, attr_value=attr_value, attrs=attrs
)
]
original_src = get_script_original_src(tag, attrs, rewritten_attrs)
if original_src is not None:
rewritten_attrs.append((WB_ORIG_SRC_ATTRIBUTE, original_src))
self.send(" ".join(format_attr(*attr) for attr in rewritten_attrs))

if auto_close:
self.send(" />")
Expand Down
71 changes: 69 additions & 2 deletions tests/rewriting/test_html_rewriting.py
Original file line number Diff line number Diff line change
Expand Up @@ -221,7 +221,9 @@ def test_escaped_content(escaped_content: ContentForTests):
'8Q7RZcHPHksttq7/GFoxjCVUjkjvPdw=="'
' crossorigin="anonymous" referrerpolicy="no-referrer"></script>',
'<script src="../cdnjs.cloudflare.com/ajax/libs/jquery/3.7.0/jquery.min.js"'
' crossorigin="anonymous" referrerpolicy="no-referrer"></script>',
' crossorigin="anonymous" referrerpolicy="no-referrer"'
' __wb_orig_src="https://cdnjs.cloudflare.com/ajax/libs/jquery/3.7.0/'
'jquery.min.js"></script>',
),
ContentForTests(
'<link rel="preload" src="https://cdnjs.cloudflare.com/jquery.min.js"'
Expand Down Expand Up @@ -711,7 +713,8 @@ def test_extract_base_href(html_content: str, expected_base_href: str):
ContentForTests(
'<html><head><base href="../"></head>'
'<body><script src="foo.js"></script></body></html>',
'<html><head></head><body><script src="../foo.js"></script></body></html>',
"<html><head></head><body>"
'<script src="../foo.js" __wb_orig_src="foo.js"></script></body></html>',
"kiwix.org/a/index.html",
),
ContentForTests(
Expand Down Expand Up @@ -1589,3 +1592,67 @@ def test_rewrite_meta_http_equiv_redirect_rule(
)
== expected_result
)


@pytest.mark.parametrize(
"input_str, expected_str",
[
pytest.param(
'<script src="/_next/static/chunks/a.js"></script>',
'<script src="../_next/static/chunks/a.js"'
' __wb_orig_src="/_next/static/chunks/a.js"></script>',
id="root_relative_src_is_remembered",
),
pytest.param(
'<script src="https://kiwix.org/_next/a.js"></script>',
'<script src="../_next/a.js"'
' __wb_orig_src="https://kiwix.org/_next/a.js"></script>',
id="absolute_src_is_remembered",
),
pytest.param(
'<script src="a.js"></script>',
'<script src="a.js"></script>',
id="an_unchanged_src_adds_nothing",
),
pytest.param(
'<script>console.log("hi")</script>',
'<script>console.log("hi")</script>',
id="an_inline_script_adds_nothing",
),
pytest.param(
'<img src="/img/a.png">',
'<img src="../img/a.png">',
id="only_scripts_get_it",
),
pytest.param(
'<script src="/_next/a.js" __wb_orig_src="/original/a.js"></script>',
'<script src="../_next/a.js" __wb_orig_src="/original/a.js"></script>',
id="rewriting_twice_keeps_the_first_original",
),
],
)
def test_script_keeps_its_original_src(input_str: str, expected_str: str):
"""A rewritten <script> remembers the src the server sent, for wombat.

Bundlers identify a chunk by the raw ``getAttribute("src")`` rather than by
the ``.src`` property, and they expect the string the server sent. Next.js
with Turbopack strips a leading ``/_next/`` to get a chunk's name; once the
src has been made relative the strip no longer matches, every chunk
registers under a name nothing is waiting for, and the page never hydrates.

wombat cannot repair this by itself — its getAttribute override un-rewrites
URLs that wombat rewrote, and this one was rewritten here. But it already
reads ``__wb_orig_src`` on script elements first, and only the rewriter can
fill that in.
"""
assert (
HtmlRewriter(
ArticleUrlRewriter(article_url=HttpUrl("https://kiwix.org/a/index.html")),
None,
None,
None,
)
.rewrite(input_str)
.content
== expected_str
)