Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
75f6e15
fix(docs): pin the generator's sort locale to en-US
waleedlatif1 Aug 28, 2026
c647676
fix(vanta): remove the MIME Type field whose value was always discarded
waleedlatif1 Aug 28, 2026
3604c23
fix(github): resolve the PR head SHA for file comments
waleedlatif1 Aug 28, 2026
77cece9
fix(google-drive): expose the page token so pagination is reachable
waleedlatif1 Aug 28, 2026
a9b476e
fix(confluence): stop documenting a cloudId users cannot supply
waleedlatif1 Aug 28, 2026
83f5cd9
fix(docs): teach the source scanner about regex literals
waleedlatif1 Aug 28, 2026
a4d9a73
Merge remote-tracking branch 'origin/staging' into fix/pre-existing-i…
waleedlatif1 Aug 28, 2026
83b16ae
fix(github): gate the commit lookup on path, coerce line, name the fa…
waleedlatif1 Aug 28, 2026
5d74dcb
test(github): cover the comment routing cases the gate changed
waleedlatif1 Aug 28, 2026
0e790bf
fix(google-drive): let an agent feed the page token back in
waleedlatif1 Aug 28, 2026
539e1ab
docs(generator): name the load-bearing newline rule and report an uns…
waleedlatif1 Aug 28, 2026
00ff23b
fix(vanta): declare the removed uploadMimeType subblock as dropped
waleedlatif1 Aug 28, 2026
a181906
fix(github): run the two-phase PR comment on the secure transport
waleedlatif1 Aug 28, 2026
06bf99f
fix(vanta): keep mimeType an ordinary upload parameter
waleedlatif1 Aug 28, 2026
cc5a264
fix(generator): lex regex-in-keyword-position and template interpolation
waleedlatif1 Aug 28, 2026
1403919
fix(github): stop forwarding the GitHub token across a redirect origin
waleedlatif1 Aug 28, 2026
68035f6
fix(github): reject a fractional comment line instead of truncating it
waleedlatif1 Aug 28, 2026
ae542d9
fix(github): select the comment endpoint by comment type, not by path
waleedlatif1 Aug 28, 2026
e2b4eae
test(confluence): drop the cloudId visibility invariant test
waleedlatif1 Aug 28, 2026
1f06127
fix(github): send an explicit User-Agent and stop downgrading a redir…
waleedlatif1 Aug 28, 2026
ec43e24
fix(vanta): drop the dead whenOperation from the removed-subblock entry
waleedlatif1 Aug 28, 2026
36d0488
fix(docs-gen): close a regex-vs-division gap and make three guards te…
waleedlatif1 Aug 28, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Prev Previous commit
Next Next commit
fix(docs): teach the source scanner about regex literals
blankStringsAndComments was a single regex with no concept of a regex
literal, so two shapes silently truncated a block's subBlock list:

  /don't/   the apostrophe opened a phantom string that swallowed the
            following entries
  /[}]/     the brace in the character class closed the enclosing object

Both returned a short list with no warning — a confident wrong answer,
which for the hidden-param filter means silently deleting a user-settable
row. No block file uses a regex literal today, so this was latent.

Replaces the regex with a linear scanner that distinguishes a regex
literal from a division by the previous significant character, blanks
regex bodies whole (their last character is arbitrary source, same reason
comments are blanked whole), and tracks ${} nesting so a backtick inside
a template expression cannot end the template early.

The scanner now returns null when it ends inside an unterminated
construct. All three call sites treat that as UNKNOWN rather than
guessing, so the filter switches off instead of stripping.

Generated artifacts are byte-identical and the warning count is unchanged.
  • Loading branch information
waleedlatif1 committed Aug 28, 2026
commit 83f5cd96a860603ee031a02645fd0f7121b1ce05
40 changes: 40 additions & 0 deletions scripts/generate-docs.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -721,3 +721,43 @@ describe('the generated catalog ordering is locale-independent', () => {
expect(source).not.toMatch(/localeCompare\(\s*[A-Za-z_$][\w$.]*\s*\)/)
})
})

describe('the scanner survives regex literals in a block config', () => {
/**
* `blankStringsAndComments` used to be a single regex with no concept of a regex literal, so
* `/don't/` opened a phantom string that swallowed the following subBlocks, and a character
* class like `/[}]/` closed the enclosing object early. Both returned a short list with no
* warning — a confident wrong answer, which is the one outcome the filter must never produce.
*/
it('does not let an apostrophe inside a regex swallow later subBlocks', () => {
const ids = extractUserSettableParamIds(
"subBlocks: [{ id: 'a', condition: (v) => /don't/.test(v) }, { id: 'b' }],"
)

expect(ids).toEqual(['a', 'b'])
})

it('does not let a brace inside a character class close the object early', () => {
const ids = extractUserSettableParamIds("subBlocks: [{ id: 'a', v: /[}]/ }, { id: 'b' }],")

expect(ids).toEqual(['a', 'b'])
})

it('still reads a division as arithmetic rather than a regex', () => {
const ids = extractUserSettableParamIds('subBlocks: [{ id: "a", n: total / 2 }, { id: "b" }],')

expect(ids).toEqual(['a', 'b'])
})

it('does not mistake a protocol slash inside a string for a comment', () => {
const ids = extractUserSettableParamIds(
"subBlocks: [{ id: 'a', url: 'https://example.com/x' }, { id: 'b' }],"
)

expect(ids).toEqual(['a', 'b'])
})

it('reports UNKNOWN rather than guessing when a literal never terminates', () => {
expect(extractUserSettableParamIds("subBlocks: [{ id: 'a }],")).toBeNull()
})
})
168 changes: 150 additions & 18 deletions scripts/generate-docs.ts
Original file line number Diff line number Diff line change
Expand Up @@ -714,6 +714,7 @@ export function extractUserSettableParamIds(
blockName = 'block'
): string[] | null {
const scannable = blankStringsAndComments(blockContent)
if (scannable === null) return null
const keyMatch = /\bsubBlocks\s*:/.exec(scannable)
if (!keyMatch) return []

Expand Down Expand Up @@ -997,6 +998,7 @@ function collectShorthandPropertyNames(body: string, into: Set<string>): void {
*/
export function extractMapperWrittenParamIds(blockContent: string): string[] {
const scannable = blankStringsAndComments(blockContent)
if (scannable === null) return []
const ids = new Set<string>()

for (const [start, end] of findMapperBodyRanges(scannable)) {
Expand Down Expand Up @@ -1253,27 +1255,156 @@ function extractAuthType(blockContent: string): 'oauth' | 'api-key' | 'none' {
return 'none'
}

/** Characters after which a `/` begins a regex literal rather than a division. */
const REGEX_ALLOWED_AFTER = new Set([
'(',
',',
'=',
':',
'[',
'!',
'&',
'|',
'?',
'{',
'}',
Comment thread
waleedlatif1 marked this conversation as resolved.
';',
'+',
'-',
'*',
'%',
'~',
'^',
'<',
'>',
'\n',
])

/**
* Length-preserving copy of `content` with string-literal and comment
* interiors blanked out, so delimiter scans cannot be tripped by braces or
* quotes inside them. Indices into the result line up with indices into
* `content`.
* Blank out string literals, template literals, comments and regex literals so a structural
* scan sees only code punctuation. Length and newlines are preserved, which the `readLiteral`
* index-mapping call sites depend on.
*
* Quoted strings keep their delimiters so callers can still see where one began; comments and
* regex literals are blanked whole, because their final character is arbitrary source text —
* commented-out code ending in `[`, or a character class like `/[}]/`, otherwise leaves an
* unbalanced bracket that derails every downstream scan.
*
* Returns `null` when the scan ends inside an unterminated construct, which means the input
* was not what we assumed and no structural conclusion drawn from it can be trusted.
*/
function blankStringsAndComments(content: string): string {
return content.replace(
/(['"`])(?:\\[\s\S]|(?!\1)[^\\])*\1|\/\/[^\n]*|\/\*[\s\S]*?\*\//g,
(match: string, quote: string | undefined) => {
const blanked = match.replace(/[^\n]/g, ' ')
/**
* A comment has no delimiters worth preserving, so it is blanked whole. Keeping
* its final character would leak arbitrary source text — commented-out code ending
* in `[` or `{` leaves an unbalanced bracket that derails every scan downstream.
* A quoted string keeps its own quotes so callers can still see where it began.
*/
if (quote === undefined) return blanked
return quote + blanked.slice(1, -1) + quote
function blankStringsAndComments(content: string): string | null {
const out = content.split('')
const blank = (start: number, end: number) => {
for (let k = start; k < end && k < out.length; k++) if (out[k] !== '\n') out[k] = ' '
}

let i = 0
let prevSignificant = ''
while (i < content.length) {
const char = content[i]

if (char === '/' && content[i + 1] === '/') {
const nl = content.indexOf('\n', i)
const end = nl === -1 ? content.length : nl
blank(i, end)
i = end
continue
}
)

if (char === '/' && content[i + 1] === '*') {
const close = content.indexOf('*/', i + 2)
if (close === -1) return null
blank(i, close + 2)
i = close + 2
continue
}

if (char === '/' && (prevSignificant === '' || REGEX_ALLOWED_AFTER.has(prevSignificant))) {
Comment thread
cubic-dev-ai[bot] marked this conversation as resolved.
Outdated
let j = i + 1
let inClass = false
let closed = false
while (j < content.length) {
const c = content[j]
if (c === '\\') {
j += 2
continue
}
if (c === '\n') break
if (c === '[') inClass = true
else if (c === ']') inClass = false
else if (c === '/' && !inClass) {
closed = true
break
}
j++
}
if (!closed) return null
j++
while (j < content.length && /[a-z]/.test(content[j])) j++
blank(i, j)
prevSignificant = ')'
i = j
continue
}

if (char === "'" || char === '"') {
let j = i + 1
let closed = false
while (j < content.length) {
if (content[j] === '\\') {
j += 2
continue
}
if (content[j] === '\n') break
if (content[j] === char) {
closed = true
break
}
j++
}
if (!closed) return null
blank(i + 1, j)
prevSignificant = char
i = j + 1
continue
}

if (char === '`') {
let j = i + 1
let depth = 0
let closed = false
while (j < content.length) {
if (content[j] === '\\') {
j += 2
continue
}
if (depth === 0 && content[j] === '`') {
closed = true
break
}
if (content[j] === '$' && content[j + 1] === '{') {
depth++
j += 2
continue
}
if (depth > 0 && content[j] === '{') depth++
Comment thread
cubic-dev-ai[bot] marked this conversation as resolved.
Outdated
else if (depth > 0 && content[j] === '}') depth--
j++
}
if (!closed) return null
blank(i + 1, j)
prevSignificant = '`'
i = j + 1
continue
}

if (!/\s/.test(char)) prevSignificant = char
else if (char === '\n') prevSignificant = '\n'
i++
}

return out.join('')
}

/**
Expand All @@ -1288,6 +1419,7 @@ function extractOAuthServiceId(blockContent: string): string | undefined {
if (!typeMatch) return undefined

const scannable = blankStringsAndComments(blockContent)
if (scannable === null) return undefined
let depth = 0
let objectStart = -1
for (let i = typeMatch.index; i >= 0; i--) {
Expand Down