diff --git a/apps/sim/lib/execution/code-placeholders/javascript.ts b/apps/sim/lib/execution/code-placeholders/javascript.ts index 4f4ce0bf91b..36cdb05f564 100644 --- a/apps/sim/lib/execution/code-placeholders/javascript.ts +++ b/apps/sim/lib/execution/code-placeholders/javascript.ts @@ -4,6 +4,7 @@ import { CodePlaceholderCompileError, createCodePlaceholderCompilationContext, createOffsetRangeLookup, + partitionPoint, type SourceEdit, } from '@/lib/execution/code-placeholders/shared' import type { @@ -762,12 +763,19 @@ export async function compileJavaScriptPlaceholders( if (ts.isTemplateExpression(node.template)) { for (const span of node.template.templateSpans) visit(span.expression) } - const nestedEdits = edits.splice(nestedEditStart) + // A stable sort: applySourceEdits keeps input order among edits with equal bounds. + const nestedEdits = edits + .splice(nestedEditStart) + .sort((left, right) => left.start - right.start) const transformedNodeText = (child: ts.Node): string => { const childStart = child.getStart(sourceFile) const childEnd = child.getEnd() const childEdits = nestedEdits - .filter((edit) => edit.start >= childStart && edit.end <= childEnd) + .slice( + partitionPoint(nestedEdits, (edit) => edit.start < childStart), + partitionPoint(nestedEdits, (edit) => edit.start <= childEnd) + ) + .filter((edit) => edit.end <= childEnd) .map((edit) => ({ ...edit, start: edit.start - childStart, diff --git a/apps/sim/lib/execution/code-placeholders/python.ts b/apps/sim/lib/execution/code-placeholders/python.ts index f77f5c7f89d..4bc25224c39 100644 --- a/apps/sim/lib/execution/code-placeholders/python.ts +++ b/apps/sim/lib/execution/code-placeholders/python.ts @@ -43,23 +43,6 @@ function findContainingToken( return token && occurrence.end <= token.end ? token : undefined } -function findLastPythonKeyword( - value: string, - keyword: string, - isIgnored: (index: number) => boolean = () => false -): number { - let index = value.lastIndexOf(keyword) - while (index >= 0) { - const before = value[index - 1] - const after = value[index + keyword.length] - if (!isIdentifierCharacter(before) && !isIdentifierCharacter(after) && !isIgnored(index)) { - return index - } - index = index === 0 ? -1 : value.lastIndexOf(keyword, index - 1) - } - return -1 -} - function readPythonStringStart( code: string, index: number @@ -558,71 +541,358 @@ function pythonRuntimeValue(bindingName: string): string { type PythonBarePlaceholderPosition = 'value' | 'attribute' | 'unsupported-name' -function classifyPythonBarePlaceholder( +const ASSIGNMENT_OPERATOR = /^(?:=(?!=)|:=|\+=|-=|\*=|\/=|\/\/=|%=|@=|&=|\|=|\^=|>>=|<<=|\*\*=)/ +const NAME_POSITION_KEYWORDS = [ + 'def', + 'class', + 'import', + 'from', + 'as', + 'global', + 'nonlocal', + 'del', + 'for', + 'lambda', +] as const + +function isWhitespace(character: string | undefined): boolean { + return character !== undefined && /\s/.test(character) +} + +interface PythonKeywordHead { + start: number + /** End of the whitespace run that follows the keyword. */ + whitespaceEnd: number + /** First carriage return at or after `whitespaceEnd`, or `Infinity`. */ + carriageReturnAfter: number +} + +interface PythonSourceIndex { + newlines: number[] + carriageReturns: number[] + openParens: number[] + closeParens: number[] + commas: number[] + equals: number[] + colons: number[] + /** `(` offsets that close a `def name(` head. */ + defParens: number[] + forHeads: PythonKeywordHead[] + delHeads: PythonKeywordHead[] + /** `in` tokens with whitespace on both sides. */ + inTokens: number[] + /** `lambda` and `if` offsets not adjoining an identifier character. */ + lambdas: number[] + ifs: number[] +} + +function firstAtOrAfter(offsets: readonly number[], offset: number): number { + return partitionPoint(offsets, (candidate) => candidate < offset) +} + +function hasOffsetIn(offsets: readonly number[], start: number, end: number): boolean { + const index = firstAtOrAfter(offsets, start) + return index < offsets.length && offsets[index] < end +} + +/** The last offset in `[start, end)`, or `start - 1` when there is none. */ +function lastOffsetIn(offsets: readonly number[], start: number, end: number): number { + const offset = offsets[firstAtOrAfter(offsets, end) - 1] + return offset !== undefined && offset >= start ? offset : start - 1 +} + +function nextOffset(offsets: readonly number[], offset: number, fallback: number): number { + return offsets[firstAtOrAfter(offsets, offset)] ?? fallback +} + +function lineStartOf(newlines: readonly number[], offset: number): number { + return lastOffsetIn(newlines, 0, offset) + 1 +} + +function standaloneKeywordOffsets(code: string, keyword: string): number[] { + const offsets: number[] = [] + for (let index = code.indexOf(keyword); index !== -1; index = code.indexOf(keyword, index + 1)) { + if ( + !isIdentifierCharacter(code[index - 1]) && + !isIdentifierCharacter(code[index + keyword.length]) + ) { + offsets.push(index) + } + } + return offsets +} + +function keywordHeads( code: string, - occurrence: CodePlaceholderOccurrence, - isIgnored: (offset: number) => boolean -): PythonBarePlaceholderPosition { - const immediatelyPrevious = code[occurrence.start - 1] - const immediatelyNext = code[occurrence.end] - let previousIndex = occurrence.start - 1 - while (previousIndex >= 0 && /[ \t\f]/.test(code[previousIndex])) previousIndex -= 1 - const previous = code[previousIndex] - const attribute = previous === '.' - if (isIdentifierCharacter(immediatelyPrevious) || isIdentifierCharacter(immediatelyNext)) { - return 'unsupported-name' + keyword: string, + carriageReturns: number[] +): PythonKeywordHead[] { + const heads: PythonKeywordHead[] = [] + for (let index = code.indexOf(keyword); index !== -1; index = code.indexOf(keyword, index + 1)) { + if (index > 0 && !isWhitespace(code[index - 1])) continue + let whitespaceEnd = index + keyword.length + if (!isWhitespace(code[whitespaceEnd])) continue + while (isWhitespace(code[whitespaceEnd])) whitespaceEnd += 1 + heads.push({ + start: index, + whitespaceEnd, + carriageReturnAfter: nextOffset(carriageReturns, whitespaceEnd, Number.POSITIVE_INFINITY), + }) } + return heads +} + +/** Whether the `(` at `paren` closes `def name(`, scanning back no further than `lineStart`. */ +function closesDefHead(code: string, paren: number, lineStart: number): boolean { + let index = paren - 1 + while (index >= lineStart && isWhitespace(code[index])) index -= 1 + const identifierEnd = index + 1 + while (index >= lineStart && isIdentifierCharacter(code[index])) index -= 1 + const identifierStart = index + 1 + if (identifierStart === identifierEnd || !/[A-Za-z_]/.test(code[identifierStart])) return false + if (index < lineStart || !isWhitespace(code[index])) return false + while (index >= lineStart && isWhitespace(code[index])) index -= 1 + const defStart = index - 2 + if (defStart < lineStart || !code.startsWith('def', defStart)) return false + return defStart === lineStart || isWhitespace(code[defStart - 1]) +} + +function indexPythonSource(code: string): PythonSourceIndex { + const newlines: number[] = [] + const carriageReturns: number[] = [] + const openParens: number[] = [] + const closeParens: number[] = [] + const commas: number[] = [] + const equals: number[] = [] + const colons: number[] = [] + for (let index = 0; index < code.length; index += 1) { + const character = code[index] + if (character === '\n') newlines.push(index) + else if (character === '\r') carriageReturns.push(index) + else if (character === '(') openParens.push(index) + else if (character === ')') closeParens.push(index) + else if (character === ',') commas.push(index) + else if (character === '=') equals.push(index) + else if (character === ':') colons.push(index) + } + return { + newlines, + carriageReturns, + openParens, + closeParens, + commas, + equals, + colons, + defParens: openParens.filter((paren) => + closesDefHead(code, paren, lineStartOf(newlines, paren)) + ), + forHeads: keywordHeads(code, 'for', carriageReturns), + delHeads: keywordHeads(code, 'del', carriageReturns), + inTokens: standaloneKeywordOffsets(code, 'in').filter( + (offset) => isWhitespace(code[offset - 1]) && isWhitespace(code[offset + 2]) + ), + lambdas: standaloneKeywordOffsets(code, 'lambda'), + ifs: standaloneKeywordOffsets(code, 'if'), + } +} - const lineStart = code.lastIndexOf('\n', occurrence.start - 1) + 1 - const nextNewline = code.indexOf('\n', occurrence.end) - const lineEnd = nextNewline === -1 ? code.length : nextNewline - const before = code.slice(lineStart, occurrence.start) - const after = code.slice(occurrence.end, lineEnd) - const isIgnoredBeforeOffset = (offset: number): boolean => isIgnored(lineStart + offset) - const parameterSegment = before.slice( - Math.max(before.lastIndexOf('('), before.lastIndexOf(',')) + 1 - ) - const inFunctionParameterName = - /(?:^|\s)(?:async\s+)?def\s+[A-Za-z_][A-Za-z0-9_]*\s*\([^)]*$/.test(before) && - !/[=:]/.test(parameterSegment) - const lambdaStart = findLastPythonKeyword(before, 'lambda', isIgnoredBeforeOffset) - const inLambdaParameterName = - lambdaStart >= 0 && - before.indexOf(':', lambdaStart) === -1 && - !before - .slice(Math.max(lambdaStart + 'lambda'.length, before.lastIndexOf(',') + 1)) - .includes('=') - const forMatch = /(?:^|\s)(?:async\s+)?for\s+([^\r\n]*)$/.exec(before) - const inForTarget = forMatch !== null && !/(?:^|\s)in\s/.test(forMatch[1]) - const followedByAssignment = - /^\s*(?:=(?!=)|:=|\+=|-=|\*=|\/=|\/\/=|%=|@=|&=|\|=|\^=|>>=|<<=|\*\*=)/.test(after) - const caseClause = /^\s*case\s+([^\r\n]*)$/.exec(before) - const casePrefix = caseClause?.[1] ?? '' - const caseStartsWithAssignment = - /^(?:=(?!=)|:=|\+=|-=|\*=|\/=|\/\/=|%=|@=|&=|\|=|\^=|>>=|<<=|\*\*=)/.test(casePrefix) - const casePrefixOffset = before.length - casePrefix.length - const inCasePattern = - caseClause !== null && - !caseStartsWithAssignment && - findLastPythonKeyword(casePrefix, 'if', (offset) => - isIgnoredBeforeOffset(casePrefixOffset + offset) - ) === -1 && - after.includes(':') - if ( - /(?:^|\s)(?:async\s+)?(?:def|class|import|from|as|global|nonlocal|del)\s*$/.test(before) || - /(?:^|\s)(?:async\s+)?for\s*$/.test(before) || - /(?:^|\s)(?:as|lambda)\s*$/.test(before) || - inFunctionParameterName || - inLambdaParameterName || - inForTarget || - inCasePattern || - followedByAssignment || - (attribute && /(?:^|\s)del\s+[^\r\n]*\.\s*$/.test(before)) - ) { - return 'unsupported-name' +/** + * The first head starting at or after `lineStart` whose text up to `end` has no carriage + * return past its whitespace. Heads are disjoint, so `carriageReturnAfter` only grows along + * the list and the qualifying heads form a suffix. + */ +function firstHeadClearOfCarriageReturn( + heads: readonly PythonKeywordHead[], + lineStart: number, + end: number +): PythonKeywordHead | undefined { + return heads[ + Math.max( + partitionPoint(heads, (head) => head.start < lineStart), + partitionPoint(heads, (head) => head.carriageReturnAfter < end) + ) + ] +} + +/** + * Classifies a bare placeholder by the Python construct around it on its line, answering each + * rule from one sorted offset index of the file instead of a regex over the line prefix. + * `isIgnored` must be a pure membership test over the whole file: keyword lookbacks memoize + * their walk per predicate and may probe offsets on earlier lines. + */ +function createPythonBarePlaceholderClassifier(code: string) { + let sourceIndex: PythonSourceIndex | undefined + const caseClauses = new Map< + number, + { keywordStart: number; prefixStart: number; carriageReturnAfter: number } | null + >() + const keywordMemos = new WeakMap< + (offset: number) => boolean, + { lambda: Map; if: Map } + >() + + /** The last keyword offset at or before `maxStart` that `isIgnored` does not exclude, or -1. */ + const lastUnignored = ( + offsets: readonly number[], + memo: Map, + isIgnored: (offset: number) => boolean, + maxStart: number + ): number => { + const visited: number[] = [] + let found = -1 + for ( + let index = partitionPoint(offsets, (offset) => offset <= maxStart) - 1; + index >= 0; + index -= 1 + ) { + const known = memo.get(index) + if (known !== undefined) { + found = known + break + } + visited.push(index) + if (!isIgnored(offsets[index])) { + found = index + break + } + } + for (const index of visited) memo.set(index, found) + return found < 0 ? -1 : offsets[found] + } + + const caseClauseAt = (source: PythonSourceIndex, lineStart: number) => { + let clause = caseClauses.get(lineStart) + if (clause === undefined) { + let keywordStart = lineStart + while (isWhitespace(code[keywordStart])) keywordStart += 1 + clause = null + if (code.startsWith('case', keywordStart) && isWhitespace(code[keywordStart + 4])) { + let prefixStart = keywordStart + 4 + while (isWhitespace(code[prefixStart])) prefixStart += 1 + clause = { + keywordStart, + prefixStart, + carriageReturnAfter: nextOffset( + source.carriageReturns, + prefixStart, + Number.POSITIVE_INFINITY + ), + } + } + caseClauses.set(lineStart, clause) + } + return clause + } + + return ( + occurrence: CodePlaceholderOccurrence, + isIgnored: (offset: number) => boolean + ): PythonBarePlaceholderPosition => { + const { start, end } = occurrence + const immediatelyPrevious = code[start - 1] + const immediatelyNext = code[end] + let previousIndex = start - 1 + while (previousIndex >= 0 && /[ \t\f]/.test(code[previousIndex])) previousIndex -= 1 + const attribute = code[previousIndex] === '.' + if (isIdentifierCharacter(immediatelyPrevious) || isIdentifierCharacter(immediatelyNext)) { + return 'unsupported-name' + } + + const source = (sourceIndex ??= indexPythonSource(code)) + let memo = keywordMemos.get(isIgnored) + if (!memo) { + memo = { lambda: new Map(), if: new Map() } + keywordMemos.set(isIgnored, memo) + } + const lineStart = lineStartOf(source.newlines, start) + const lineEnd = nextOffset(source.newlines, end, code.length) + + let trailingWhitespace = start + while (trailingWhitespace > lineStart && isWhitespace(code[trailingWhitespace - 1])) { + trailingWhitespace -= 1 + } + const endsWithKeyword = (keyword: string) => { + const keywordStart = trailingWhitespace - keyword.length + return ( + keywordStart >= lineStart && + code.startsWith(keyword, keywordStart) && + (keywordStart === lineStart || isWhitespace(code[keywordStart - 1])) + ) + } + + const defParen = lastOffsetIn(source.defParens, lineStart, start) + const parameterStart = + Math.max( + lastOffsetIn(source.openParens, lineStart, start), + lastOffsetIn(source.commas, lineStart, start) + ) + 1 + const inFunctionParameterName = + defParen >= lineStart && + !hasOffsetIn(source.closeParens, defParen + 1, start) && + !hasOffsetIn(source.equals, parameterStart, start) && + !hasOffsetIn(source.colons, parameterStart, start) + + const lambdaStart = lastUnignored( + source.lambdas, + memo.lambda, + isIgnored, + start - 'lambda'.length + ) + const inLambdaParameterName = + lambdaStart >= lineStart && + !hasOffsetIn(source.colons, lambdaStart, start) && + !hasOffsetIn( + source.equals, + Math.max(lambdaStart + 'lambda'.length, lastOffsetIn(source.commas, lineStart, start) + 1), + start + ) + + const forHead = firstHeadClearOfCarriageReturn(source.forHeads, lineStart, start) + const inForTarget = + forHead !== undefined && + forHead.start <= start - 4 && + !hasOffsetIn(source.inTokens, forHead.whitespaceEnd, start - 2) + + let assignmentStart = end + while (assignmentStart < lineEnd && isWhitespace(code[assignmentStart])) assignmentStart += 1 + const followedByAssignment = ASSIGNMENT_OPERATOR.test( + code.slice(assignmentStart, Math.min(lineEnd, assignmentStart + 3)) + ) + + const caseClause = caseClauseAt(source, lineStart) + const inCaseClause = + caseClause !== null && + caseClause.keywordStart + 4 < start && + caseClause.carriageReturnAfter >= start + const inCasePattern = + inCaseClause && + !ASSIGNMENT_OPERATOR.test( + code.slice(caseClause.prefixStart, Math.min(start, caseClause.prefixStart + 3)) + ) && + lastUnignored(source.ifs, memo.if, isIgnored, start - 'if'.length) < caseClause.prefixStart && + hasOffsetIn(source.colons, end, lineEnd) + + const deletesAttribute = () => { + const dot = trailingWhitespace - 1 + if (dot < lineStart || code[dot] !== '.') return false + const delHead = firstHeadClearOfCarriageReturn(source.delHeads, lineStart, dot) + return delHead !== undefined && delHead.start <= dot - 4 + } + + if ( + NAME_POSITION_KEYWORDS.some(endsWithKeyword) || + inFunctionParameterName || + inLambdaParameterName || + inForTarget || + inCasePattern || + followedByAssignment || + (attribute && deletesAttribute()) + ) { + return 'unsupported-name' + } + if (attribute) return 'attribute' + return 'value' } - if (attribute) return 'attribute' - return 'value' } /** @@ -710,6 +980,7 @@ export async function compilePythonPlaceholders( } const lexed = lexPython(input.code) + const classifyBarePlaceholder = createPythonBarePlaceholderClassifier(input.code) recordPythonDirectEnvironmentReads(input.code, () => lexed, context) const edits: SourceEdit[] = [] const consumed = new Set() @@ -806,8 +1077,7 @@ export async function compilePythonPlaceholders( if (fStringPosition === 'expression' || fStringPosition === 'format') { if ( fStringPosition === 'expression' && - classifyPythonBarePlaceholder(input.code, occurrence, isInNestedString) === - 'unsupported-name' + classifyBarePlaceholder(occurrence, isInNestedString) === 'unsupported-name' ) { if (input.analysisOnly) { context.resolveValue(occurrence) @@ -884,7 +1154,7 @@ export async function compilePythonPlaceholders( if (consumed.has(occurrence) || isInComment(occurrence.start)) continue if (findContainingToken(lexed.strings, occurrence)) continue if (!context.hasValue(occurrence.name)) continue - const position = classifyPythonBarePlaceholder(input.code, occurrence, isInStringOrComment) + const position = classifyBarePlaceholder(occurrence, isInStringOrComment) if (position === 'unsupported-name') { if (input.analysisOnly) { context.resolveValue(occurrence)