mirror of
https://github.com/stablyai/orca.git
synced 2026-10-08 00:02:38 +00:00
#14397 split `shared/types.ts` into 46 per-domain modules but kept the path as a re-export barrel so the import sites did not have to change. This removes the barrel: every consumer now imports from the module that actually declares the type, and `src/shared/types.ts` is deleted. Barrels hide where a type lives, make every consumer look like it depends on the whole domain, and let an unrelated edit invalidate a module that ~2,000 files transitively import. 2,323 import declarations across 2,321 files. Rewritten mechanically: each specifier was resolved to an absolute path via the TypeScript AST and recomputed, rather than string-substituted, so alias forms (`@/../../shared/ types`) and per-specifier `type` modifiers survive. Four cases the mechanical pass had to handle, each found by a gate rather than by reading the diff: - Modules inside `src/shared` import the barrel as `./types`, not `shared/types`. A pre-filter on the latter string skipped 176 of them and left imports dangling at a deleted file, which surfaced as confusing `Property 'x' is optional in type 'Repo' but required in Pick<Repo, ...>` errors rather than "module not found". - The barrel RENAMED one type on the way through (`WorkspaceSource as WorkspaceCreateTelemetrySource`), so the original name in the owning module has to be re-aliased at each consumer. - Three test files put `;(globalThis as ...)` on the line after the import. TypeScript parses that `;` as the import statement's terminator, so replacing through `statement.getEnd()` deletes it and breaks ASI. The rewrite now stops at the module specifier. - A file that already imported directly from a module got a SECOND import from it, because the barrel re-exported those same names — which trips `import/no-duplicates` under `--deny-warnings`. A post-pass merges declarations sharing a specifier and type-only-ness; the `import type` plus `import` pair from one module is left alone, since that form is allowed. Splitting one barrel import into several genuinely adds lines, which pushed `terminal-layout-pty-ownership.ts` to 301 counted lines: its 107-character import must wrap, and neither local type collapses onto one line (101 and 116 characters). Rather than contort a type declaration to fit a line budget, `collectLeafIds` and `pruneLeaves` move to `terminal-pane-layout-tree.ts` — they are pure structural operations on the layout tree and independent of PTY ownership. `visible-worktrees.ts` similarly loses its own mini-barrel re-export of `isDefaultBranchWorkspace`, with the four real consumers repointed at the declaring module. No `max-lines` bypass added. Verified: cold `tsc --noEmit` green on node, cli, and web (buildinfo deleted first — these projects are `composite: true` and reuse stale caches); the full `pnpm lint` green, not just bare oxlint — the narrower local check is what let the duplicate imports reach CI; max-lines ratchet OK at 344.
469 lines
15 KiB
TypeScript
469 lines
15 KiB
TypeScript
/* oxlint-disable max-lines -- Why: single source of truth for rg/git-grep arg
|
||
* construction and --json/submatch parsing shared by main process and SSH relay;
|
||
* re-splitting would re-introduce the maxBuffer divergence the design doc calls out. */
|
||
/**
|
||
* Shared, pure text-search helpers used by both the local main process and the
|
||
* SSH relay. No Electron, child_process, or fs — the caller owns process
|
||
* execution and transport-specific path translation (WSL).
|
||
*
|
||
* Centralizes rg/git-grep arg construction and parsing so the local and relay paths
|
||
* can't re-diverge (notably the relay's old execFile maxBuffer that dropped matches).
|
||
* Design doc: docs/design/share-text-search.md.
|
||
*/
|
||
import { posix, win32 } from 'node:path'
|
||
import { assertJsonTextStructureWithinLimits } from './json-text-structure-limit'
|
||
import { normalizeSearchResult } from './search-match-count'
|
||
import { escapeRegex } from './string-utils'
|
||
import type {
|
||
SearchFileResult,
|
||
SearchMatch,
|
||
SearchOptions,
|
||
SearchResult
|
||
} from './code-search-types'
|
||
|
||
export type SearchAccumulator = {
|
||
fileMap: Map<string, SearchFileResult>
|
||
totalMatches: number
|
||
truncated: boolean
|
||
}
|
||
|
||
export function createAccumulator(): SearchAccumulator {
|
||
return { fileMap: new Map(), totalMatches: 0, truncated: false }
|
||
}
|
||
|
||
function acceptMatch(fileResult: SearchFileResult): void {
|
||
fileResult.matchCount = (fileResult.matchCount ?? 0) + 1
|
||
}
|
||
|
||
// Why: normalize separators and strip leading `/` so results are cross-platform stable and don't break callers' `join(rootPath, relPath)`.
|
||
export function normalizeRelativePath(path: string): string {
|
||
return path.replace(/[\\/]+/g, '/').replace(/^\/+/, '')
|
||
}
|
||
|
||
function pathFlavor(rootPath: string): typeof posix | typeof win32 {
|
||
if (/^[a-zA-Z]:[\\/]/.test(rootPath) || rootPath.startsWith('\\\\')) {
|
||
return win32
|
||
}
|
||
return posix
|
||
}
|
||
|
||
function relativeToSearchRoot(rootPath: string, absPath: string): string {
|
||
return pathFlavor(rootPath).relative(rootPath, absPath)
|
||
}
|
||
|
||
function joinSearchRoot(rootPath: string, relPath: string): string {
|
||
return pathFlavor(rootPath).join(rootPath, relPath)
|
||
}
|
||
|
||
// ─── Constants shared by both callers ────────────────────────────────
|
||
|
||
export const MAX_MATCHES_PER_FILE = 100
|
||
export const DEFAULT_SEARCH_MAX_RESULTS = 2000
|
||
export const SEARCH_TIMEOUT_MS = 15_000
|
||
export const SEARCH_JSON_STRUCTURE_LIMITS = {
|
||
structuralTokens: 32 * 1024,
|
||
nestingDepth: 16
|
||
} as const
|
||
|
||
// Why: keep search cheaper than opening a file; the editor read path has a larger cap (Monaco large-file handling).
|
||
const SEARCH_MAX_FILE_SIZE = 5 * 1024 * 1024
|
||
|
||
// Why: mega-byte lines (minified/generated files) × 2000-match caps blow past the 16MB SSH relay MAX_MESSAGE_SIZE; clamp each match's context.
|
||
export const MAX_LINE_CONTENT_LENGTH = 500
|
||
const TRUNCATION_MARKER = '…'
|
||
|
||
function clampLineContext(
|
||
text: string,
|
||
matchStart: number,
|
||
matchLength: number
|
||
): {
|
||
lineContent: string
|
||
column: number
|
||
matchLength: number
|
||
displayColumn?: number
|
||
displayMatchLength?: number
|
||
} {
|
||
if (text.length <= MAX_LINE_CONTENT_LENGTH) {
|
||
return { lineContent: text, column: matchStart + 1, matchLength }
|
||
}
|
||
// Clamp the match first so a pathological multi-MB regex hit can't defeat the windowing below.
|
||
const clampedMatchLength = Math.min(matchLength, MAX_LINE_CONTENT_LENGTH)
|
||
const remaining = MAX_LINE_CONTENT_LENGTH - clampedMatchLength
|
||
const leftBudget = Math.floor(remaining / 2)
|
||
let windowStart = Math.max(0, matchStart - leftBudget)
|
||
let windowEnd = Math.min(text.length, windowStart + MAX_LINE_CONTENT_LENGTH)
|
||
windowStart = Math.max(0, windowEnd - MAX_LINE_CONTENT_LENGTH)
|
||
|
||
let snippet = text.slice(windowStart, windowEnd)
|
||
let column = matchStart - windowStart + 1
|
||
if (windowStart > 0) {
|
||
snippet = TRUNCATION_MARKER + snippet
|
||
column += TRUNCATION_MARKER.length
|
||
}
|
||
if (windowEnd < text.length) {
|
||
snippet = snippet + TRUNCATION_MARKER
|
||
}
|
||
return {
|
||
lineContent: snippet,
|
||
column: matchStart + 1,
|
||
matchLength,
|
||
displayColumn: column,
|
||
displayMatchLength: clampedMatchLength
|
||
}
|
||
}
|
||
|
||
// Why: shared by rg and git-grep to preserve the synchronous truncation ordering callers require.
|
||
function pushMatch(
|
||
fileResult: SearchFileResult,
|
||
acc: SearchAccumulator,
|
||
clamped: ReturnType<typeof clampLineContext>,
|
||
lineNumber: number,
|
||
maxResults: number
|
||
): 'continue' | 'stop' {
|
||
// Why: direct assignment avoids conditional-spread allocations on the per-match hot path.
|
||
const match: SearchMatch = {
|
||
line: lineNumber,
|
||
column: clamped.column,
|
||
matchLength: clamped.matchLength,
|
||
lineContent: clamped.lineContent
|
||
}
|
||
if (clamped.displayColumn !== undefined) {
|
||
match.displayColumn = clamped.displayColumn
|
||
}
|
||
if (clamped.displayMatchLength !== undefined) {
|
||
match.displayMatchLength = clamped.displayMatchLength
|
||
}
|
||
fileResult.matches.push(match)
|
||
acceptMatch(fileResult)
|
||
acc.totalMatches++
|
||
if (acc.totalMatches >= maxResults) {
|
||
acc.truncated = true
|
||
return 'stop'
|
||
}
|
||
return 'continue'
|
||
}
|
||
|
||
// ─── rg ─────────────────────────────────────────────────────────────
|
||
|
||
export type SearchOptionsLike = Pick<
|
||
SearchOptions,
|
||
'caseSensitive' | 'wholeWord' | 'useRegex' | 'includePattern' | 'excludePattern'
|
||
>
|
||
|
||
export function splitSearchGlobPatterns(patterns: string): string[] {
|
||
const out: string[] = []
|
||
let current = ''
|
||
let escaping = false
|
||
for (const ch of patterns) {
|
||
if (escaping) {
|
||
current += `\\${ch}`
|
||
escaping = false
|
||
continue
|
||
}
|
||
if (ch === '\\') {
|
||
escaping = true
|
||
continue
|
||
}
|
||
if (ch === ',') {
|
||
const trimmed = current.trim()
|
||
if (trimmed) {
|
||
out.push(trimmed)
|
||
}
|
||
current = ''
|
||
continue
|
||
}
|
||
current += ch
|
||
}
|
||
if (escaping) {
|
||
current += '\\'
|
||
}
|
||
const trimmed = current.trim()
|
||
if (trimmed) {
|
||
out.push(trimmed)
|
||
}
|
||
return out
|
||
}
|
||
|
||
/**
|
||
* Build the complete rg argv (flags + `--` + query + target) for both callers to spawn as-is.
|
||
*
|
||
* Constraint: pass `rootPath` unchanged as `target` — do NOT WSL-translate it; only the rg
|
||
* invocation is routed through `wslAwareSpawn`, and output paths are translated back in `ingestRgJsonLine`.
|
||
*/
|
||
export function buildRgArgs(query: string, target: string, opts: SearchOptionsLike): string[] {
|
||
const args: string[] = [
|
||
'--json',
|
||
'--hidden',
|
||
'--glob',
|
||
'!.git',
|
||
'--max-count',
|
||
String(MAX_MATCHES_PER_FILE),
|
||
'--max-filesize',
|
||
`${Math.floor(SEARCH_MAX_FILE_SIZE / 1024 / 1024)}M`
|
||
]
|
||
if (!opts.caseSensitive) {
|
||
args.push('--ignore-case')
|
||
}
|
||
if (opts.wholeWord) {
|
||
args.push('--word-regexp')
|
||
}
|
||
if (!opts.useRegex) {
|
||
args.push('--fixed-strings')
|
||
}
|
||
if (opts.includePattern) {
|
||
for (const pat of splitSearchGlobPatterns(opts.includePattern)) {
|
||
args.push('--glob', pat)
|
||
}
|
||
}
|
||
if (opts.excludePattern) {
|
||
for (const pat of splitSearchGlobPatterns(opts.excludePattern)) {
|
||
args.push('--glob', `!${pat}`)
|
||
}
|
||
}
|
||
args.push('--', query, target)
|
||
return args
|
||
}
|
||
|
||
/**
|
||
* Ingest a single line of rg `--json` stdout, mutating `acc`. Returns 'stop' when
|
||
* `maxResults` is reached (so the caller can kill the child), else 'continue'.
|
||
* `transformAbsPath` lets the local caller apply WSL translation; the relay passes none.
|
||
*
|
||
* Invariant: sets `acc.truncated = true` synchronously in the same tick it returns
|
||
* 'stop'; callers must not flip `truncated` or resolve before that tick (see design doc).
|
||
*/
|
||
export function ingestRgJsonLine(
|
||
line: string,
|
||
rootPath: string,
|
||
acc: SearchAccumulator,
|
||
maxResults: number,
|
||
transformAbsPath?: (p: string) => string
|
||
): 'continue' | 'stop' {
|
||
if (acc.totalMatches >= maxResults) {
|
||
return 'stop'
|
||
}
|
||
if (!line) {
|
||
return 'continue'
|
||
}
|
||
let msg: {
|
||
type?: string
|
||
data?: {
|
||
path?: { text?: string }
|
||
submatches?: { start: number; end: number }[]
|
||
line_number?: number
|
||
lines?: { text?: string }
|
||
}
|
||
}
|
||
try {
|
||
assertJsonTextStructureWithinLimits(line, SEARCH_JSON_STRUCTURE_LIMITS)
|
||
msg = JSON.parse(line)
|
||
} catch {
|
||
return 'continue'
|
||
}
|
||
if (msg.type !== 'match' || !msg.data) {
|
||
return 'continue'
|
||
}
|
||
const data = msg.data
|
||
const rawPath = data.path?.text
|
||
if (typeof rawPath !== 'string') {
|
||
return 'continue'
|
||
}
|
||
const absPath = transformAbsPath ? transformAbsPath(rawPath) : rawPath
|
||
const relPath = normalizeRelativePath(relativeToSearchRoot(rootPath, absPath))
|
||
const lineContent = (data.lines?.text ?? '').replace(/\n$/, '')
|
||
const lineNumber = data.line_number ?? 0
|
||
let submatches = data.submatches ?? []
|
||
if (submatches.length === 0) {
|
||
// Why: some rg matches report a line but no submatch ranges; surface a navigable line-level result instead of a count-0 row.
|
||
submatches = [{ start: 0, end: lineContent.length > 0 ? 1 : 0 }]
|
||
}
|
||
|
||
for (const sub of submatches) {
|
||
let fileResult = acc.fileMap.get(absPath)
|
||
if (!fileResult) {
|
||
fileResult = { filePath: absPath, relativePath: relPath, matches: [], matchCount: 0 }
|
||
acc.fileMap.set(absPath, fileResult)
|
||
}
|
||
const clamped = clampLineContext(lineContent, sub.start, sub.end - sub.start)
|
||
if (pushMatch(fileResult, acc, clamped, lineNumber, maxResults) === 'stop') {
|
||
return 'stop'
|
||
}
|
||
}
|
||
return 'continue'
|
||
}
|
||
|
||
// ─── git grep ───────────────────────────────────────────────────────
|
||
|
||
/**
|
||
* Convert a user-facing glob pattern into a git pathspec.
|
||
*
|
||
* Why: bare git pathspecs only match the repo root, so wrap with `:(glob)` and prepend `**\/` to replicate rg's recursive-by-default globbing.
|
||
*/
|
||
export function toGitGlobPathspec(glob: string, exclude?: boolean): string {
|
||
const needsRecursive = !glob.includes('/')
|
||
const pattern = needsRecursive ? `**/${glob}` : glob
|
||
return exclude ? `:(exclude,glob)${pattern}` : `:(glob)${pattern}`
|
||
}
|
||
|
||
export function buildGitGrepArgs(query: string, opts: SearchOptionsLike): string[] {
|
||
// Why: --no-recurse-submodules avoids failing when submodule.recurse=true conflicts with --untracked; --null disambiguates colon-containing filenames.
|
||
const gitArgs: string[] = [
|
||
'-c',
|
||
'submodule.recurse=false',
|
||
'grep',
|
||
'-n',
|
||
'-I',
|
||
'--null',
|
||
'--no-color',
|
||
'--untracked',
|
||
'--no-recurse-submodules'
|
||
]
|
||
if (!opts.caseSensitive) {
|
||
gitArgs.push('-i')
|
||
}
|
||
if (opts.wholeWord) {
|
||
gitArgs.push('-w')
|
||
}
|
||
if (!opts.useRegex) {
|
||
gitArgs.push('--fixed-strings')
|
||
} else {
|
||
gitArgs.push('--extended-regexp')
|
||
}
|
||
|
||
gitArgs.push('-e', query, '--')
|
||
|
||
let hasPathspecs = false
|
||
if (opts.includePattern) {
|
||
for (const pat of splitSearchGlobPatterns(opts.includePattern)) {
|
||
gitArgs.push(toGitGlobPathspec(pat))
|
||
hasPathspecs = true
|
||
}
|
||
}
|
||
if (opts.excludePattern) {
|
||
for (const pat of splitSearchGlobPatterns(opts.excludePattern)) {
|
||
gitArgs.push(toGitGlobPathspec(pat, true))
|
||
hasPathspecs = true
|
||
}
|
||
}
|
||
// Why: git grep needs a pathspec to search the working tree; '.' means everything under cwd.
|
||
if (!hasPathspecs) {
|
||
gitArgs.push('.')
|
||
}
|
||
return gitArgs
|
||
}
|
||
|
||
/**
|
||
* Build the JS regex to locate all submatch column positions in a matched line
|
||
* (git grep reports only the first hit per line).
|
||
*
|
||
* @returns `null` when the query is valid git-grep ERE but not a valid JS RegExp
|
||
* (POSIX classes, back-ref numbering, `\<`/`\>` anchors); callers then fall back to a whole-line highlight.
|
||
*/
|
||
export function buildSubmatchRegex(
|
||
query: string,
|
||
opts: { useRegex?: boolean; wholeWord?: boolean; caseSensitive?: boolean }
|
||
): RegExp | null {
|
||
let pattern = opts.useRegex ? query : escapeRegex(query)
|
||
if (opts.wholeWord) {
|
||
pattern = `\\b${pattern}\\b`
|
||
}
|
||
try {
|
||
return new RegExp(pattern, `g${opts.caseSensitive ? '' : 'i'}`)
|
||
} catch {
|
||
return null
|
||
}
|
||
}
|
||
|
||
export function ingestGitGrepLine(
|
||
line: string,
|
||
rootPath: string,
|
||
submatchRegex: RegExp | null,
|
||
acc: SearchAccumulator,
|
||
maxResults: number
|
||
): 'continue' | 'stop' {
|
||
if (acc.totalMatches >= maxResults) {
|
||
return 'stop'
|
||
}
|
||
if (!line) {
|
||
return 'continue'
|
||
}
|
||
|
||
// Why: modern git with --null -n emits filename\0linenum\0content; keep the colon parser too for hosts with older git output.
|
||
const nullIdx = line.indexOf('\0')
|
||
if (nullIdx === -1) {
|
||
return 'continue'
|
||
}
|
||
const relPath = normalizeRelativePath(line.substring(0, nullIdx))
|
||
const rest = line.substring(nullIdx + 1)
|
||
const secondNullIdx = rest.indexOf('\0')
|
||
let lineNumberText: string
|
||
let lineContent: string
|
||
if (secondNullIdx !== -1) {
|
||
lineNumberText = rest.substring(0, secondNullIdx)
|
||
lineContent = rest.substring(secondNullIdx + 1).replace(/\n$/, '')
|
||
} else {
|
||
const colonIdx = rest.indexOf(':')
|
||
if (colonIdx === -1) {
|
||
return 'continue'
|
||
}
|
||
lineNumberText = rest.substring(0, colonIdx)
|
||
lineContent = rest.substring(colonIdx + 1).replace(/\n$/, '')
|
||
}
|
||
if (!/^\d+$/.test(lineNumberText)) {
|
||
return 'continue'
|
||
}
|
||
const lineNum = Number(lineNumberText)
|
||
|
||
const absPath = joinSearchRoot(rootPath, relPath)
|
||
const getFileResult = (): SearchFileResult => {
|
||
let fileResult = acc.fileMap.get(absPath)
|
||
if (!fileResult) {
|
||
fileResult = { filePath: absPath, relativePath: relPath, matches: [], matchCount: 0 }
|
||
acc.fileMap.set(absPath, fileResult)
|
||
}
|
||
return fileResult
|
||
}
|
||
|
||
// Why: no JS-side submatch regex (git accepts patterns JS RegExp rejects); fall back to whole-line highlight so the hit still shows.
|
||
if (submatchRegex === null) {
|
||
const clamped = clampLineContext(lineContent, 0, lineContent.length)
|
||
const fileResult = getFileResult()
|
||
return pushMatch(fileResult, acc, clamped, lineNum, maxResults)
|
||
}
|
||
|
||
submatchRegex.lastIndex = 0
|
||
let m: RegExpExecArray | null
|
||
let acceptedLineMatch = false
|
||
while ((m = submatchRegex.exec(lineContent)) !== null) {
|
||
const clamped = clampLineContext(lineContent, m.index, m[0].length)
|
||
const fileResult = getFileResult()
|
||
acceptedLineMatch = true
|
||
if (pushMatch(fileResult, acc, clamped, lineNum, maxResults) === 'stop') {
|
||
return 'stop'
|
||
}
|
||
// Prevent infinite loop on zero-length regex matches.
|
||
if (m[0].length === 0) {
|
||
submatchRegex.lastIndex++
|
||
}
|
||
}
|
||
// Why: git grep confirmed the line but JS regex found no occurrence; keep it navigable, don't drop a git-confirmed hit.
|
||
if (!acceptedLineMatch) {
|
||
const clamped = clampLineContext(lineContent, 0, lineContent.length)
|
||
const fileResult = getFileResult()
|
||
if (pushMatch(fileResult, acc, clamped, lineNum, maxResults) === 'stop') {
|
||
return 'stop'
|
||
}
|
||
}
|
||
return 'continue'
|
||
}
|
||
|
||
// ─── finalize ───────────────────────────────────────────────────────
|
||
|
||
export function finalize(acc: SearchAccumulator): SearchResult {
|
||
return normalizeSearchResult({
|
||
files: Array.from(acc.fileMap.values()).filter((file) => file.matches.length > 0),
|
||
totalMatches: acc.totalMatches,
|
||
truncated: acc.truncated
|
||
})
|
||
}
|