diff --git a/CHANGELOG.md b/CHANGELOG.md index fe753f8e..a2a3507d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,25 @@ # Changelog +## 1.0.2 — 2026-09-29 + +### Fixed + +- Inspecting an outline item (`cursor` + `matchIndex` from `mode: "outline"`) now returns the symbol's version-checked source with its syntax boundary instead of metadata only ([#96](https://github.com/lightsifter/sift-light/issues/96)). +- `mode: "files"` no longer reports `complete` coverage when ignore rules skipped files. It reports `policy-filtered` coverage, the number of ignored files and how many of them match the query with examples, and accepts `ignorePolicy: "include"` to list them. File candidates show their score and ranking reason in text and compact model output ([#97](https://github.com/lightsifter/sift-light/issues/97)). +- Display redaction keeps the source's own quoting. An unquoted secret such as `DB_PASSWORD=value` is shown as `DB_PASSWORD=[REDACTED]` instead of `DB_PASSWORD="[REDACTED]"`, so masked output no longer suggests quotes that are not in the file. +- Display redaction no longer masks source code that reads a secret. An unquoted value that starts as a call or index expression, such as `password = env.get("DB_PASS", "")` or `api_key = os.environ["API_KEY"]`, stays visible, so the key being read remains available as evidence. A literal secret that is unquoted and itself begins with letters followed by `(` or `[` is therefore not masked. +- `mode: "inspect"` with a direct `path` and no `line` opens the file from line 1 instead of failing. Cursor inspection still requires the exact retained line. + +### Changed + +These request shapes were observed from real agent sessions; each was rejected although its intent was unambiguous. Rewrites are disclosed in `details.requestNotes` and as `[Request note: …]` lines. + +- `mode: "files"` accepts `limit` (files per page, default 30), the redundant `scope: "strict"` and `ignorePolicy`. A wildcard-only `query` (`*`, `**`, `.*`) lists every file, and a plain glob query such as `*.py` is applied as a glob filter; regex-like queries still fail with guidance. `pattern: ".*"` gets an exact retry request. +- `mode: "inspect"` accepts `paths` without a cursor to open several files from line 1, and the redundant `scope: "strict"` (`scope: "expand"` fails). `targets` and `matchIndices` accept up to 20 entries: five are inspected per response and the rest are returned as an exact `nextRequest`. A `sourceCursor` request may repeat its own `path` (a different path fails) and ignores `line` with a note. +- `anyOf` accepts one term and `allOf` accepts one to three terms. `mode: "anyOf"` and `mode: "allOf"` are accepted as aliases for omitting `mode`, and `anyOf` with `mode: "summary"` is served by its analysis page. +- `roles` supports Python through a bounded lexical scanner: comments, strings and code are exact; declarations and imports are line facts; calls are candidates. `mode: "capabilities"` lists it. +- A request without `pattern` explains how to list files or read a known file. + ## 1.0.1 — 2026-09-23 ### Changed diff --git a/README.md b/README.md index a5ffb71e..916f414d 100644 --- a/README.md +++ b/README.md @@ -53,11 +53,11 @@ Long results arrive in pages with a way to continue. When the original material A complete snapshot describes match retention, not complete source text. Truncated matching-line excerpts show their limit and an executable `inspectRequest`, which remains usable after the final match page. Follow pagination cursors instead of repeating the query with a different limit. In Pi and OMP, the passive session status includes the loaded package version, counts returned new queries, distinguishes complete, partial and unfinished results, and reports non-cancelled failed calls. Cursor and operation continuations do not inflate the new-query count. -Ordinary searches keep repository ignore rules, but now report policy-filtered filesystem coverage instead of calling an absence exhaustive when ignored files exist. Use `mode: "audit"` with named literal `patterns` for one bounded receipt covering declared scope, enumerated/searched/skipped files, ignored-file policy, per-pattern `present`/`absent_with_complete_coverage`/`unknown` findings, and start/end source stability. Set `ignorePolicy: "include"` when an audit must include ignored configuration and generated files; `.git` internals and protected paths remain excluded. +Ordinary searches keep repository ignore rules, but now report policy-filtered filesystem coverage instead of calling an absence exhaustive when ignored files exist. Filename discovery (`mode: "files"`) does the same: it says how many files were ignored and which of them match the query, and `ignorePolicy: "include"` lists them. Each file candidate shows its score and whether it is an exact, substring or fuzzy match. Use `mode: "audit"` with named literal `patterns` for one bounded receipt covering declared scope, enumerated/searched/skipped files, ignored-file policy, per-pattern `present`/`absent_with_complete_coverage`/`unknown` findings, and start/end source stability. Set `ignorePolicy: "include"` when an audit must include ignored configuration and generated files; `.git` internals and protected paths remain excluded. ### Discover language capabilities before loading a provider -Use `mode: "capabilities"` with the project root for a compact, names-only inventory. JavaScript, TypeScript and TSX support AST structure, roles, outline, static imports and related-test candidates. Go supports AST structure and roles. Python supports bounded indentation-based outline. Swift and other languages remain available to ordinary content search, filename discovery and source inspection. Capability inventory does not start parsers or the Concept model. +Use `mode: "capabilities"` with the project root for a compact, names-only inventory. JavaScript, TypeScript and TSX support AST structure, roles, outline, static imports and related-test candidates. Go supports AST structure and roles. Python supports bounded indentation-based outline and lexical roles (comment, string, code, declaration, import, and call candidates). Swift and other languages remain available to ordinary content search, filename discovery and source inspection. Capability inventory does not start parsers or the Concept model. Language-service navigation is out of scope. Asking for `definitions`, `references`, `implementations`, `callers`, `callees`, `dependencies`, `dependents`, `trace` or `impact` fails explicitly, because a text search dressed up as precise navigation would be a worse answer than a clear refusal. No language server runs while you work. @@ -69,7 +69,7 @@ Use `mode: "validate"` with a saved ordinary-search or analysis `cursor`, option Worktree searches can use `modifiedAfter` and `modifiedBefore` as Unix millisecond bounds. The lower bound is inclusive and the upper bound is exclusive, so a time window can be expressed without changing the search pattern. The same filter applies to content and filename searches; unavailable file metadata is reported as incomplete evidence rather than silently treated as a match. -Use `mode: "outline"` with a concrete JS/TS/TSX or Python file to see bounded symbol ranges. JS/TS/TSX use ast-grep; Python uses indentation-based class, function and method evidence. These ranges do not prove compiler bindings, runtime calls or test coverage. `mode: "tests"` provides JS/TS/TSX related-test candidates; unsupported language operations fail explicitly. Use ordinary search and `inspect` for Swift source. +Use `mode: "outline"` with a concrete JS/TS/TSX or Python file to see bounded symbol ranges. JS/TS/TSX use ast-grep; Python uses indentation-based class, function and method evidence. Inspecting an outline item returns that symbol's version-checked source. These ranges do not prove compiler bindings, runtime calls or test coverage. `mode: "tests"` provides JS/TS/TSX related-test candidates; unsupported language operations fail explicitly. Use ordinary search and `inspect` for Swift source. The readable result keeps the main evidence compact. Per-item ranges, counts, coverage and continuation requests remain in structured `details`, so a client can use the structured fields without requiring a second search. diff --git a/README.zh-CN.md b/README.zh-CN.md index 8e4e2181..4385e53d 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -53,11 +53,11 @@ Concept 或 hybrid 较慢时,会在默认五秒等待窗口内返回 `status: 完整快照表示匹配保留完整,不代表源码正文没有截断。长行摘录会明确标记限制,并给出最后一页之后仍可执行的 `inspectRequest`。需要更多结果时沿游标继续,不要仅为翻页而修改 limit 重搜。 Pi 和 OMP 的被动 session 状态会显示当前加载的包版本,统计已返回的新查询,区分完整、部分和未完成结果,并报告未取消的失败调用。cursor 与 operation 续接不会重复计入新查询。 -普通搜索继续遵循仓库 ignore 规则,但只要存在被忽略文件,就会明确说明文件系统覆盖受策略过滤,不再把“接纳文件里没找到”说成“整个目录绝对不存在”。需要发布前收口时,可以用 `mode: "audit"` 配合带名字的字面量 `patterns`,一次拿到声明范围、枚举/搜索/跳过文件、ignore 策略、每个模式的 `present`、`absent_with_complete_coverage` 或 `unknown` 结论,以及搜索前后的来源稳定性。审计必须包含被忽略的配置或生成文件时,设置 `ignorePolicy: "include"`;`.git` 内部和受保护路径仍然不会开放。 +普通搜索继续遵循仓库 ignore 规则,但只要存在被忽略文件,就会明确说明文件系统覆盖受策略过滤,不再把“接纳文件里没找到”说成“整个目录绝对不存在”。文件名发现(`mode: "files"`)同样会说明有多少文件被忽略、其中哪些匹配查询,`ignorePolicy: "include"` 可以把它们列出来;每个候选都显示分数以及是精确、子串还是模糊匹配。需要发布前收口时,可以用 `mode: "audit"` 配合带名字的字面量 `patterns`,一次拿到声明范围、枚举/搜索/跳过文件、ignore 策略、每个模式的 `present`、`absent_with_complete_coverage` 或 `unknown` 结论,以及搜索前后的来源稳定性。审计必须包含被忽略的配置或生成文件时,设置 `ignorePolicy: "include"`;`.git` 内部和受保护路径仍然不会开放。 ### 先发现语言能力,再按需加载提供方 -使用 `mode: "capabilities"` 和项目根目录,可以获取紧凑的文件语言清单。JavaScript、TypeScript 和 TSX 支持 AST 结构、角色、outline、静态 imports 和关联测试候选;Go 支持 AST 结构和角色;Python 支持基于缩进的有界 outline。Swift 和其他语言仍可使用普通内容搜索、文件发现和源码 inspect。能力清单不会启动 parser 或 Concept 模型。 +使用 `mode: "capabilities"` 和项目根目录,可以获取紧凑的文件语言清单。JavaScript、TypeScript 和 TSX 支持 AST 结构、角色、outline、静态 imports 和关联测试候选;Go 支持 AST 结构和角色;Python 支持基于缩进的有界 outline 和词法角色(注释、字符串、代码、声明、导入和调用候选)。Swift 和其他语言仍可使用普通内容搜索、文件发现和源码 inspect。能力清单不会启动 parser 或 Concept 模型。 语言服务导航不在范围内。请求 `definitions`、`references`、`implementations`、`callers`、`callees`、`dependencies`、`dependents`、`trace` 或 `impact` 都会明确报错——用文本搜索伪装精确导航,比直接说清楚更糟。使用期间不会启动任何语言服务。 @@ -69,7 +69,7 @@ Pi 和 OMP 的被动 session 状态会显示当前加载的包版本,统计已 工作区搜索支持用 Unix 毫秒时间戳传入 `modifiedAfter` 和 `modifiedBefore`。下界包含、上界不包含,因此可以准确表示一个时间窗口,不必改动搜索关键词。内容搜索和文件名搜索使用同一过滤条件;无法核验文件元数据时会明确报告证据不完整,不会静默当作命中。 -对具体的 JS/TS/TSX 或 Python 文件使用 `mode: "outline"`,可以查看有界符号范围。JS/TS/TSX 使用 ast-grep,Python 使用基于缩进的类、函数和方法范围;它们不证明编译器绑定、运行时调用或测试覆盖。`mode: "tests"` 提供 JS/TS/TSX 关联测试候选,不支持的语言操作会明确报错。Swift 源码可使用普通搜索和 `inspect`。 +对具体的 JS/TS/TSX 或 Python 文件使用 `mode: "outline"`,可以查看有界符号范围。JS/TS/TSX 使用 ast-grep,Python 使用基于缩进的类、函数和方法范围;对 outline 条目执行 inspect 会返回该符号经版本校验的源码。它们不证明编译器绑定、运行时调用或测试覆盖。`mode: "tests"` 提供 JS/TS/TSX 关联测试候选,不支持的语言操作会明确报错。Swift 源码可使用普通搜索和 `inspect`。 可读正文会保持精简;每项证据的范围、计数、覆盖状态和继续请求仍保留在结构化 `details` 中,客户端无需为了拿到这些字段再次搜索。 diff --git a/package.json b/package.json index 8c289eda..da2e1c8c 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "sift-light", - "version": "1.0.1", + "version": "1.0.2", "description": "Context-efficient local search for files, documents, notes and logs across Pi, OMP and MCP clients", "keywords": [ "ai-agent", diff --git a/plugins/sift-light/.claude-plugin/plugin.json b/plugins/sift-light/.claude-plugin/plugin.json index 6309f297..7d00d244 100644 --- a/plugins/sift-light/.claude-plugin/plugin.json +++ b/plugins/sift-light/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sift-light", - "version": "1.0.1", + "version": "1.0.2", "description": "Require sift-light for conventional local searches while keeping development tools available.", "author": { "name": "baoer" diff --git a/plugins/sift-light/.codex-plugin/plugin.json b/plugins/sift-light/.codex-plugin/plugin.json index 3f2a8716..8ca36db8 100644 --- a/plugins/sift-light/.codex-plugin/plugin.json +++ b/plugins/sift-light/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sift-light", - "version": "1.0.1", + "version": "1.0.2", "description": "Require sift-light for conventional local searches while keeping development tools available.", "author": { "name": "baoer" diff --git a/plugins/sift-light/.mcp.json b/plugins/sift-light/.mcp.json index 16aa32da..8d7085b6 100644 --- a/plugins/sift-light/.mcp.json +++ b/plugins/sift-light/.mcp.json @@ -5,7 +5,7 @@ "args": [ "--yes", "--package", - "sift-light-runtime@npm:sift-light@1.0.1", + "sift-light-runtime@npm:sift-light@1.0.2", "sift-light-mcp", "--stdio" ], diff --git a/plugins/sift-light/kimi.plugin.json b/plugins/sift-light/kimi.plugin.json index 69b6f2c4..ab5824bc 100644 --- a/plugins/sift-light/kimi.plugin.json +++ b/plugins/sift-light/kimi.plugin.json @@ -1,6 +1,6 @@ { "name": "sift-light", - "version": "1.0.1", + "version": "1.0.2", "description": "Require sift-light for conventional local searches while keeping development tools available.", "author": { "name": "baoer" @@ -13,7 +13,7 @@ "args": [ "--yes", "--package", - "sift-light-runtime@npm:sift-light@1.0.1", + "sift-light-runtime@npm:sift-light@1.0.2", "sift-light-mcp", "--stdio" ], diff --git a/plugins/sift-light/omp-extension.mjs b/plugins/sift-light/omp-extension.mjs index f69b45a4..e14179d8 100644 --- a/plugins/sift-light/omp-extension.mjs +++ b/plugins/sift-light/omp-extension.mjs @@ -229,6 +229,7 @@ var ESTIMATED_CHARACTERS_PER_TOKEN = 4; var DEFAULT_SUMMARY_FILE_LIMIT = 30; var MAX_SELECTED_PATHS = 20; var MAX_INSPECT_TARGETS = 5; +var MAX_INSPECT_REQUEST_TARGETS = 20; var MAX_DISPLAYED_OCCURRENCES = 20; var MAX_STORED_MATCHES = 50000; var MAX_STORED_OCCURRENCES = 200000; @@ -808,6 +809,7 @@ var PRIVATE_KEY = /-----BEGIN ([^-\r\n]*PRIVATE KEY)-----[\s\S]*?-----END \1---- var SENSITIVE_NAME = String.raw`(?:(?:[A-Za-z][A-Za-z0-9]*[_-])*(?:password|passwd|secret|token|api[_-]?key|access[_-]?(?:key|token)|secret[_-]?access[_-]?key|private[_-]?key|service[_-]?key)(?:[_-][A-Za-z0-9]+)*)`; var SENSITIVE_ASSIGNMENT = new RegExp(String.raw`((? { const unquoted = rawValue.replace(/^["']|["']$/g, "").toLowerCase(); - if (TYPE_ONLY_VALUES.has(unquoted)) + if (TYPE_ONLY_VALUES.has(unquoted) || CODE_EXPRESSION.test(rawValue)) return match; count += 1; - return `${prefix}"[REDACTED]"`; + const quote = rawValue.startsWith('"') || rawValue.startsWith("'") ? rawValue[0] : ""; + return `${prefix}${quote}[REDACTED]${quote}`; }); redacted = redacted.replace(SENSITIVE_TOKEN, () => { count += 1; @@ -1819,7 +1822,7 @@ function createCtagsStructureProvider(options = {}) { // package.json var package_default = { name: "sift-light", - version: "1.0.1", + version: "1.0.2", description: "Context-efficient local search for files, documents, notes and logs across Pi, OMP and MCP clients", keywords: [ "ai-agent", @@ -2281,7 +2284,7 @@ function normalizeRequest(input) { throw new SiftLightError("ignorePolicy must be respect or include"); const pattern = input.pattern; if (pattern === undefined) { - throw new SiftLightError("pattern is required when cursor is not provided"); + throw new SiftLightError('pattern is required when cursor is not provided; to list files use mode="files" (query optional), to read a known file use mode="inspect" with path'); } const path = input.path?.replace(/^@/, ""); return { @@ -2321,7 +2324,7 @@ var MAX_ANALYSIS_STORAGE_BYTES = 32 * 1024 * 1024; var ANALYSIS_METADATA_RESERVE_BYTES = 64 * 1024; var MAX_ANALYSIS_REASONS = 64; var MAX_ANALYSIS_REASON_BYTES = 4 * 1024; -var MIN_ANY_OF_TERMS = 2; +var MIN_ANY_OF_TERMS = 1; var MAX_ANY_OF_TERMS = 8; var MAX_ANY_OF_TOTAL_TERMS = 64; var MAX_LITERAL_TERM_BYTES = 256; @@ -4038,6 +4041,15 @@ var PYTHON_LANGUAGE_CAPABILITIES = [ evidence: "syntax", availability: "implemented", load: "lazy" + }, + { + id: "python-lexical.roles", + name: "roles", + provider: "bounded Python lexical role scanner", + providerKind: "builtin", + evidence: "syntax", + availability: "implemented", + load: "lazy" } ]; var DEFAULT_LANGUAGE_CAPABILITIES = [ @@ -5487,7 +5499,7 @@ function publicAnalysisItem(result, item, index, storedId, modelOutput) { return { path: item.path, line: item.line, - label: result.kind === "outline" && modelOutput ? item.label : publicAnalysisLabel(result), + label: result.kind === "outline" && modelOutput || result.kind === "files" ? item.label : publicAnalysisLabel(result), index: index + 1, ...inspect ? { inspect } : {}, ...publicDetails ? { details: publicDetails } : {} @@ -5698,7 +5710,8 @@ Inspect: ${JSON.stringify(inspect)}` : ""}`; break; } } else { - for (let index = offset;index < result.items.length && items.length < 30; index += 1) { + const pageSize = result.pageSize ?? 30; + for (let index = offset;index < result.items.length && items.length < pageSize; index += 1) { if (!appendItem(index)) break; } @@ -5812,6 +5825,8 @@ function resolveInspectionTarget(input, cwd, snapshots) { } if (!path) throw new SiftLightError("path is required when mode=inspect"); + if (line === undefined && input.cursor === undefined) + line = 1; if (line === undefined || !Number.isSafeInteger(line) || line < 1) { throw new SiftLightError("line must be a positive integer when mode=inspect"); } @@ -7540,6 +7555,28 @@ async function filterPathsByModificationTime(cwd, paths, modifiedAfterMs, modifi // src/file-discovery.ts var graphemes = new Intl.Segmenter(undefined, { granularity: "grapheme" }); var FILE_QUERY_GLOB_WILDCARD = /[*?]/u; +var FILE_QUERY_MATCH_ALL = /^(?:[*/]+|\.\*|\.\+|\*\.\*)$/u; +var FILE_QUERY_REGEX_SYNTAX = /[()|^$\\+]/u; +var IGNORED_MATCH_SAMPLES = 5; +function normalizeFileQuery(query, hasGlob) { + const trimmed = query.trim(); + if (FILE_QUERY_MATCH_ALL.test(trimmed)) + return { + query: "", + note: `Query ${JSON.stringify(query)} was treated as listing every file under path; omit query for the same result.` + }; + if (!FILE_QUERY_GLOB_WILDCARD.test(trimmed)) + return { query }; + if (!hasGlob && !FILE_QUERY_REGEX_SYNTAX.test(trimmed) && !/\s/u.test(trimmed)) { + const glob = trimmed.includes("/") ? trimmed : `**/${trimmed}`; + return { + query: "", + glob, + note: `Query ${JSON.stringify(query)} contains glob wildcards and was applied as glob ${JSON.stringify(glob)}; mode=files query otherwise matches filename text.` + }; + } + throw new SiftLightError(`File query ${JSON.stringify(query)} uses wildcard or regex syntax; mode=files matches filename and path text, not patterns. Omit query to retain every file under path, use glob for one name pattern, or run one files request per name.`); +} function subsequenceScore(text, query) { const characters = Array.from(graphemes.segment(text), (item) => item.segment); const queryCharacters = Array.from(graphemes.segment(query), (item) => item.segment); @@ -7590,39 +7627,79 @@ function pathRelativeToDiscoveryRoot(cwd, root, path) { return scoped || platformBasename(absolutePath); } async function discoverFiles(input, cwd, signal) { - const query = input.query ?? ""; - if (query.length > 256 || !query.isWellFormed() || /[\r\n\0]/.test(query)) + const rawQuery = input.query ?? ""; + if (rawQuery.length > 256 || !rawQuery.isWellFormed() || /[\r\n\0]/.test(rawQuery)) throw new SiftLightError("File query must be well-formed single-line text of at most 256 characters"); - if (FILE_QUERY_GLOB_WILDCARD.test(query)) - throw new SiftLightError(`File query ${JSON.stringify(query)} uses glob wildcards; mode=files matches filename and path text, not glob patterns. Omit query to retain every file under path, or use glob to filter by name pattern.`); - const request = normalizeRequest({ ...input, pattern: "" }); + const requestedGlobs = input.glob === undefined ? [] : Array.isArray(input.glob) ? input.glob : [input.glob]; + const normalizedQuery = normalizeFileQuery(rawQuery, requestedGlobs.length > 0); + const query = normalizedQuery.query; + const request = normalizeRequest({ + ...input, + pattern: "", + ...normalizedQuery.glob ? { glob: [...requestedGlobs, normalizedQuery.glob] } : {} + }); const policy = new SearchPathPolicy(cwd); const discoveryRoot = request.path ?? "."; const scoringRoot = await policy.resolveSearchTarget(discoveryRoot); - const files = await listWorkspaceFiles(cwd, signal, { + const includeIgnored = request.ignorePolicy === "include"; + const listOptions = { path: scoringRoot, glob: request.glob, exclude: request.exclude, hidden: request.hidden + }; + const files = await listWorkspaceFiles(cwd, signal, { + ...listOptions, + ...includeIgnored ? { ignore: false, ignoreParents: false } : {} }); const filtered = await filterPathsByModificationTime(cwd, files.paths, request.modifiedAfterMs, request.modifiedBeforeMs, signal); + const rank = (path) => scoreFilePath(pathRelativeToDiscoveryRoot(cwd, scoringRoot, path), query); const selected = filtered.paths.flatMap((path) => { - const rank = scoreFilePath(pathRelativeToDiscoveryRoot(cwd, scoringRoot, path), query); - return rank ? [{ path, ...rank }] : []; + const score = rank(path); + return score ? [{ path, ...score }] : []; }).toSorted((left, right) => right.score - left.score || left.path.localeCompare(right.path)); for (let offset = 0;offset < selected.length; offset += 16) { await Promise.all(selected.slice(offset, offset + 16).map((item) => policy.assertExistingPath(item.path))); signal?.throwIfAborted(); } + const reasons = new Set([...files.reasons, ...filtered.reasons]); + if (normalizedQuery.note) + reasons.add(normalizedQuery.note); + let ignoredFiles = 0; + let ignoredMatches = 0; + let ignoredComparisonPartial = false; + if (!includeIgnored) { + const everything = await listWorkspaceFiles(cwd, signal, { + ...listOptions, + ignore: false, + ignoreParents: false + }); + ignoredComparisonPartial = everything.partial; + const admitted = new Set(files.paths); + const ignored = everything.paths.filter((path) => !admitted.has(path)); + ignoredFiles = ignored.length; + const ignoredCandidates = ignored.flatMap((path) => { + const score = rank(path); + return score ? [{ path, ...score }] : []; + }).toSorted((left, right) => right.score - left.score || left.path.localeCompare(right.path)); + ignoredMatches = ignoredCandidates.length; + if (ignoredFiles > 0) { + const samples = ignoredCandidates.slice(0, IGNORED_MATCH_SAMPLES).map((item) => `${JSON.stringify(item.path)} (${item.reason})`); + reasons.add(ignoredMatches > 0 ? `${String(ignoredFiles)} file(s) were excluded by ignore rules; ${String(ignoredMatches)} of them match this query${samples.length ? `, e.g. ${samples.join(", ")}` : ""}. Retry with ignorePolicy "include" to list them.` : `${String(ignoredFiles)} file(s) were excluded by ignore rules; none of them match this query.`); + } + if (ignoredComparisonPartial) + reasons.add("Ignored-file comparison was incomplete; the ignored-file count is a lower bound"); + } + const enumerationPartial = files.partial || filtered.partial; return { kind: "files", unit: "files", - partial: files.partial || filtered.partial, - reasons: [...new Set([...files.reasons, ...filtered.reasons])], + partial: enumerationPartial, + reasons: [...reasons], items: selected.map((item) => ({ path: item.path, line: 1, - label: `File candidate (${item.reason})`, + label: `File candidate (score ${String(item.score)}: ${item.reason})`, details: { kind: "file", score: item.score, @@ -7630,8 +7707,13 @@ async function discoverFiles(input, cwd, signal) { inspect: { mode: "inspect", path: item.path, line: 1 } } })), - coverage: { fileEnumeration: files.partial || filtered.partial ? "partial" : "complete" }, - stats: { filesEnumerated: files.paths.length }, + coverage: { + fileEnumeration: enumerationPartial ? "partial" : ignoredFiles > 0 || ignoredComparisonPartial ? "policy-filtered" : "complete" + }, + stats: { + filesEnumerated: files.paths.length, + ...includeIgnored ? {} : { ignoredFiles, ignoredMatches } + }, scope: { path: request.path ?? ".", requestedPath: request.path ?? ".", @@ -7644,7 +7726,8 @@ async function discoverFiles(input, cwd, signal) { ...request.modifiedAfterMs !== undefined ? { modifiedAfterMs: request.modifiedAfterMs } : {}, ...request.modifiedBeforeMs !== undefined ? { modifiedBeforeMs: request.modifiedBeforeMs } : {} }, - redact: input.redact ?? false + redact: input.redact ?? false, + ...input.limit !== undefined ? { pageSize: request.pageSize } : {} }; } @@ -7885,7 +7968,19 @@ async function prepare(target, access, structure) { let range = target.range; let details = { status: "no-symbol" }; const language = syntaxLanguage(document2.path); - if (document2.utf8 && language && language !== "go") { + if (target.range && target.structure) { + document2.checkRange(target.range); + const lines = { + startLine: document2.lineAt(target.range.start), + endLine: document2.lineAt(Math.max(target.range.start, target.range.end - 1)) + }; + range = document2.lineRange(lines.startLine, lines.endLine); + details = { + ...target.structure, + range: lines, + ...target.structure.symbol ? { symbol: { ...target.structure.symbol, range: lines } } : {} + }; + } else if (document2.utf8 && language && language !== "go") { const syntax = await access.syntax(document2); details = { status: syntax.status === "ok" ? "no-symbol" : syntax.status === "unsupported" ? "provider-unavailable" : "parse-error", @@ -7937,7 +8032,7 @@ async function prepare(target, access, structure) { details = { status: "provider-unavailable", ...language ? { language } : {} }; } range ??= document2.lineRange(Math.max(1, target.line - 10), Math.min(document2.lineStarts.length, target.line + 10)); - const boundary = target.range ? "requested-range" : details.status === "available" && details.range ? "syntax" : "line-window"; + const boundary = target.range ? target.structure ? "syntax" : "requested-range" : details.status === "available" && details.range ? "syntax" : "line-window"; document2.checkRange(range); return { target, document: document2, range, structure: details, boundary, focus }; } @@ -8296,8 +8391,10 @@ ${preview.text}`); } }; } -async function continueSource(cursor, access, continuations) { +async function continueSource(cursor, access, continuations, expectedPath) { const state = continuations.resolve(cursor); + if (expectedPath !== undefined && resolve18(access.cwd, expectedPath.replace(/^@/, "")) !== resolve18(access.cwd, state.source.path)) + throw new SiftLightError(`sourceCursor continues ${JSON.stringify(state.source.path)}, not ${JSON.stringify(expectedPath)}; copy the returned nextRequest exactly`); const document2 = await access.load(state.source.path, state.source); const page = sourcePage(document2, state.remaining, MAX_RESULT_BYTES - 1400); const next = continuations.advance(cursor, page.fragment); @@ -9174,6 +9271,221 @@ function parsePythonOutline(document2) { }); } +// src/python-roles.ts +var KEYWORDS = new Set([ + "and", + "as", + "assert", + "async", + "await", + "case", + "del", + "elif", + "else", + "except", + "for", + "from", + "global", + "if", + "import", + "in", + "is", + "lambda", + "match", + "nonlocal", + "not", + "or", + "raise", + "return", + "while", + "with", + "yield" +]); +var STRING_PREFIX = /^(?:[rRbBuUfF]|[rR][bBfF]|[bBfF][rR])?$/u; +var IDENTIFIER_START = /[\p{ID_Start}_]/u; +var IDENTIFIER_PART = /[\p{ID_Continue}]/u; +function lex(text) { + const comments = []; + const strings = []; + const code = new Uint8Array(text.length).fill(1); + const mark = (start2, end) => { + code.fill(0, start2, end); + }; + let index = 0; + while (index < text.length) { + const character = text[index]; + if (character === "#") { + const newline = text.indexOf(` +`, index); + const end = newline < 0 ? text.length : newline; + comments.push({ start: index, end }); + mark(index, end); + index = end; + continue; + } + if (character !== "'" && character !== '"') { + index += 1; + continue; + } + let prefixStart = index; + while (prefixStart > 0 && /[rRbBuUfF]/u.test(text[prefixStart - 1] ?? "")) + prefixStart -= 1; + const prefix = text.slice(prefixStart, index); + const beforePrefix = text[prefixStart - 1] ?? ""; + const start2 = STRING_PREFIX.test(prefix) && !IDENTIFIER_PART.test(beforePrefix) ? prefixStart : index; + const triple = text.slice(index, index + 3) === character.repeat(3); + const delimiter = triple ? character.repeat(3) : character; + let cursor = index + delimiter.length; + let end = text.length; + while (cursor < text.length) { + const current = text[cursor]; + if (current === "\\") { + cursor += 2; + continue; + } + if (!triple && current === ` +`) { + end = cursor; + break; + } + if (text.startsWith(delimiter, cursor)) { + end = cursor + delimiter.length; + break; + } + cursor += 1; + } + strings.push({ start: start2, end }); + mark(start2, end); + index = end; + } + return { comments, strings, code }; +} +function codeRanges(code) { + const ranges = []; + let start2 = -1; + for (let index = 0;index <= code.length; index += 1) { + const inCode = index < code.length && code[index] === 1; + if (inCode && start2 < 0) + start2 = index; + else if (!inCode && start2 >= 0) { + ranges.push({ start: start2, end: index }); + start2 = -1; + } + } + return ranges; +} +function identifierEnd(text, start2) { + let end = start2; + while (end < text.length && IDENTIFIER_PART.test(text[end] ?? "")) + end += 1; + return end; +} +function skipSpaces(text, index, code) { + let cursor = index; + while (cursor < text.length && code[cursor] === 1 && /[ \t]/u.test(text[cursor] ?? "")) + cursor += 1; + return cursor; +} +function statementEnd(text, start2, code) { + let depth = 0; + for (let index = start2;index < text.length; index += 1) { + if (code[index] !== 1) + continue; + const character = text[index]; + if (character === "(" || character === "[" || character === "{") + depth += 1; + else if (character === ")" || character === "]" || character === "}") + depth = Math.max(0, depth - 1); + else if (character === "\\" && text[index + 1] === ` +`) + index += 1; + else if (character === ` +` && depth === 0) + return index; + } + return text.length; +} +function pythonRoleAnalysis(document2) { + const text = document2.text; + const { comments, strings, code } = lex(text); + const roles = []; + const push = (start2, end, role, certainty, subkind) => { + if (start2 < end) + roles.push({ start: start2, end, role, certainty, node: 0, ...subkind ? { subkind } : {} }); + }; + for (const span of comments) + push(span.start, span.end, "comment", "syntax"); + for (const span of strings) + push(span.start, span.end, "string", "syntax"); + for (const span of codeRanges(code)) + push(span.start, span.end, "code", "syntax"); + let index = 0; + let lineStart = true; + while (index < text.length) { + const character = text[index] ?? ""; + if (character === ` +`) { + lineStart = true; + index += 1; + continue; + } + if (code[index] !== 1 || !IDENTIFIER_START.test(character)) { + if (!/[ \t]/u.test(character)) + lineStart = false; + index += 1; + continue; + } + const previous = text[index - 1] ?? ""; + if (IDENTIFIER_PART.test(previous)) { + index += 1; + continue; + } + const end = identifierEnd(text, index); + const word = text.slice(index, end); + const atStatementStart = lineStart; + lineStart = false; + if (atStatementStart && (word === "import" || word === "from")) { + const statement = statementEnd(text, index, code); + const body2 = text.slice(index, statement); + if (word === "import" || /\bimport\b/u.test(body2)) { + push(index, statement, "import", "syntax", word === "from" ? "from-import" : "import"); + index = statement; + continue; + } + } + if (word === "def" || word === "class") { + const nameStart = skipSpaces(text, end, code); + if (IDENTIFIER_START.test(text[nameStart] ?? "")) { + const nameEnd = identifierEnd(text, nameStart); + push(nameStart, nameEnd, "declaration", "syntax", word === "def" ? "function" : "class"); + index = nameEnd; + continue; + } + } + let chainEnd = end; + while (text[chainEnd] === "." && IDENTIFIER_START.test(text[chainEnd + 1] ?? "")) { + chainEnd = identifierEnd(text, chainEnd + 1); + } + const open = skipSpaces(text, chainEnd, code); + const lastSegment = text.slice(text.lastIndexOf(".", chainEnd - 1) + 1, chainEnd); + const before = text.slice(Math.max(0, index - 4), index); + if (text[open] === "(" && code[open] === 1 && !KEYWORDS.has(word) && !KEYWORDS.has(lastSegment) && !/\bdef\s+$|\bclass\s+$/u.test(before)) { + push(index, open + 1, "call", "candidate", "lexical-call"); + } + index = chainEnd; + } + roles.sort((a, b) => a.start - b.start || a.end - b.end || a.role.localeCompare(b.role)); + return { + status: "ok", + nodes: [], + children: [], + symbols: [], + roles, + diagnostics: [], + limited: false + }; +} + // src/hybrid-search.ts import { resolve as resolve20 } from "path"; @@ -10080,7 +10392,17 @@ var MODE_FIELDS_BY_MODE = { auto: ordinaryFields, summary: ordinaryFields, matches: ordinaryFields, - inspect: [...commonFields, "path", "line", "cursor", "matchIndex", "matchIndices", "targets"], + inspect: [ + ...commonFields, + "path", + "paths", + "line", + "cursor", + "matchIndex", + "matchIndices", + "targets", + "scope" + ], outline: [...commonFields, "path", "line", "symbol", "cursor", "matchIndex", "maxFilesToParse"], imports: [ ...commonFields, @@ -10100,7 +10422,16 @@ var MODE_FIELDS_BY_MODE = { "matchIndex", "maxFilesToParse" ], - files: [...commonFields, "query", ...sourceFilters, "modifiedAfter", "modifiedBefore"], + files: [ + ...commonFields, + "query", + ...sourceFilters, + "ignorePolicy", + "limit", + "scope", + "modifiedAfter", + "modifiedBefore" + ], structure: [...commonFields, "pattern", ...sourceFilters, "maxFilesToParse"], concept: [...commonFields, "query", ...sourceFilters, "maxFilesToParse"], hybrid: [...commonFields, "query", ...sourceFilters, "conceptLimit", "maxFilesToParse"], @@ -10109,9 +10440,7 @@ var MODE_FIELDS_BY_MODE = { await: ["mode", "operationId"], cancel: ["mode", "operationId"] }; -var SAFE_DROP_FIELDS = { - files: ["scope"] -}; +var SAFE_DROP_FIELDS = {}; var SUPPORTED_OUTLINE_EXTENSIONS = new Set(DEFAULT_LANGUAGE_CAPABILITIES.flatMap((descriptor) => descriptor.capabilities.some((capability) => capability.name === "outline") ? descriptor.extensions : [])); function modeFields(mode) { return MODE_FIELDS_BY_MODE[mode]; @@ -10140,8 +10469,8 @@ var REQUEST_FIELD_GUIDANCE = { allOf: "allOf is case-sensitive exact-literal AND; omit pattern, anyOf, literal, ignoreCase, roles", context: "context is an output-context budget and is never silently dropped", limit: "limit is an output/page budget and is never silently dropped", - scope: "scope applies to ordinary content search; mode=files rejects this field because its scope is fixed strict, and only redundant strict may be removed", - ignorePolicy: "respect keeps repository ignore rules; include searches ignored files but always excludes .git internals and protected paths", + scope: "scope applies to ordinary content search; mode=files is always strict, so it accepts only the redundant scope=strict", + ignorePolicy: "respect keeps repository ignore rules and discloses ignored files; include also searches or lists ignored files but always excludes .git internals and protected paths", patterns: "audit accepts named exact-literal patterns and returns one closure receipt", maxFilesToParse: "concept/hybrid automatically batch the requested scope; this optional field sets an advanced hard file ceiling" }; @@ -10240,12 +10569,15 @@ function schemaError(input, field, reason) { function plainFilePatternRecovery(mode, input, invalid) { return mode === "files" && invalid.includes("pattern") && input.query === undefined && typeof input.pattern === "string" && input.pattern.trim().length > 0 && input.pattern.length <= 256 && input.pattern.isWellFormed() && !/[\\^$.*+?()[\]{}|\r\n\0]/u.test(input.pattern); } +function matchAllFilePatternRecovery(mode, input, invalid) { + return mode === "files" && invalid.includes("pattern") && input.query === undefined && typeof input.pattern === "string" && /^(?:\.\*|\.\+|\*+)$/u.test(input.pattern.trim()); +} function safeNextRequest(input, mode, invalid) { if (input.redact === true && containsSensitiveText(input)) return; const safe = new Set(SAFE_DROP_FIELDS[mode] ?? []); const plainFilePattern = plainFilePatternRecovery(mode, input, invalid); - if (plainFilePattern) + if (plainFilePattern || matchAllFilePatternRecovery(mode, input, invalid)) safe.add("pattern"); if (invalid.some((field) => !safe.has(field))) return; @@ -10291,7 +10623,7 @@ function fieldsError(input, mode, invalid, selectorIssues = []) { issues, recovery: nextRequest ? { action: "retry", - reason: plainFilePatternRecovery(mode, input, invalid) ? "For filename discovery, move the plain text from pattern to query and copy nextRequest; all filters are preserved." : `Remove only ${visibleFields} and copy the exact nextRequest; all other fields are preserved.`, + reason: plainFilePatternRecovery(mode, input, invalid) ? "For filename discovery, move the plain text from pattern to query and copy nextRequest; all filters are preserved." : matchAllFilePatternRecovery(mode, input, invalid) ? "A match-all pattern lists every file; omitting query does that, so copy nextRequest; all filters are preserved." : `Remove only ${visibleFields} and copy the exact nextRequest; all other fields are preserved.`, nextRequest } : { action: "manual", @@ -10447,7 +10779,7 @@ function validateRequestContract(input) { const allowed = new Set(MODE_FIELDS_BY_MODE[mode]); if (mode === "inspect" && raw.sourceCursor !== undefined) { allowed.clear(); - for (const field of ["mode", "sourceCursor", "redact"]) + for (const field of ["mode", "sourceCursor", "redact", "path"]) allowed.add(field); } if ((mode === "auto" || mode === "summary" || mode === "matches") && typeof raw.cursor === "string" && raw.cursor.includes(".analysis")) { @@ -10475,6 +10807,69 @@ import { realpath as realpath6 } from "fs/promises"; function isEvidenceRequest(input) { return input.mode === "concept" || input.mode === "hybrid" || input.mode === "structure" || input.mode === "files" || input.mode === "inspect" || input.mode === "outline" || input.mode === "imports" || input.mode === "tests" || input.mode === "validate" || input.sourceCursor !== undefined || input.anyOf !== undefined || input.allOf !== undefined || input.within !== undefined || input.roles !== undefined || input.changes !== undefined || input.symbol !== undefined || input.conceptLimit !== undefined || (input.cursor?.includes(".analysis") ?? false); } +function pageInspectRequest(input) { + if (input.scope === "expand") + throw new SiftLightError("mode=inspect reads exact locations; scope=expand does not apply (omit scope)"); + const { scope: _scope, paths, ...rest } = input; + let request = rest; + if (paths !== undefined) { + if (input.cursor !== undefined || input.targets !== undefined || input.matchIndices !== undefined || input.path !== undefined || input.line !== undefined || input.matchIndex !== undefined) + throw new SiftLightError("mode=inspect paths opens files from line 1 and cannot be combined with cursor, targets, matchIndices, path, line or matchIndex"); + request = { ...rest, targets: paths.map((path) => ({ path, line: 1 })) }; + } + const redact = input.redact ? { redact: true } : {}; + const requested = request.targets?.length ?? request.matchIndices?.length ?? 0; + if (requested > MAX_INSPECT_REQUEST_TARGETS) + throw new SiftLightError(`mode=inspect accepts at most ${String(MAX_INSPECT_REQUEST_TARGETS)} targets per request; ${String(MAX_INSPECT_TARGETS)} are inspected per response`); + if (request.targets && request.targets.length > MAX_INSPECT_TARGETS) + return { + current: { ...request, targets: request.targets.slice(0, MAX_INSPECT_TARGETS) }, + remaining: { + mode: "inspect", + targets: request.targets.slice(MAX_INSPECT_TARGETS), + ...redact + } + }; + if (request.matchIndices && request.matchIndices.length > MAX_INSPECT_TARGETS) + return { + current: { ...request, matchIndices: request.matchIndices.slice(0, MAX_INSPECT_TARGETS) }, + remaining: { + mode: "inspect", + ...request.cursor !== undefined ? { cursor: request.cursor } : {}, + matchIndices: request.matchIndices.slice(MAX_INSPECT_TARGETS), + ...redact + } + }; + return { current: request }; +} +function withRemainingInspection(result, remaining) { + const count = remaining.targets?.length ?? remaining.matchIndices?.length ?? 0; + return { + text: `${result.text} + +[${String(count)} more target(s) were not inspected; one response inspects ${String(MAX_INSPECT_TARGETS)}.] +Next request: ${JSON.stringify(remaining)}`, + details: { + ...result.details, + status: "partial", + snapshotComplete: false, + nextRequest: remaining + } + }; +} +function analysisItemStructure(item) { + const details = item.details ?? {}; + const language = typeof details.language === "string" ? details.language : undefined; + const name2 = typeof details.name === "string" ? details.name : undefined; + const scope = Array.isArray(details.scope) ? details.scope.filter((entry) => typeof entry === "string") : typeof details.scope === "string" ? [details.scope] : []; + const kind = details.kind === "function" ? "function" : item.label.split(" ", 1)[0] ?? "symbol"; + return { + status: "available", + provider: language === "python" ? "python-outline" : "tree-sitter", + ...language ? { language } : {}, + ...name2 ? { symbol: { name: name2, kind, scope, range: { startLine: item.line, endLine: item.line } } } : {} + }; +} function maxFilesToParse(value, defaultValue = MAX_STRUCTURE_FILES) { const candidate = value ?? defaultValue; if (!Number.isSafeInteger(candidate) || candidate < 1 || candidate > MAX_CONFIGURABLE_STRUCTURE_FILES) { @@ -10490,8 +10885,8 @@ function validateTerms(input) { } return; } - if (!Array.isArray(terms) || terms.length < 2 || terms.length > 3 || terms.some((term) => typeof term !== "string" || !term.trim() || /[\r\n\0]/.test(term)) || new Set(terms).size !== terms.length) - throw new SiftLightError("allOf requires 2\u20133 distinct, nonempty, single-line literal terms"); + if (!Array.isArray(terms) || terms.length < 1 || terms.length > 3 || terms.some((term) => typeof term !== "string" || !term.trim() || /[\r\n\0]/.test(term)) || new Set(terms).size !== terms.length) + throw new SiftLightError("allOf requires 1\u20133 distinct, nonempty, single-line literal terms"); if (input.pattern !== undefined || input.roles !== undefined || input.literal !== undefined || input.ignoreCase !== undefined || input.wholeWord !== undefined) throw new SiftLightError("allOf is an explicit case-sensitive literal conjunction; omit pattern, roles, literal and ignoreCase"); if (input.within !== undefined && input.within !== "file" && input.within !== "function") @@ -10689,11 +11084,13 @@ class EvidenceService { throw new CursorError("A nonempty sourceCursor is required"); if (input.mode !== "inspect") throw new SiftLightError("sourceCursor requires mode=inspect"); - return continueSource(input.sourceCursor, access, this.#continuations); + return continueSource(input.sourceCursor, access, this.#continuations, input.path); } if (input.mode === "inspect") { - const targets = this.#inspectionTargets(input, cwd); - return targets.some((target) => target.range !== undefined) ? inspectDocumentsMetadata(targets, access, this.#structure) : inspectDocuments(targets, access, this.#continuations, this.#structure); + const { current, remaining } = pageInspectRequest(input); + const targets = this.#inspectionTargets(current, cwd); + const result = await inspectDocuments(targets, access, this.#continuations, this.#structure); + return remaining ? withRemainingInspection(result, remaining) : result; } if (input.mode === "concept") { const execution = await this.#conceptSearch(input, access, options.onProgress); @@ -10782,7 +11179,7 @@ class EvidenceService { if (input.pattern !== undefined || input.allOf !== undefined || input.within !== undefined || input.roles !== undefined || input.literal !== undefined || input.ignoreCase !== undefined || input.wholeWord !== undefined) throw new SiftLightError("anyOf is an explicit case-sensitive literal union; omit pattern, allOf, within, roles, literal and ignoreCase"); if (input.mode !== undefined && input.mode !== "auto" && input.mode !== "matches") - throw new SiftLightError("anyOf mode must be omitted, auto, or matches"); + throw new SiftLightError("anyOf mode must be omitted, auto, matches or summary"); const chunks = Array.from({ length: Math.ceil(anyOf.length / MAX_ANY_OF_TERMS) }, (_, index) => anyOf.slice(index * MAX_ANY_OF_TERMS, (index + 1) * MAX_ANY_OF_TERMS)); const { path: _inputPath, ...unscopedInput } = input; let chunkAccess = access; @@ -10920,9 +11317,10 @@ class EvidenceService { if (item) result.items.push(item); } else if (terms || input.roles) { - if (syntaxLanguage(file.document.path)) + const pythonRoles = !terms && /\.py$/iu.test(file.document.path); + if (pythonRoles || syntaxLanguage(file.document.path)) syntaxCapableFiles += 1; - const syntax = await access.syntax(file.document); + const syntax = pythonRoles ? pythonRoleAnalysis(file.document) : await access.syntax(file.document); const classified = terms ? findFunctionConjunctions(file.document, syntax, terms, input.changes?.scope === "lines" ? file.changedRanges : undefined) : filterRoleOccurrences(file.document, syntax, file.occurrences, input.roles ?? []); result.items.push(...classified.items); result.partial ||= classified.partial; @@ -10999,12 +11397,12 @@ class EvidenceService { const item = this.#analyses.item(input.cursor, input.matchIndex); if (!item.source || !item.range) throw new CursorError("This analysis item has no verified source range"); - const metadataOnly = item.details?.kind === "symbol" || item.details?.kind === "function"; + const bounded = item.details?.kind === "symbol" || item.details?.kind === "function"; return { path: item.path, line: item.line, reference: item.source, - ...metadataOnly ? { range: item.range } : { absoluteFocus: item.range.start } + ...bounded ? { range: item.range, structure: analysisItemStructure(item) } : { absoluteFocus: item.range.start } }; } return { @@ -11213,6 +11611,29 @@ import { resolve as resolve24 } from "path"; // src/discovery-errors.ts var DISCOVERY_MODE_REQUIRED_ERROR = 'query requires an explicit discovery mode: use mode=files for filename/path discovery or mode=concept for semantic discovery; for example {"mode":"files","query":""}'; +// src/request-aliases.ts +var MODE_ALIASES = ["anyOf", "allOf"]; +function normalizeRequestAliases(request) { + const notes = []; + const { mode, ...rest } = request; + let input = rest; + if (mode === "anyOf" || mode === "allOf") { + if (request[mode] === undefined) + throw new SiftLightError(`mode=${mode} is not a mode; pass ${mode}:[...terms] and omit mode`); + notes.push(`mode="${mode}" is not a mode; the ${mode} field alone selects this search.`); + } else if (mode === "summary" && request.anyOf !== undefined) { + notes.push("anyOf returns one analysis page whose header already carries per-file statistics; mode=summary was served by that page."); + } else if (mode !== undefined) { + input = { ...rest, mode }; + } + if (input.sourceCursor !== undefined && input.line !== undefined) { + const { line: _line, ...withoutLine } = input; + input = withoutLine; + notes.push("line is ignored with sourceCursor; the continuation selects its own source range."); + } + return { input, notes }; +} + // src/format.ts import { readFile as readFile3 } from "fs/promises"; var RESULT_METADATA_RESERVE_BYTES = 1024; @@ -12566,6 +12987,16 @@ class LanguageCapabilityCatalog { } // src/service.ts +function withRequestNotes(result, notes) { + if (notes.length === 0) + return result; + return { + text: `${notes.map((note) => `[Request note: ${note}]`).join(` +`)} +${result.text}`, + details: { ...result.details, requestNotes: [...notes] } + }; +} function filterList(value) { return value === undefined ? [] : Array.isArray(value) ? [...value] : [value]; } @@ -12794,7 +13225,11 @@ class SiftLightService { this.#operations = new OperationLifecycle({ deadlineMs: resolveConceptTimeoutMs() }); this.#evidence = new EvidenceService(this.#runRipgrep, this.#snapshots, options.structure, options.conceptSearch, options.semanticJudge); } - async search(input, cwd, signal, options = {}) { + async search(request, cwd, signal, options = {}) { + const { input, notes } = normalizeRequestAliases(request); + return withRequestNotes(await this.#searchNormalized(input, cwd, signal, options), notes); + } + async#searchNormalized(input, cwd, signal, options) { validateRawSearchInput(input); validateRequestContract(input); if (!this.#vectorSearchEnabled && (input.mode === "concept" || input.mode === "hybrid")) { @@ -20586,9 +21021,9 @@ var siftLightSchema = _Object_({ description: `${fieldGuidance("anyOf")}. ${String(MIN_ANY_OF_TERMS)}-${String(MAX_ANY_OF_TOTAL_TERMS)} distinct case-sensitive single-line terms, at most ${String(MAX_LITERAL_TERM_BYTES)} UTF-8 bytes each. Requests above ${String(MAX_ANY_OF_TERMS)} terms are split into version-checked chunks and merged. Returns every retained occurrence attributed to its term.` })), allOf: Optional(_Array_(String2({ maxLength: MAX_PATH_CHARACTERS }), { - minItems: 2, + minItems: 1, maxItems: 3, - description: `${fieldGuidance("allOf")}. 2-3 distinct terms must occur in one file (default) or one function.` + description: `${fieldGuidance("allOf")}. 1-3 distinct terms must occur in one file (default) or one function; one term lists the files or functions containing it.` })), within: Optional(stringEnum(["file", "function"], { description: "Only valid with allOf; omit for ordinary single-pattern searches. function requires JS/TS/TSX and counts only that implementation's own code, excluding nested callbacks, strings/comments/types. Not proof of a shared execution path." @@ -20605,7 +21040,7 @@ var siftLightSchema = _Object_({ "unknown" ]), { minItems: 1, - description: "Filter each single-pattern occurrence by syntax role (JS/TS/TSX/Go). Roles may be candidates, especially Go call/conversion ambiguity. Cannot combine with allOf." + description: "Filter each single-pattern occurrence by syntax role (JS/TS/TSX/Go parsed; Python lexical: comment, string, code, declaration, import, call candidates). Roles may be candidates, especially Go call/conversion and Python calls. Cannot combine with allOf." })), changes: Optional(_Object_({ base: Optional(String2({ @@ -20638,7 +21073,7 @@ var siftLightSchema = _Object_({ paths: Optional(_Array_(String2(), { minItems: 1, maxItems: MAX_SELECTED_PATHS, - description: "Exact retained files to select together from a cursor. A new search accepts one path; split multiple roots into separate requests." + description: "Exact retained files to select together from a cursor. A new search accepts one path; split multiple roots into separate requests. With mode=inspect and no cursor, opens each file from line 1 (same as targets with line 1)." })), glob: Optional(Union([ String2({ maxLength: MAX_PATH_CHARACTERS }), @@ -20662,7 +21097,7 @@ var siftLightSchema = _Object_({ })), hidden: Optional(Boolean2({ description: "Search hidden files (default true; .git is always excluded)." })), ignorePolicy: Optional(stringEnum(["respect", "include"], { - description: "respect (default) honors ignore rules and reports policy-filtered coverage when files are omitted. include searches ignored files while still excluding .git internals and protected paths." + description: "respect (default) honors ignore rules and reports policy-filtered coverage when files are omitted, including mode=files. include searches or lists ignored files while still excluding .git internals and protected paths." })), patterns: Optional(_Array_(_Object_({ id: String2({ minLength: 1, maxLength: 64 }), @@ -20707,29 +21142,29 @@ var siftLightSchema = _Object_({ limit: Optional(Integer({ minimum: 1, maximum: MAX_PAGE_SIZE, - description: `${fieldGuidance("limit")}. Ordinary search only: explicit detail-page match limit (max 100). Normally omit to preserve automatic summarization; analysis and inspect modes reject it.` + description: `${fieldGuidance("limit")}. Ordinary search: explicit detail-page match limit (max 100); mode=files: files per page (default 30). Normally omit to preserve automatic summarization; other analysis and inspect modes reject it.` })), - mode: Optional(stringEnum(SIFT_LIGHT_MODES, { - description: `Ordinary search defaults to auto; summary/matches request explicit pages. capabilities returns a compact names-only project inventory and per-language supported modes without loading providers. files uses query, structure uses an AST pattern, concept uses a required natural-language query, and hybrid uses one query for exact literal plus concept evidence in a single snapshot. validate rechecks saved search or analysis sources against their recorded origin. await waits for an existing long-running concept or hybrid operation without restarting it; cancel explicitly cancels one and waits for owned cleanup. Copy the returned nextRequest exactly and do not repeat the original query. Waiting is an operation state, not evidence. Validation details report the requested scope, comparison target, coverage, and freshness as current, stale, or unknown; partial coverage is retained during validation. inspect/outline/imports/tests retain their documented location selectors. Syntax results are static evidence; concept and related-test results remain candidates. ${MODE_CONTRACT_DESCRIPTION}` + mode: Optional(stringEnum([...SIFT_LIGHT_MODES, ...MODE_ALIASES], { + description: `Ordinary search defaults to auto; summary/matches request explicit pages. capabilities returns a compact names-only project inventory and per-language supported modes without loading providers. files uses query, structure uses an AST pattern, concept uses a required natural-language query, and hybrid uses one query for exact literal plus concept evidence in a single snapshot. validate rechecks saved search or analysis sources against their recorded origin. await waits for an existing long-running concept or hybrid operation without restarting it; cancel explicitly cancels one and waits for owned cleanup. Copy the returned nextRequest exactly and do not repeat the original query. Waiting is an operation state, not evidence. Validation details report the requested scope, comparison target, coverage, and freshness as current, stale, or unknown; partial coverage is retained during validation. inspect/outline/imports/tests retain their documented location selectors. Syntax results are static evidence; concept and related-test results remain candidates. ${MODE_CONTRACT_DESCRIPTION} anyOf/allOf are field names, not modes; as mode values they are accepted aliases for omitting mode, disclosed in a request note.` })), line: Optional(Number2({ - description: "1-indexed source line for path inspection/navigation. Omit with matchIndex, matchIndices or targets." + description: "1-indexed source line for path inspection/navigation; path-only inspect starts at line 1. Omit with matchIndex, matchIndices or targets." })), matchIndex: Optional(Number2({ description: "1-based retained match index for cursor-scoped inspect; replaces path and line." })), matchIndices: Optional(_Array_(Integer({ minimum: 1 }), { minItems: 1, - maxItems: MAX_INSPECT_TARGETS, - description: "Inspect up to five visible match numbers together using the same cursor; mutually exclusive with matchIndex, path, line and targets." + maxItems: MAX_INSPECT_REQUEST_TARGETS, + description: `Inspect visible match numbers together using the same cursor; mutually exclusive with matchIndex, path, line and targets. One response inspects ${String(MAX_INSPECT_TARGETS)}; the rest are returned as an exact nextRequest.` })), targets: Optional(_Array_(_Object_({ path: String2({ maxLength: MAX_PATH_CHARACTERS }), line: Integer({ minimum: 1 }) }), { minItems: 1, - maxItems: MAX_INSPECT_TARGETS, - description: "Inspect known path/line locations together without a cursor. The complete batch shares one 16 KiB response budget." + maxItems: MAX_INSPECT_REQUEST_TARGETS, + description: `Inspect known path/line locations together without a cursor. One response inspects ${String(MAX_INSPECT_TARGETS)} within a shared 16 KiB budget; the rest are returned as an exact nextRequest.` })), cursor: Optional(String2({ description: "Opaque cursor from a previous stable search snapshot." })), operationId: Optional(String2({ diff --git a/plugins/sift-light/package.json b/plugins/sift-light/package.json index 4ac4f7c0..10f7fcb9 100644 --- a/plugins/sift-light/package.json +++ b/plugins/sift-light/package.json @@ -1,6 +1,6 @@ { "name": "sift-light", - "version": "1.0.1", + "version": "1.0.2", "private": true, "type": "module", "omp": { diff --git a/src/analysis-limits.ts b/src/analysis-limits.ts index 1cc9fe12..ce7d6c02 100644 --- a/src/analysis-limits.ts +++ b/src/analysis-limits.ts @@ -13,7 +13,7 @@ export const MAX_ANALYSIS_STORAGE_BYTES = 32 * 1024 * 1024; export const ANALYSIS_METADATA_RESERVE_BYTES = 64 * 1024; export const MAX_ANALYSIS_REASONS = 64; export const MAX_ANALYSIS_REASON_BYTES = 4 * 1024; -export const MIN_ANY_OF_TERMS = 2; +export const MIN_ANY_OF_TERMS = 1; export const MAX_ANY_OF_TERMS = 8; /** Public union size; execution keeps the proven eight-term scan boundary per chunk. */ export const MAX_ANY_OF_TOTAL_TERMS = 64; diff --git a/src/analysis-store.ts b/src/analysis-store.ts index 6de6acfd..d8db67a7 100644 --- a/src/analysis-store.ts +++ b/src/analysis-store.ts @@ -206,7 +206,10 @@ function publicAnalysisItem( return { path: item.path, line: item.line, - label: result.kind === "outline" && modelOutput ? item.label : publicAnalysisLabel(result), + label: + (result.kind === "outline" && modelOutput) || result.kind === "files" + ? item.label + : publicAnalysisLabel(result), index: index + 1, ...(inspect ? { inspect } : {}), ...(publicDetails ? { details: publicDetails } : {}), @@ -455,7 +458,8 @@ export class AnalysisStore { if (items.length >= 30 || !appendItem(index)) break; } } else { - for (let index = offset; index < result.items.length && items.length < 30; index += 1) { + const pageSize = result.pageSize ?? 30; + for (let index = offset; index < result.items.length && items.length < pageSize; index += 1) { if (!appendItem(index)) break; } } diff --git a/src/analysis-types.ts b/src/analysis-types.ts index 5aebab7e..a6ac7060 100644 --- a/src/analysis-types.ts +++ b/src/analysis-types.ts @@ -4,7 +4,12 @@ import type { SearchScopeDetails, ResultStatistics } from "./types.js"; import type { ValidationDetails } from "./validation-types.js"; import type { ConceptSourceSummary } from "./concept-source-generation.js"; -export type CoverageStatus = "complete" | "partial" | "skipped" | "not-applicable"; +export type CoverageStatus = + | "complete" + | "partial" + | "policy-filtered" + | "skipped" + | "not-applicable"; export const SEMANTIC_JUDGE_CLASSIFICATIONS = [ "implementation-candidate", @@ -125,6 +130,8 @@ export interface AnalysisDetails { passagesRanked?: number; elapsedMs?: number; filesEnumerated?: number; + ignoredFiles?: number; + ignoredMatches?: number; filesAdmitted?: number; filesParsed?: number; filesSkipped?: number; @@ -159,4 +166,6 @@ export interface AnalysisResultSet { stats?: AnalysisDetails["stats"]; sourceGeneration?: ConceptSourceSummary; redact?: boolean; + /** Explicit output page size (`limit`); pages default to 30 items. */ + pageSize?: number; } diff --git a/src/evidence-service.ts b/src/evidence-service.ts index 582b2e74..d3e99d0c 100644 --- a/src/evidence-service.ts +++ b/src/evidence-service.ts @@ -39,7 +39,6 @@ import { continueSource, inspectDocuments, matchInspectionTarget, - inspectDocumentsMetadata, type SourceInspectionTarget, } from "./source-inspection.js"; import { validateSavedEvidence } from "./evidence-validation.js"; @@ -48,6 +47,7 @@ import type { CodeStructureProvider } from "./structure.js"; import { filterRoleOccurrences, findFunctionConjunctions } from "./syntax-search.js"; import { syntaxLanguage } from "./syntax.js"; import { parsePythonOutline } from "./python-outline.js"; +import { pythonRoleAnalysis } from "./python-roles.js"; import { combineHybridSearch, hybridConceptLimit, @@ -66,10 +66,12 @@ import { MAX_STRUCTURE_FILES, } from "./analysis-limits.js"; import { + MAX_INSPECT_REQUEST_TARGETS, MAX_INSPECT_TARGETS, type SearchRequest, type SearchScopeDetails, type SiftLightResult, + type StructureDetails, } from "./types.js"; export function isEvidenceRequest(input: SiftLightInput): boolean { @@ -95,6 +97,100 @@ export function isEvidenceRequest(input: SiftLightInput): boolean { ); } +/** + * Splits one inspect request into the page served now and an exact request for + * the rest. `paths` without a cursor opens each file from line 1. + */ +function pageInspectRequest(input: SiftLightInput): { + current: SiftLightInput; + remaining?: SiftLightInput; +} { + if (input.scope === "expand") + throw new SiftLightError( + "mode=inspect reads exact locations; scope=expand does not apply (omit scope)", + ); + const { scope: _scope, paths, ...rest } = input; + let request: SiftLightInput = rest; + if (paths !== undefined) { + if ( + input.cursor !== undefined || + input.targets !== undefined || + input.matchIndices !== undefined || + input.path !== undefined || + input.line !== undefined || + input.matchIndex !== undefined + ) + throw new SiftLightError( + "mode=inspect paths opens files from line 1 and cannot be combined with cursor, targets, matchIndices, path, line or matchIndex", + ); + request = { ...rest, targets: paths.map((path) => ({ path, line: 1 })) }; + } + const redact = input.redact ? { redact: true } : {}; + const requested = request.targets?.length ?? request.matchIndices?.length ?? 0; + if (requested > MAX_INSPECT_REQUEST_TARGETS) + throw new SiftLightError( + `mode=inspect accepts at most ${String(MAX_INSPECT_REQUEST_TARGETS)} targets per request; ${String(MAX_INSPECT_TARGETS)} are inspected per response`, + ); + if (request.targets && request.targets.length > MAX_INSPECT_TARGETS) + return { + current: { ...request, targets: request.targets.slice(0, MAX_INSPECT_TARGETS) }, + remaining: { + mode: "inspect", + targets: request.targets.slice(MAX_INSPECT_TARGETS), + ...redact, + }, + }; + if (request.matchIndices && request.matchIndices.length > MAX_INSPECT_TARGETS) + return { + current: { ...request, matchIndices: request.matchIndices.slice(0, MAX_INSPECT_TARGETS) }, + remaining: { + mode: "inspect", + ...(request.cursor !== undefined ? { cursor: request.cursor } : {}), + matchIndices: request.matchIndices.slice(MAX_INSPECT_TARGETS), + ...redact, + }, + }; + return { current: request }; +} + +/** The unserved targets stay explicit: the page is partial and names its exact continuation. */ +function withRemainingInspection( + result: SiftLightResult, + remaining: SiftLightInput, +): SiftLightResult { + const count = remaining.targets?.length ?? remaining.matchIndices?.length ?? 0; + return { + text: `${result.text}\n\n[${String(count)} more target(s) were not inspected; one response inspects ${String(MAX_INSPECT_TARGETS)}.]\nNext request: ${JSON.stringify(remaining)}`, + details: { + ...result.details, + status: "partial", + snapshotComplete: false, + nextRequest: remaining, + }, + }; +} + +/** Structure facts already proven by the analysis that produced a bounded item. */ +function analysisItemStructure(item: AnalysisItem): StructureDetails { + const details = item.details ?? {}; + const language = typeof details.language === "string" ? details.language : undefined; + const name = typeof details.name === "string" ? details.name : undefined; + const scope = Array.isArray(details.scope) + ? details.scope.filter((entry): entry is string => typeof entry === "string") + : typeof details.scope === "string" + ? [details.scope] + : []; + const kind = details.kind === "function" ? "function" : (item.label.split(" ", 1)[0] ?? "symbol"); + return { + status: "available", + provider: language === "python" ? "python-outline" : "tree-sitter", + ...(language ? { language } : {}), + ...(name + ? { symbol: { name, kind, scope, range: { startLine: item.line, endLine: item.line } } } + : {}), + }; +} + export interface EvidenceSearchOptions { modelOutput?: boolean; onProgress?: (progress: OperationProgress) => void; @@ -126,12 +222,12 @@ function validateTerms(input: SiftLightInput): string[] | undefined { } if ( !Array.isArray(terms) || - terms.length < 2 || + terms.length < 1 || terms.length > 3 || terms.some((term) => typeof term !== "string" || !term.trim() || /[\r\n\0]/.test(term)) || new Set(terms).size !== terms.length ) - throw new SiftLightError("allOf requires 2–3 distinct, nonempty, single-line literal terms"); + throw new SiftLightError("allOf requires 1–3 distinct, nonempty, single-line literal terms"); if ( input.pattern !== undefined || input.roles !== undefined || @@ -396,13 +492,13 @@ export class EvidenceService { if (typeof input.sourceCursor !== "string" || !input.sourceCursor.trim()) throw new CursorError("A nonempty sourceCursor is required"); if (input.mode !== "inspect") throw new SiftLightError("sourceCursor requires mode=inspect"); - return continueSource(input.sourceCursor, access, this.#continuations); + return continueSource(input.sourceCursor, access, this.#continuations, input.path); } if (input.mode === "inspect") { - const targets = this.#inspectionTargets(input, cwd); - return targets.some((target) => target.range !== undefined) - ? inspectDocumentsMetadata(targets, access, this.#structure) - : inspectDocuments(targets, access, this.#continuations, this.#structure); + const { current, remaining } = pageInspectRequest(input); + const targets = this.#inspectionTargets(current, cwd); + const result = await inspectDocuments(targets, access, this.#continuations, this.#structure); + return remaining ? withRemainingInspection(result, remaining) : result; } if (input.mode === "concept") { const execution = await this.#conceptSearch(input, access, options.onProgress); @@ -518,7 +614,7 @@ export class EvidenceService { "anyOf is an explicit case-sensitive literal union; omit pattern, allOf, within, roles, literal and ignoreCase", ); if (input.mode !== undefined && input.mode !== "auto" && input.mode !== "matches") - throw new SiftLightError("anyOf mode must be omitted, auto, or matches"); + throw new SiftLightError("anyOf mode must be omitted, auto, matches or summary"); const chunks = Array.from( { length: Math.ceil(anyOf.length / MAX_ANY_OF_TERMS) }, (_, index) => anyOf.slice(index * MAX_ANY_OF_TERMS, (index + 1) * MAX_ANY_OF_TERMS), @@ -696,8 +792,12 @@ export class EvidenceService { ); if (item) result.items.push(item); } else if (terms || input.roles) { - if (syntaxLanguage(file.document.path)) syntaxCapableFiles += 1; - const syntax = await access.syntax(file.document); + // Python roles come from a bounded lexical scanner; other languages parse. + const pythonRoles = !terms && /\.py$/iu.test(file.document.path); + if (pythonRoles || syntaxLanguage(file.document.path)) syntaxCapableFiles += 1; + const syntax = pythonRoles + ? pythonRoleAnalysis(file.document) + : await access.syntax(file.document); const classified = terms ? findFunctionConjunctions( file.document, @@ -790,12 +890,16 @@ export class EvidenceService { const item = this.#analyses.item(input.cursor, input.matchIndex); if (!item.source || !item.range) throw new CursorError("This analysis item has no verified source range"); - const metadataOnly = item.details?.kind === "symbol" || item.details?.kind === "function"; + // Outline symbols and function conjunctions carry their exact syntax range, + // so inspection returns that version-checked range as source (issue #96). + const bounded = item.details?.kind === "symbol" || item.details?.kind === "function"; return { path: item.path, line: item.line, reference: item.source, - ...(metadataOnly ? { range: item.range } : { absoluteFocus: item.range.start }), + ...(bounded + ? { range: item.range, structure: analysisItemStructure(item) } + : { absoluteFocus: item.range.start }), }; } return { diff --git a/src/file-discovery.ts b/src/file-discovery.ts index e0f579bf..ed67905e 100644 --- a/src/file-discovery.ts +++ b/src/file-discovery.ts @@ -13,6 +13,42 @@ interface FileScore { const graphemes = new Intl.Segmenter(undefined, { granularity: "grapheme" }); /** Bracket classes remain valid query text, so dynamic route names stay searchable. */ const FILE_QUERY_GLOB_WILDCARD = /[*?]/u; +/** Queries such as `*`, `**` or `.*` mean "every file" rather than filename text. */ +const FILE_QUERY_MATCH_ALL = /^(?:[*/]+|\.\*|\.\+|\*\.\*)$/u; +/** Regex syntax that cannot be a simple glob; such a query is rejected with guidance. */ +const FILE_QUERY_REGEX_SYNTAX = /[()|^$\\+]/u; +const IGNORED_MATCH_SAMPLES = 5; + +interface NormalizedFileQuery { + query: string; + glob?: string; + note?: string; +} + +/** + * Filename discovery matches text. A wildcard-only query means "all files" and a + * plain glob such as `*.py` becomes an explicit glob filter; both are disclosed. + */ +function normalizeFileQuery(query: string, hasGlob: boolean): NormalizedFileQuery { + const trimmed = query.trim(); + if (FILE_QUERY_MATCH_ALL.test(trimmed)) + return { + query: "", + note: `Query ${JSON.stringify(query)} was treated as listing every file under path; omit query for the same result.`, + }; + if (!FILE_QUERY_GLOB_WILDCARD.test(trimmed)) return { query }; + if (!hasGlob && !FILE_QUERY_REGEX_SYNTAX.test(trimmed) && !/\s/u.test(trimmed)) { + const glob = trimmed.includes("/") ? trimmed : `**/${trimmed}`; + return { + query: "", + glob, + note: `Query ${JSON.stringify(query)} contains glob wildcards and was applied as glob ${JSON.stringify(glob)}; mode=files query otherwise matches filename text.`, + }; + } + throw new SiftLightError( + `File query ${JSON.stringify(query)} uses wildcard or regex syntax; mode=files matches filename and path text, not patterns. Omit query to retain every file under path, use glob for one name pattern, or run one files request per name.`, + ); +} function subsequenceScore(text: string, query: string): number | undefined { const characters = Array.from(graphemes.segment(text), (item) => item.segment); @@ -70,26 +106,33 @@ export async function discoverFiles( cwd: string, signal?: AbortSignal, ): Promise { - const query = input.query ?? ""; - if (query.length > 256 || !query.isWellFormed() || /[\r\n\0]/.test(query)) + const rawQuery = input.query ?? ""; + if (rawQuery.length > 256 || !rawQuery.isWellFormed() || /[\r\n\0]/.test(rawQuery)) throw new SiftLightError( "File query must be well-formed single-line text of at most 256 characters", ); - // A wildcard query can never match filename text, so scoring it would discard - // every enumerated file and report an empty result as complete. - if (FILE_QUERY_GLOB_WILDCARD.test(query)) - throw new SiftLightError( - `File query ${JSON.stringify(query)} uses glob wildcards; mode=files matches filename and path text, not glob patterns. Omit query to retain every file under path, or use glob to filter by name pattern.`, - ); - const request = normalizeRequest({ ...input, pattern: "" }); + const requestedGlobs = + input.glob === undefined ? [] : Array.isArray(input.glob) ? input.glob : [input.glob]; + const normalizedQuery = normalizeFileQuery(rawQuery, requestedGlobs.length > 0); + const query = normalizedQuery.query; + const request = normalizeRequest({ + ...input, + pattern: "", + ...(normalizedQuery.glob ? { glob: [...requestedGlobs, normalizedQuery.glob] } : {}), + }); const policy = new SearchPathPolicy(cwd); const discoveryRoot = request.path ?? "."; const scoringRoot = await policy.resolveSearchTarget(discoveryRoot); - const files = await listWorkspaceFiles(cwd, signal, { + const includeIgnored = request.ignorePolicy === "include"; + const listOptions = { path: scoringRoot, glob: request.glob, exclude: request.exclude, hidden: request.hidden, + }; + const files = await listWorkspaceFiles(cwd, signal, { + ...listOptions, + ...(includeIgnored ? { ignore: false, ignoreParents: false } : {}), }); const filtered = await filterPathsByModificationTime( cwd, @@ -98,10 +141,12 @@ export async function discoverFiles( request.modifiedBeforeMs, signal, ); + const rank = (path: string) => + scoreFilePath(pathRelativeToDiscoveryRoot(cwd, scoringRoot, path), query); const selected = filtered.paths .flatMap((path) => { - const rank = scoreFilePath(pathRelativeToDiscoveryRoot(cwd, scoringRoot, path), query); - return rank ? [{ path, ...rank }] : []; + const score = rank(path); + return score ? [{ path, ...score }] : []; }) .toSorted((left, right) => right.score - left.score || left.path.localeCompare(right.path)); for (let offset = 0; offset < selected.length; offset += 16) { @@ -111,15 +156,54 @@ export async function discoverFiles( ); signal?.throwIfAborted(); } + const reasons = new Set([...files.reasons, ...filtered.reasons]); + if (normalizedQuery.note) reasons.add(normalizedQuery.note); + // Issue #97: ignore rules must never turn a filtered absence into a complete one. + let ignoredFiles = 0; + let ignoredMatches = 0; + let ignoredComparisonPartial = false; + if (!includeIgnored) { + const everything = await listWorkspaceFiles(cwd, signal, { + ...listOptions, + ignore: false, + ignoreParents: false, + }); + ignoredComparisonPartial = everything.partial; + const admitted = new Set(files.paths); + const ignored = everything.paths.filter((path) => !admitted.has(path)); + ignoredFiles = ignored.length; + const ignoredCandidates = ignored + .flatMap((path) => { + const score = rank(path); + return score ? [{ path, ...score }] : []; + }) + .toSorted((left, right) => right.score - left.score || left.path.localeCompare(right.path)); + ignoredMatches = ignoredCandidates.length; + if (ignoredFiles > 0) { + const samples = ignoredCandidates + .slice(0, IGNORED_MATCH_SAMPLES) + .map((item) => `${JSON.stringify(item.path)} (${item.reason})`); + reasons.add( + ignoredMatches > 0 + ? `${String(ignoredFiles)} file(s) were excluded by ignore rules; ${String(ignoredMatches)} of them match this query${samples.length ? `, e.g. ${samples.join(", ")}` : ""}. Retry with ignorePolicy "include" to list them.` + : `${String(ignoredFiles)} file(s) were excluded by ignore rules; none of them match this query.`, + ); + } + if (ignoredComparisonPartial) + reasons.add( + "Ignored-file comparison was incomplete; the ignored-file count is a lower bound", + ); + } + const enumerationPartial = files.partial || filtered.partial; return { kind: "files", unit: "files", - partial: files.partial || filtered.partial, - reasons: [...new Set([...files.reasons, ...filtered.reasons])], + partial: enumerationPartial, + reasons: [...reasons], items: selected.map((item) => ({ path: item.path, line: 1, - label: `File candidate (${item.reason})`, + label: `File candidate (score ${String(item.score)}: ${item.reason})`, details: { kind: "file", score: item.score, @@ -127,8 +211,17 @@ export async function discoverFiles( inspect: { mode: "inspect", path: item.path, line: 1 }, }, })), - coverage: { fileEnumeration: files.partial || filtered.partial ? "partial" : "complete" }, - stats: { filesEnumerated: files.paths.length }, + coverage: { + fileEnumeration: enumerationPartial + ? "partial" + : ignoredFiles > 0 || ignoredComparisonPartial + ? "policy-filtered" + : "complete", + }, + stats: { + filesEnumerated: files.paths.length, + ...(includeIgnored ? {} : { ignoredFiles, ignoredMatches }), + }, scope: { path: request.path ?? ".", requestedPath: request.path ?? ".", @@ -146,5 +239,6 @@ export async function discoverFiles( : {}), }, redact: input.redact ?? false, + ...(input.limit !== undefined ? { pageSize: request.pageSize } : {}), }; } diff --git a/src/inspect.ts b/src/inspect.ts index a8fb2cbf..ce9b599c 100644 --- a/src/inspect.ts +++ b/src/inspect.ts @@ -78,6 +78,9 @@ export function resolveInspectionTarget( line = retainedMatch.lineNumber; } if (!path) throw new SiftLightError("path is required when mode=inspect"); + // A direct path without a line opens the file from its first line. Cursor + // inspection still needs the exact retained line, so it keeps failing closed. + if (line === undefined && input.cursor === undefined) line = 1; if (line === undefined || !Number.isSafeInteger(line) || line < 1) { throw new SiftLightError("line must be a positive integer when mode=inspect"); } diff --git a/src/language-capability-definitions.ts b/src/language-capability-definitions.ts index 058d4fa5..f502bbfa 100644 --- a/src/language-capability-definitions.ts +++ b/src/language-capability-definitions.ts @@ -119,6 +119,15 @@ const PYTHON_LANGUAGE_CAPABILITIES: readonly LanguageCapabilitySpec[] = [ availability: "implemented", load: "lazy", }, + { + id: "python-lexical.roles", + name: "roles", + provider: "bounded Python lexical role scanner", + providerKind: "builtin", + evidence: "syntax", + availability: "implemented", + load: "lazy", + }, ]; export const DEFAULT_LANGUAGE_CAPABILITIES: readonly LanguageCapabilityDescriptor[] = [ diff --git a/src/mcp-model-output.ts b/src/mcp-model-output.ts index 9f8966cd..c47fca13 100644 --- a/src/mcp-model-output.ts +++ b/src/mcp-model-output.ts @@ -87,7 +87,10 @@ function compactRows(analysis: AnalysisDetails): string[] { rows.push(JSON.stringify(item.path)); previousPath = item.path; } - const label = analysis.kind === "outline" && analysis.modelOutput ? item.label : "metadata"; + const label = + (analysis.kind === "outline" && analysis.modelOutput) || analysis.kind === "files" + ? item.label + : "metadata"; const semanticJudge = item.details?.semanticJudge; const judgment = isRecord(semanticJudge) && @@ -149,6 +152,7 @@ export function compactMcpModelText(result: SiftLightResult): string { const inspect = compactInspectInstruction(analysis); const nextRequest = distinctNextRequest(result.details, analysis); const compact = [ + ...(result.details.requestNotes ?? []).map((note) => `[Request note: ${note}]`), header, ...compactMetadata(result.details, analysis), ...compactRows(analysis), @@ -156,6 +160,7 @@ export function compactMcpModelText(result: SiftLightResult): string { ...(nextRequest ? [`Next request: ${nextRequest}`] : []), ].join("\n"); const standard = result.text.replace(" Structured output retains per-item evidence details.", ""); + // Both projections carry request notes, so the byte comparison stays fair. if (analysis.semanticJudge) return compact; return Buffer.byteLength(compact) < Buffer.byteLength(standard) ? compact : standard; } diff --git a/src/mcp-server.mjs b/src/mcp-server.mjs index fbad10ef..f5e17d42 100755 --- a/src/mcp-server.mjs +++ b/src/mcp-server.mjs @@ -7,7 +7,7 @@ import { URL as URL2 } from "node:url"; // package.json var package_default = { name: "sift-light", - version: "1.0.1", + version: "1.0.2", description: "Context-efficient local search for files, documents, notes and logs across Pi, OMP and MCP clients", keywords: [ "ai-agent", @@ -346,7 +346,7 @@ function compactRows(analysis) { rows.push(JSON.stringify(item.path)); previousPath = item.path; } - const label = analysis.kind === "outline" && analysis.modelOutput ? item.label : "metadata"; + const label = analysis.kind === "outline" && analysis.modelOutput || analysis.kind === "files" ? item.label : "metadata"; const semanticJudge = item.details?.semanticJudge; const judgment = isRecord(semanticJudge) && typeof semanticJudge.classification === "string" && typeof semanticJudge.probability === "number" && typeof semanticJudge.model === "string" ? ` Jev: ${semanticJudge.classification}; probability=${String(semanticJudge.probability)};${typeof semanticJudge.confidence === "number" ? ` confidence=${String(semanticJudge.confidence)};` : ""} model=${JSON.stringify(semanticJudge.model)}.` : ""; rows.push(`#${String(item.index)} L${String(item.line)} ${label}${judgment}`); @@ -386,6 +386,7 @@ function compactMcpModelText(result) { const inspect = compactInspectInstruction(analysis); const nextRequest = distinctNextRequest(result.details, analysis); const compact = [ + ...(result.details.requestNotes ?? []).map((note) => `[Request note: ${note}]`), header, ...compactMetadata(result.details, analysis), ...compactRows(analysis), @@ -417,7 +418,7 @@ var MAX_ANALYSIS_STORAGE_BYTES = 32 * 1024 * 1024; var ANALYSIS_METADATA_RESERVE_BYTES = 64 * 1024; var MAX_ANALYSIS_REASONS = 64; var MAX_ANALYSIS_REASON_BYTES = 4 * 1024; -var MIN_ANY_OF_TERMS = 2; +var MIN_ANY_OF_TERMS = 1; var MAX_ANY_OF_TERMS = 8; var MAX_ANY_OF_TOTAL_TERMS = 64; var MAX_LITERAL_TERM_BYTES = 256; @@ -437,6 +438,7 @@ var ESTIMATED_CHARACTERS_PER_TOKEN = 4; var DEFAULT_SUMMARY_FILE_LIMIT = 30; var MAX_SELECTED_PATHS = 20; var MAX_INSPECT_TARGETS = 5; +var MAX_INSPECT_REQUEST_TARGETS = 20; var MAX_DISPLAYED_OCCURRENCES = 20; var MAX_STORED_MATCHES = 50000; var MAX_STORED_OCCURRENCES = 200000; @@ -458,6 +460,7 @@ var PRIVATE_KEY = /-----BEGIN ([^-\r\n]*PRIVATE KEY)-----[\s\S]*?-----END \1---- var SENSITIVE_NAME = String.raw`(?:(?:[A-Za-z][A-Za-z0-9]*[_-])*(?:password|passwd|secret|token|api[_-]?key|access[_-]?(?:key|token)|secret[_-]?access[_-]?key|private[_-]?key|service[_-]?key)(?:[_-][A-Za-z0-9]+)*)`; var SENSITIVE_ASSIGNMENT = new RegExp(String.raw`((? { const unquoted = rawValue.replace(/^["']|["']$/g, "").toLowerCase(); - if (TYPE_ONLY_VALUES.has(unquoted)) + if (TYPE_ONLY_VALUES.has(unquoted) || CODE_EXPRESSION.test(rawValue)) return match; count += 1; - return `${prefix}"[REDACTED]"`; + const quote = rawValue.startsWith('"') || rawValue.startsWith("'") ? rawValue[0] : ""; + return `${prefix}${quote}[REDACTED]${quote}`; }); redacted = redacted.replace(SENSITIVE_TOKEN, () => { count += 1; @@ -738,6 +742,15 @@ var PYTHON_LANGUAGE_CAPABILITIES = [ evidence: "syntax", availability: "implemented", load: "lazy" + }, + { + id: "python-lexical.roles", + name: "roles", + provider: "bounded Python lexical role scanner", + providerKind: "builtin", + evidence: "syntax", + availability: "implemented", + load: "lazy" } ]; var DEFAULT_LANGUAGE_CAPABILITIES = [ @@ -865,7 +878,17 @@ var MODE_FIELDS_BY_MODE = { auto: ordinaryFields, summary: ordinaryFields, matches: ordinaryFields, - inspect: [...commonFields, "path", "line", "cursor", "matchIndex", "matchIndices", "targets"], + inspect: [ + ...commonFields, + "path", + "paths", + "line", + "cursor", + "matchIndex", + "matchIndices", + "targets", + "scope" + ], outline: [...commonFields, "path", "line", "symbol", "cursor", "matchIndex", "maxFilesToParse"], imports: [ ...commonFields, @@ -885,7 +908,16 @@ var MODE_FIELDS_BY_MODE = { "matchIndex", "maxFilesToParse" ], - files: [...commonFields, "query", ...sourceFilters, "modifiedAfter", "modifiedBefore"], + files: [ + ...commonFields, + "query", + ...sourceFilters, + "ignorePolicy", + "limit", + "scope", + "modifiedAfter", + "modifiedBefore" + ], structure: [...commonFields, "pattern", ...sourceFilters, "maxFilesToParse"], concept: [...commonFields, "query", ...sourceFilters, "maxFilesToParse"], hybrid: [...commonFields, "query", ...sourceFilters, "conceptLimit", "maxFilesToParse"], @@ -894,9 +926,7 @@ var MODE_FIELDS_BY_MODE = { await: ["mode", "operationId"], cancel: ["mode", "operationId"] }; -var SAFE_DROP_FIELDS = { - files: ["scope"] -}; +var SAFE_DROP_FIELDS = {}; var SUPPORTED_OUTLINE_EXTENSIONS = new Set(DEFAULT_LANGUAGE_CAPABILITIES.flatMap((descriptor) => descriptor.capabilities.some((capability) => capability.name === "outline") ? descriptor.extensions : [])); function modeFields(mode) { return MODE_FIELDS_BY_MODE[mode]; @@ -925,8 +955,8 @@ var REQUEST_FIELD_GUIDANCE = { allOf: "allOf is case-sensitive exact-literal AND; omit pattern, anyOf, literal, ignoreCase, roles", context: "context is an output-context budget and is never silently dropped", limit: "limit is an output/page budget and is never silently dropped", - scope: "scope applies to ordinary content search; mode=files rejects this field because its scope is fixed strict, and only redundant strict may be removed", - ignorePolicy: "respect keeps repository ignore rules; include searches ignored files but always excludes .git internals and protected paths", + scope: "scope applies to ordinary content search; mode=files is always strict, so it accepts only the redundant scope=strict", + ignorePolicy: "respect keeps repository ignore rules and discloses ignored files; include also searches or lists ignored files but always excludes .git internals and protected paths", patterns: "audit accepts named exact-literal patterns and returns one closure receipt", maxFilesToParse: "concept/hybrid automatically batch the requested scope; this optional field sets an advanced hard file ceiling" }; @@ -1025,12 +1055,15 @@ function schemaError(input, field, reason) { function plainFilePatternRecovery(mode, input, invalid) { return mode === "files" && invalid.includes("pattern") && input.query === undefined && typeof input.pattern === "string" && input.pattern.trim().length > 0 && input.pattern.length <= 256 && input.pattern.isWellFormed() && !/[\\^$.*+?()[\]{}|\r\n\0]/u.test(input.pattern); } +function matchAllFilePatternRecovery(mode, input, invalid) { + return mode === "files" && invalid.includes("pattern") && input.query === undefined && typeof input.pattern === "string" && /^(?:\.\*|\.\+|\*+)$/u.test(input.pattern.trim()); +} function safeNextRequest(input, mode, invalid) { if (input.redact === true && containsSensitiveText(input)) return; const safe = new Set(SAFE_DROP_FIELDS[mode] ?? []); const plainFilePattern = plainFilePatternRecovery(mode, input, invalid); - if (plainFilePattern) + if (plainFilePattern || matchAllFilePatternRecovery(mode, input, invalid)) safe.add("pattern"); if (invalid.some((field) => !safe.has(field))) return; @@ -1076,7 +1109,7 @@ function fieldsError(input, mode, invalid, selectorIssues = []) { issues, recovery: nextRequest ? { action: "retry", - reason: plainFilePatternRecovery(mode, input, invalid) ? "For filename discovery, move the plain text from pattern to query and copy nextRequest; all filters are preserved." : `Remove only ${visibleFields} and copy the exact nextRequest; all other fields are preserved.`, + reason: plainFilePatternRecovery(mode, input, invalid) ? "For filename discovery, move the plain text from pattern to query and copy nextRequest; all filters are preserved." : matchAllFilePatternRecovery(mode, input, invalid) ? "A match-all pattern lists every file; omitting query does that, so copy nextRequest; all filters are preserved." : `Remove only ${visibleFields} and copy the exact nextRequest; all other fields are preserved.`, nextRequest } : { action: "manual", @@ -1232,7 +1265,7 @@ function validateRequestContract(input) { const allowed = new Set(MODE_FIELDS_BY_MODE[mode]); if (mode === "inspect" && raw.sourceCursor !== undefined) { allowed.clear(); - for (const field of ["mode", "sourceCursor", "redact"]) + for (const field of ["mode", "sourceCursor", "redact", "path"]) allowed.add(field); } if ((mode === "auto" || mode === "summary" || mode === "matches") && typeof raw.cursor === "string" && raw.cursor.includes(".analysis")) { @@ -2948,7 +2981,7 @@ function normalizeRequest(input) { throw new SiftLightError("ignorePolicy must be respect or include"); const pattern = input.pattern; if (pattern === undefined) { - throw new SiftLightError("pattern is required when cursor is not provided"); + throw new SiftLightError('pattern is required when cursor is not provided; to list files use mode="files" (query optional), to read a known file use mode="inspect" with path'); } const path = input.path?.replace(/^@/, ""); return { @@ -5872,7 +5905,7 @@ function publicAnalysisItem(result, item, index, storedId, modelOutput) { return { path: item.path, line: item.line, - label: result.kind === "outline" && modelOutput ? item.label : publicAnalysisLabel(result), + label: result.kind === "outline" && modelOutput || result.kind === "files" ? item.label : publicAnalysisLabel(result), index: index + 1, ...inspect ? { inspect } : {}, ...publicDetails ? { details: publicDetails } : {} @@ -6083,7 +6116,8 @@ Inspect: ${JSON.stringify(inspect)}` : ""}`; break; } } else { - for (let index = offset;index < result.items.length && items.length < 30; index += 1) { + const pageSize = result.pageSize ?? 30; + for (let index = offset;index < result.items.length && items.length < pageSize; index += 1) { if (!appendItem(index)) break; } @@ -6197,6 +6231,8 @@ function resolveInspectionTarget(input, cwd, snapshots) { } if (!path) throw new SiftLightError("path is required when mode=inspect"); + if (line === undefined && input.cursor === undefined) + line = 1; if (line === undefined || !Number.isSafeInteger(line) || line < 1) { throw new SiftLightError("line must be a positive integer when mode=inspect"); } @@ -7925,6 +7961,28 @@ async function filterPathsByModificationTime(cwd, paths, modifiedAfterMs, modifi // src/file-discovery.ts var graphemes = new Intl.Segmenter(undefined, { granularity: "grapheme" }); var FILE_QUERY_GLOB_WILDCARD = /[*?]/u; +var FILE_QUERY_MATCH_ALL = /^(?:[*/]+|\.\*|\.\+|\*\.\*)$/u; +var FILE_QUERY_REGEX_SYNTAX = /[()|^$\\+]/u; +var IGNORED_MATCH_SAMPLES = 5; +function normalizeFileQuery(query, hasGlob) { + const trimmed = query.trim(); + if (FILE_QUERY_MATCH_ALL.test(trimmed)) + return { + query: "", + note: `Query ${JSON.stringify(query)} was treated as listing every file under path; omit query for the same result.` + }; + if (!FILE_QUERY_GLOB_WILDCARD.test(trimmed)) + return { query }; + if (!hasGlob && !FILE_QUERY_REGEX_SYNTAX.test(trimmed) && !/\s/u.test(trimmed)) { + const glob = trimmed.includes("/") ? trimmed : `**/${trimmed}`; + return { + query: "", + glob, + note: `Query ${JSON.stringify(query)} contains glob wildcards and was applied as glob ${JSON.stringify(glob)}; mode=files query otherwise matches filename text.` + }; + } + throw new SiftLightError(`File query ${JSON.stringify(query)} uses wildcard or regex syntax; mode=files matches filename and path text, not patterns. Omit query to retain every file under path, use glob for one name pattern, or run one files request per name.`); +} function subsequenceScore(text, query) { const characters = Array.from(graphemes.segment(text), (item) => item.segment); const queryCharacters = Array.from(graphemes.segment(query), (item) => item.segment); @@ -7975,39 +8033,79 @@ function pathRelativeToDiscoveryRoot(cwd, root, path) { return scoped || platformBasename(absolutePath); } async function discoverFiles(input, cwd, signal) { - const query = input.query ?? ""; - if (query.length > 256 || !query.isWellFormed() || /[\r\n\0]/.test(query)) + const rawQuery = input.query ?? ""; + if (rawQuery.length > 256 || !rawQuery.isWellFormed() || /[\r\n\0]/.test(rawQuery)) throw new SiftLightError("File query must be well-formed single-line text of at most 256 characters"); - if (FILE_QUERY_GLOB_WILDCARD.test(query)) - throw new SiftLightError(`File query ${JSON.stringify(query)} uses glob wildcards; mode=files matches filename and path text, not glob patterns. Omit query to retain every file under path, or use glob to filter by name pattern.`); - const request = normalizeRequest({ ...input, pattern: "" }); + const requestedGlobs = input.glob === undefined ? [] : Array.isArray(input.glob) ? input.glob : [input.glob]; + const normalizedQuery = normalizeFileQuery(rawQuery, requestedGlobs.length > 0); + const query = normalizedQuery.query; + const request = normalizeRequest({ + ...input, + pattern: "", + ...normalizedQuery.glob ? { glob: [...requestedGlobs, normalizedQuery.glob] } : {} + }); const policy = new SearchPathPolicy(cwd); const discoveryRoot = request.path ?? "."; const scoringRoot = await policy.resolveSearchTarget(discoveryRoot); - const files = await listWorkspaceFiles(cwd, signal, { + const includeIgnored = request.ignorePolicy === "include"; + const listOptions = { path: scoringRoot, glob: request.glob, exclude: request.exclude, hidden: request.hidden + }; + const files = await listWorkspaceFiles(cwd, signal, { + ...listOptions, + ...includeIgnored ? { ignore: false, ignoreParents: false } : {} }); const filtered = await filterPathsByModificationTime(cwd, files.paths, request.modifiedAfterMs, request.modifiedBeforeMs, signal); + const rank = (path) => scoreFilePath(pathRelativeToDiscoveryRoot(cwd, scoringRoot, path), query); const selected = filtered.paths.flatMap((path) => { - const rank = scoreFilePath(pathRelativeToDiscoveryRoot(cwd, scoringRoot, path), query); - return rank ? [{ path, ...rank }] : []; + const score = rank(path); + return score ? [{ path, ...score }] : []; }).toSorted((left, right) => right.score - left.score || left.path.localeCompare(right.path)); for (let offset = 0;offset < selected.length; offset += 16) { await Promise.all(selected.slice(offset, offset + 16).map((item) => policy.assertExistingPath(item.path))); signal?.throwIfAborted(); } + const reasons = new Set([...files.reasons, ...filtered.reasons]); + if (normalizedQuery.note) + reasons.add(normalizedQuery.note); + let ignoredFiles = 0; + let ignoredMatches = 0; + let ignoredComparisonPartial = false; + if (!includeIgnored) { + const everything = await listWorkspaceFiles(cwd, signal, { + ...listOptions, + ignore: false, + ignoreParents: false + }); + ignoredComparisonPartial = everything.partial; + const admitted = new Set(files.paths); + const ignored = everything.paths.filter((path) => !admitted.has(path)); + ignoredFiles = ignored.length; + const ignoredCandidates = ignored.flatMap((path) => { + const score = rank(path); + return score ? [{ path, ...score }] : []; + }).toSorted((left, right) => right.score - left.score || left.path.localeCompare(right.path)); + ignoredMatches = ignoredCandidates.length; + if (ignoredFiles > 0) { + const samples = ignoredCandidates.slice(0, IGNORED_MATCH_SAMPLES).map((item) => `${JSON.stringify(item.path)} (${item.reason})`); + reasons.add(ignoredMatches > 0 ? `${String(ignoredFiles)} file(s) were excluded by ignore rules; ${String(ignoredMatches)} of them match this query${samples.length ? `, e.g. ${samples.join(", ")}` : ""}. Retry with ignorePolicy "include" to list them.` : `${String(ignoredFiles)} file(s) were excluded by ignore rules; none of them match this query.`); + } + if (ignoredComparisonPartial) + reasons.add("Ignored-file comparison was incomplete; the ignored-file count is a lower bound"); + } + const enumerationPartial = files.partial || filtered.partial; return { kind: "files", unit: "files", - partial: files.partial || filtered.partial, - reasons: [...new Set([...files.reasons, ...filtered.reasons])], + partial: enumerationPartial, + reasons: [...reasons], items: selected.map((item) => ({ path: item.path, line: 1, - label: `File candidate (${item.reason})`, + label: `File candidate (score ${String(item.score)}: ${item.reason})`, details: { kind: "file", score: item.score, @@ -8015,8 +8113,13 @@ async function discoverFiles(input, cwd, signal) { inspect: { mode: "inspect", path: item.path, line: 1 } } })), - coverage: { fileEnumeration: files.partial || filtered.partial ? "partial" : "complete" }, - stats: { filesEnumerated: files.paths.length }, + coverage: { + fileEnumeration: enumerationPartial ? "partial" : ignoredFiles > 0 || ignoredComparisonPartial ? "policy-filtered" : "complete" + }, + stats: { + filesEnumerated: files.paths.length, + ...includeIgnored ? {} : { ignoredFiles, ignoredMatches } + }, scope: { path: request.path ?? ".", requestedPath: request.path ?? ".", @@ -8029,7 +8132,8 @@ async function discoverFiles(input, cwd, signal) { ...request.modifiedAfterMs !== undefined ? { modifiedAfterMs: request.modifiedAfterMs } : {}, ...request.modifiedBeforeMs !== undefined ? { modifiedBeforeMs: request.modifiedBeforeMs } : {} }, - redact: input.redact ?? false + redact: input.redact ?? false, + ...input.limit !== undefined ? { pageSize: request.pageSize } : {} }; } @@ -8270,7 +8374,19 @@ async function prepare(target, access, structure) { let range = target.range; let details = { status: "no-symbol" }; const language = syntaxLanguage(document.path); - if (document.utf8 && language && language !== "go") { + if (target.range && target.structure) { + document.checkRange(target.range); + const lines = { + startLine: document.lineAt(target.range.start), + endLine: document.lineAt(Math.max(target.range.start, target.range.end - 1)) + }; + range = document.lineRange(lines.startLine, lines.endLine); + details = { + ...target.structure, + range: lines, + ...target.structure.symbol ? { symbol: { ...target.structure.symbol, range: lines } } : {} + }; + } else if (document.utf8 && language && language !== "go") { const syntax = await access.syntax(document); details = { status: syntax.status === "ok" ? "no-symbol" : syntax.status === "unsupported" ? "provider-unavailable" : "parse-error", @@ -8322,7 +8438,7 @@ async function prepare(target, access, structure) { details = { status: "provider-unavailable", ...language ? { language } : {} }; } range ??= document.lineRange(Math.max(1, target.line - 10), Math.min(document.lineStarts.length, target.line + 10)); - const boundary = target.range ? "requested-range" : details.status === "available" && details.range ? "syntax" : "line-window"; + const boundary = target.range ? target.structure ? "syntax" : "requested-range" : details.status === "available" && details.range ? "syntax" : "line-window"; document.checkRange(range); return { target, document, range, structure: details, boundary, focus }; } @@ -8681,8 +8797,10 @@ ${preview.text}`); } }; } -async function continueSource(cursor, access, continuations) { +async function continueSource(cursor, access, continuations, expectedPath) { const state = continuations.resolve(cursor); + if (expectedPath !== undefined && resolve18(access.cwd, expectedPath.replace(/^@/, "")) !== resolve18(access.cwd, state.source.path)) + throw new SiftLightError(`sourceCursor continues ${JSON.stringify(state.source.path)}, not ${JSON.stringify(expectedPath)}; copy the returned nextRequest exactly`); const document = await access.load(state.source.path, state.source); const page = sourcePage(document, state.remaining, MAX_RESULT_BYTES - 1400); const next = continuations.advance(cursor, page.fragment); @@ -9559,6 +9677,221 @@ function parsePythonOutline(document) { }); } +// src/python-roles.ts +var KEYWORDS = new Set([ + "and", + "as", + "assert", + "async", + "await", + "case", + "del", + "elif", + "else", + "except", + "for", + "from", + "global", + "if", + "import", + "in", + "is", + "lambda", + "match", + "nonlocal", + "not", + "or", + "raise", + "return", + "while", + "with", + "yield" +]); +var STRING_PREFIX = /^(?:[rRbBuUfF]|[rR][bBfF]|[bBfF][rR])?$/u; +var IDENTIFIER_START = /[\p{ID_Start}_]/u; +var IDENTIFIER_PART = /[\p{ID_Continue}]/u; +function lex(text) { + const comments = []; + const strings = []; + const code = new Uint8Array(text.length).fill(1); + const mark = (start, end) => { + code.fill(0, start, end); + }; + let index = 0; + while (index < text.length) { + const character = text[index]; + if (character === "#") { + const newline = text.indexOf(` +`, index); + const end = newline < 0 ? text.length : newline; + comments.push({ start: index, end }); + mark(index, end); + index = end; + continue; + } + if (character !== "'" && character !== '"') { + index += 1; + continue; + } + let prefixStart = index; + while (prefixStart > 0 && /[rRbBuUfF]/u.test(text[prefixStart - 1] ?? "")) + prefixStart -= 1; + const prefix = text.slice(prefixStart, index); + const beforePrefix = text[prefixStart - 1] ?? ""; + const start = STRING_PREFIX.test(prefix) && !IDENTIFIER_PART.test(beforePrefix) ? prefixStart : index; + const triple = text.slice(index, index + 3) === character.repeat(3); + const delimiter = triple ? character.repeat(3) : character; + let cursor = index + delimiter.length; + let end = text.length; + while (cursor < text.length) { + const current = text[cursor]; + if (current === "\\") { + cursor += 2; + continue; + } + if (!triple && current === ` +`) { + end = cursor; + break; + } + if (text.startsWith(delimiter, cursor)) { + end = cursor + delimiter.length; + break; + } + cursor += 1; + } + strings.push({ start, end }); + mark(start, end); + index = end; + } + return { comments, strings, code }; +} +function codeRanges(code) { + const ranges = []; + let start = -1; + for (let index = 0;index <= code.length; index += 1) { + const inCode = index < code.length && code[index] === 1; + if (inCode && start < 0) + start = index; + else if (!inCode && start >= 0) { + ranges.push({ start, end: index }); + start = -1; + } + } + return ranges; +} +function identifierEnd(text, start) { + let end = start; + while (end < text.length && IDENTIFIER_PART.test(text[end] ?? "")) + end += 1; + return end; +} +function skipSpaces(text, index, code) { + let cursor = index; + while (cursor < text.length && code[cursor] === 1 && /[ \t]/u.test(text[cursor] ?? "")) + cursor += 1; + return cursor; +} +function statementEnd(text, start, code) { + let depth = 0; + for (let index = start;index < text.length; index += 1) { + if (code[index] !== 1) + continue; + const character = text[index]; + if (character === "(" || character === "[" || character === "{") + depth += 1; + else if (character === ")" || character === "]" || character === "}") + depth = Math.max(0, depth - 1); + else if (character === "\\" && text[index + 1] === ` +`) + index += 1; + else if (character === ` +` && depth === 0) + return index; + } + return text.length; +} +function pythonRoleAnalysis(document) { + const text = document.text; + const { comments, strings, code } = lex(text); + const roles = []; + const push = (start, end, role, certainty, subkind) => { + if (start < end) + roles.push({ start, end, role, certainty, node: 0, ...subkind ? { subkind } : {} }); + }; + for (const span of comments) + push(span.start, span.end, "comment", "syntax"); + for (const span of strings) + push(span.start, span.end, "string", "syntax"); + for (const span of codeRanges(code)) + push(span.start, span.end, "code", "syntax"); + let index = 0; + let lineStart = true; + while (index < text.length) { + const character = text[index] ?? ""; + if (character === ` +`) { + lineStart = true; + index += 1; + continue; + } + if (code[index] !== 1 || !IDENTIFIER_START.test(character)) { + if (!/[ \t]/u.test(character)) + lineStart = false; + index += 1; + continue; + } + const previous = text[index - 1] ?? ""; + if (IDENTIFIER_PART.test(previous)) { + index += 1; + continue; + } + const end = identifierEnd(text, index); + const word = text.slice(index, end); + const atStatementStart = lineStart; + lineStart = false; + if (atStatementStart && (word === "import" || word === "from")) { + const statement = statementEnd(text, index, code); + const body = text.slice(index, statement); + if (word === "import" || /\bimport\b/u.test(body)) { + push(index, statement, "import", "syntax", word === "from" ? "from-import" : "import"); + index = statement; + continue; + } + } + if (word === "def" || word === "class") { + const nameStart = skipSpaces(text, end, code); + if (IDENTIFIER_START.test(text[nameStart] ?? "")) { + const nameEnd = identifierEnd(text, nameStart); + push(nameStart, nameEnd, "declaration", "syntax", word === "def" ? "function" : "class"); + index = nameEnd; + continue; + } + } + let chainEnd = end; + while (text[chainEnd] === "." && IDENTIFIER_START.test(text[chainEnd + 1] ?? "")) { + chainEnd = identifierEnd(text, chainEnd + 1); + } + const open = skipSpaces(text, chainEnd, code); + const lastSegment = text.slice(text.lastIndexOf(".", chainEnd - 1) + 1, chainEnd); + const before = text.slice(Math.max(0, index - 4), index); + if (text[open] === "(" && code[open] === 1 && !KEYWORDS.has(word) && !KEYWORDS.has(lastSegment) && !/\bdef\s+$|\bclass\s+$/u.test(before)) { + push(index, open + 1, "call", "candidate", "lexical-call"); + } + index = chainEnd; + } + roles.sort((a, b) => a.start - b.start || a.end - b.end || a.role.localeCompare(b.role)); + return { + status: "ok", + nodes: [], + children: [], + symbols: [], + roles, + diagnostics: [], + limited: false + }; +} + // src/hybrid-search.ts import { resolve as resolve20 } from "node:path"; @@ -10506,6 +10839,69 @@ import { realpath as realpath6 } from "node:fs/promises"; function isEvidenceRequest(input) { return input.mode === "concept" || input.mode === "hybrid" || input.mode === "structure" || input.mode === "files" || input.mode === "inspect" || input.mode === "outline" || input.mode === "imports" || input.mode === "tests" || input.mode === "validate" || input.sourceCursor !== undefined || input.anyOf !== undefined || input.allOf !== undefined || input.within !== undefined || input.roles !== undefined || input.changes !== undefined || input.symbol !== undefined || input.conceptLimit !== undefined || (input.cursor?.includes(".analysis") ?? false); } +function pageInspectRequest(input) { + if (input.scope === "expand") + throw new SiftLightError("mode=inspect reads exact locations; scope=expand does not apply (omit scope)"); + const { scope: _scope, paths, ...rest } = input; + let request = rest; + if (paths !== undefined) { + if (input.cursor !== undefined || input.targets !== undefined || input.matchIndices !== undefined || input.path !== undefined || input.line !== undefined || input.matchIndex !== undefined) + throw new SiftLightError("mode=inspect paths opens files from line 1 and cannot be combined with cursor, targets, matchIndices, path, line or matchIndex"); + request = { ...rest, targets: paths.map((path) => ({ path, line: 1 })) }; + } + const redact = input.redact ? { redact: true } : {}; + const requested = request.targets?.length ?? request.matchIndices?.length ?? 0; + if (requested > MAX_INSPECT_REQUEST_TARGETS) + throw new SiftLightError(`mode=inspect accepts at most ${String(MAX_INSPECT_REQUEST_TARGETS)} targets per request; ${String(MAX_INSPECT_TARGETS)} are inspected per response`); + if (request.targets && request.targets.length > MAX_INSPECT_TARGETS) + return { + current: { ...request, targets: request.targets.slice(0, MAX_INSPECT_TARGETS) }, + remaining: { + mode: "inspect", + targets: request.targets.slice(MAX_INSPECT_TARGETS), + ...redact + } + }; + if (request.matchIndices && request.matchIndices.length > MAX_INSPECT_TARGETS) + return { + current: { ...request, matchIndices: request.matchIndices.slice(0, MAX_INSPECT_TARGETS) }, + remaining: { + mode: "inspect", + ...request.cursor !== undefined ? { cursor: request.cursor } : {}, + matchIndices: request.matchIndices.slice(MAX_INSPECT_TARGETS), + ...redact + } + }; + return { current: request }; +} +function withRemainingInspection(result, remaining) { + const count = remaining.targets?.length ?? remaining.matchIndices?.length ?? 0; + return { + text: `${result.text} + +[${String(count)} more target(s) were not inspected; one response inspects ${String(MAX_INSPECT_TARGETS)}.] +Next request: ${JSON.stringify(remaining)}`, + details: { + ...result.details, + status: "partial", + snapshotComplete: false, + nextRequest: remaining + } + }; +} +function analysisItemStructure(item) { + const details = item.details ?? {}; + const language = typeof details.language === "string" ? details.language : undefined; + const name = typeof details.name === "string" ? details.name : undefined; + const scope = Array.isArray(details.scope) ? details.scope.filter((entry) => typeof entry === "string") : typeof details.scope === "string" ? [details.scope] : []; + const kind = details.kind === "function" ? "function" : item.label.split(" ", 1)[0] ?? "symbol"; + return { + status: "available", + provider: language === "python" ? "python-outline" : "tree-sitter", + ...language ? { language } : {}, + ...name ? { symbol: { name, kind, scope, range: { startLine: item.line, endLine: item.line } } } : {} + }; +} function maxFilesToParse(value, defaultValue = MAX_STRUCTURE_FILES) { const candidate = value ?? defaultValue; if (!Number.isSafeInteger(candidate) || candidate < 1 || candidate > MAX_CONFIGURABLE_STRUCTURE_FILES) { @@ -10521,8 +10917,8 @@ function validateTerms(input) { } return; } - if (!Array.isArray(terms) || terms.length < 2 || terms.length > 3 || terms.some((term) => typeof term !== "string" || !term.trim() || /[\r\n\0]/.test(term)) || new Set(terms).size !== terms.length) - throw new SiftLightError("allOf requires 2–3 distinct, nonempty, single-line literal terms"); + if (!Array.isArray(terms) || terms.length < 1 || terms.length > 3 || terms.some((term) => typeof term !== "string" || !term.trim() || /[\r\n\0]/.test(term)) || new Set(terms).size !== terms.length) + throw new SiftLightError("allOf requires 1–3 distinct, nonempty, single-line literal terms"); if (input.pattern !== undefined || input.roles !== undefined || input.literal !== undefined || input.ignoreCase !== undefined || input.wholeWord !== undefined) throw new SiftLightError("allOf is an explicit case-sensitive literal conjunction; omit pattern, roles, literal and ignoreCase"); if (input.within !== undefined && input.within !== "file" && input.within !== "function") @@ -10720,11 +11116,13 @@ class EvidenceService { throw new CursorError("A nonempty sourceCursor is required"); if (input.mode !== "inspect") throw new SiftLightError("sourceCursor requires mode=inspect"); - return continueSource(input.sourceCursor, access, this.#continuations); + return continueSource(input.sourceCursor, access, this.#continuations, input.path); } if (input.mode === "inspect") { - const targets = this.#inspectionTargets(input, cwd); - return targets.some((target) => target.range !== undefined) ? inspectDocumentsMetadata(targets, access, this.#structure) : inspectDocuments(targets, access, this.#continuations, this.#structure); + const { current, remaining } = pageInspectRequest(input); + const targets = this.#inspectionTargets(current, cwd); + const result = await inspectDocuments(targets, access, this.#continuations, this.#structure); + return remaining ? withRemainingInspection(result, remaining) : result; } if (input.mode === "concept") { const execution = await this.#conceptSearch(input, access, options.onProgress); @@ -10813,7 +11211,7 @@ class EvidenceService { if (input.pattern !== undefined || input.allOf !== undefined || input.within !== undefined || input.roles !== undefined || input.literal !== undefined || input.ignoreCase !== undefined || input.wholeWord !== undefined) throw new SiftLightError("anyOf is an explicit case-sensitive literal union; omit pattern, allOf, within, roles, literal and ignoreCase"); if (input.mode !== undefined && input.mode !== "auto" && input.mode !== "matches") - throw new SiftLightError("anyOf mode must be omitted, auto, or matches"); + throw new SiftLightError("anyOf mode must be omitted, auto, matches or summary"); const chunks = Array.from({ length: Math.ceil(anyOf.length / MAX_ANY_OF_TERMS) }, (_, index) => anyOf.slice(index * MAX_ANY_OF_TERMS, (index + 1) * MAX_ANY_OF_TERMS)); const { path: _inputPath, ...unscopedInput } = input; let chunkAccess = access; @@ -10951,9 +11349,10 @@ class EvidenceService { if (item) result.items.push(item); } else if (terms || input.roles) { - if (syntaxLanguage(file.document.path)) + const pythonRoles = !terms && /\.py$/iu.test(file.document.path); + if (pythonRoles || syntaxLanguage(file.document.path)) syntaxCapableFiles += 1; - const syntax = await access.syntax(file.document); + const syntax = pythonRoles ? pythonRoleAnalysis(file.document) : await access.syntax(file.document); const classified = terms ? findFunctionConjunctions(file.document, syntax, terms, input.changes?.scope === "lines" ? file.changedRanges : undefined) : filterRoleOccurrences(file.document, syntax, file.occurrences, input.roles ?? []); result.items.push(...classified.items); result.partial ||= classified.partial; @@ -11030,12 +11429,12 @@ class EvidenceService { const item = this.#analyses.item(input.cursor, input.matchIndex); if (!item.source || !item.range) throw new CursorError("This analysis item has no verified source range"); - const metadataOnly = item.details?.kind === "symbol" || item.details?.kind === "function"; + const bounded = item.details?.kind === "symbol" || item.details?.kind === "function"; return { path: item.path, line: item.line, reference: item.source, - ...metadataOnly ? { range: item.range } : { absoluteFocus: item.range.start } + ...bounded ? { range: item.range, structure: analysisItemStructure(item) } : { absoluteFocus: item.range.start } }; } return { @@ -11244,6 +11643,29 @@ import { resolve as resolve24 } from "node:path"; // src/discovery-errors.ts var DISCOVERY_MODE_REQUIRED_ERROR = 'query requires an explicit discovery mode: use mode=files for filename/path discovery or mode=concept for semantic discovery; for example {"mode":"files","query":""}'; +// src/request-aliases.ts +var MODE_ALIASES = ["anyOf", "allOf"]; +function normalizeRequestAliases(request) { + const notes = []; + const { mode, ...rest } = request; + let input = rest; + if (mode === "anyOf" || mode === "allOf") { + if (request[mode] === undefined) + throw new SiftLightError(`mode=${mode} is not a mode; pass ${mode}:[...terms] and omit mode`); + notes.push(`mode="${mode}" is not a mode; the ${mode} field alone selects this search.`); + } else if (mode === "summary" && request.anyOf !== undefined) { + notes.push("anyOf returns one analysis page whose header already carries per-file statistics; mode=summary was served by that page."); + } else if (mode !== undefined) { + input = { ...rest, mode }; + } + if (input.sourceCursor !== undefined && input.line !== undefined) { + const { line: _line, ...withoutLine } = input; + input = withoutLine; + notes.push("line is ignored with sourceCursor; the continuation selects its own source range."); + } + return { input, notes }; +} + // src/format.ts import { readFile as readFile3 } from "node:fs/promises"; var RESULT_METADATA_RESERVE_BYTES = 1024; @@ -12597,6 +13019,16 @@ class LanguageCapabilityCatalog { } // src/service.ts +function withRequestNotes(result, notes) { + if (notes.length === 0) + return result; + return { + text: `${notes.map((note) => `[Request note: ${note}]`).join(` +`)} +${result.text}`, + details: { ...result.details, requestNotes: [...notes] } + }; +} function filterList(value) { return value === undefined ? [] : Array.isArray(value) ? [...value] : [value]; } @@ -12825,7 +13257,11 @@ class SiftLightService { this.#operations = new OperationLifecycle({ deadlineMs: resolveConceptTimeoutMs() }); this.#evidence = new EvidenceService(this.#runRipgrep, this.#snapshots, options.structure, options.conceptSearch, options.semanticJudge); } - async search(input, cwd, signal, options = {}) { + async search(request, cwd, signal, options = {}) { + const { input, notes } = normalizeRequestAliases(request); + return withRequestNotes(await this.#searchNormalized(input, cwd, signal, options), notes); + } + async#searchNormalized(input, cwd, signal, options) { validateRawSearchInput(input); validateRequestContract(input); if (!this.#vectorSearchEnabled && (input.mode === "concept" || input.mode === "hybrid")) { @@ -13297,9 +13733,9 @@ var siftLightSchema = Type.Object({ description: `${fieldGuidance("anyOf")}. ${String(MIN_ANY_OF_TERMS)}-${String(MAX_ANY_OF_TOTAL_TERMS)} distinct case-sensitive single-line terms, at most ${String(MAX_LITERAL_TERM_BYTES)} UTF-8 bytes each. Requests above ${String(MAX_ANY_OF_TERMS)} terms are split into version-checked chunks and merged. Returns every retained occurrence attributed to its term.` })), allOf: Type.Optional(Type.Array(Type.String({ maxLength: MAX_PATH_CHARACTERS }), { - minItems: 2, + minItems: 1, maxItems: 3, - description: `${fieldGuidance("allOf")}. 2-3 distinct terms must occur in one file (default) or one function.` + description: `${fieldGuidance("allOf")}. 1-3 distinct terms must occur in one file (default) or one function; one term lists the files or functions containing it.` })), within: Type.Optional(stringEnum(["file", "function"], { description: "Only valid with allOf; omit for ordinary single-pattern searches. function requires JS/TS/TSX and counts only that implementation's own code, excluding nested callbacks, strings/comments/types. Not proof of a shared execution path." @@ -13316,7 +13752,7 @@ var siftLightSchema = Type.Object({ "unknown" ]), { minItems: 1, - description: "Filter each single-pattern occurrence by syntax role (JS/TS/TSX/Go). Roles may be candidates, especially Go call/conversion ambiguity. Cannot combine with allOf." + description: "Filter each single-pattern occurrence by syntax role (JS/TS/TSX/Go parsed; Python lexical: comment, string, code, declaration, import, call candidates). Roles may be candidates, especially Go call/conversion and Python calls. Cannot combine with allOf." })), changes: Type.Optional(Type.Object({ base: Type.Optional(Type.String({ @@ -13349,7 +13785,7 @@ var siftLightSchema = Type.Object({ paths: Type.Optional(Type.Array(Type.String(), { minItems: 1, maxItems: MAX_SELECTED_PATHS, - description: "Exact retained files to select together from a cursor. A new search accepts one path; split multiple roots into separate requests." + description: "Exact retained files to select together from a cursor. A new search accepts one path; split multiple roots into separate requests. With mode=inspect and no cursor, opens each file from line 1 (same as targets with line 1)." })), glob: Type.Optional(Type.Union([ Type.String({ maxLength: MAX_PATH_CHARACTERS }), @@ -13373,7 +13809,7 @@ var siftLightSchema = Type.Object({ })), hidden: Type.Optional(Type.Boolean({ description: "Search hidden files (default true; .git is always excluded)." })), ignorePolicy: Type.Optional(stringEnum(["respect", "include"], { - description: "respect (default) honors ignore rules and reports policy-filtered coverage when files are omitted. include searches ignored files while still excluding .git internals and protected paths." + description: "respect (default) honors ignore rules and reports policy-filtered coverage when files are omitted, including mode=files. include searches or lists ignored files while still excluding .git internals and protected paths." })), patterns: Type.Optional(Type.Array(Type.Object({ id: Type.String({ minLength: 1, maxLength: 64 }), @@ -13418,29 +13854,29 @@ var siftLightSchema = Type.Object({ limit: Type.Optional(Type.Integer({ minimum: 1, maximum: MAX_PAGE_SIZE, - description: `${fieldGuidance("limit")}. Ordinary search only: explicit detail-page match limit (max 100). Normally omit to preserve automatic summarization; analysis and inspect modes reject it.` + description: `${fieldGuidance("limit")}. Ordinary search: explicit detail-page match limit (max 100); mode=files: files per page (default 30). Normally omit to preserve automatic summarization; other analysis and inspect modes reject it.` })), - mode: Type.Optional(stringEnum(SIFT_LIGHT_MODES, { - description: `Ordinary search defaults to auto; summary/matches request explicit pages. capabilities returns a compact names-only project inventory and per-language supported modes without loading providers. files uses query, structure uses an AST pattern, concept uses a required natural-language query, and hybrid uses one query for exact literal plus concept evidence in a single snapshot. validate rechecks saved search or analysis sources against their recorded origin. await waits for an existing long-running concept or hybrid operation without restarting it; cancel explicitly cancels one and waits for owned cleanup. Copy the returned nextRequest exactly and do not repeat the original query. Waiting is an operation state, not evidence. Validation details report the requested scope, comparison target, coverage, and freshness as current, stale, or unknown; partial coverage is retained during validation. inspect/outline/imports/tests retain their documented location selectors. Syntax results are static evidence; concept and related-test results remain candidates. ${MODE_CONTRACT_DESCRIPTION}` + mode: Type.Optional(stringEnum([...SIFT_LIGHT_MODES, ...MODE_ALIASES], { + description: `Ordinary search defaults to auto; summary/matches request explicit pages. capabilities returns a compact names-only project inventory and per-language supported modes without loading providers. files uses query, structure uses an AST pattern, concept uses a required natural-language query, and hybrid uses one query for exact literal plus concept evidence in a single snapshot. validate rechecks saved search or analysis sources against their recorded origin. await waits for an existing long-running concept or hybrid operation without restarting it; cancel explicitly cancels one and waits for owned cleanup. Copy the returned nextRequest exactly and do not repeat the original query. Waiting is an operation state, not evidence. Validation details report the requested scope, comparison target, coverage, and freshness as current, stale, or unknown; partial coverage is retained during validation. inspect/outline/imports/tests retain their documented location selectors. Syntax results are static evidence; concept and related-test results remain candidates. ${MODE_CONTRACT_DESCRIPTION} anyOf/allOf are field names, not modes; as mode values they are accepted aliases for omitting mode, disclosed in a request note.` })), line: Type.Optional(Type.Number({ - description: "1-indexed source line for path inspection/navigation. Omit with matchIndex, matchIndices or targets." + description: "1-indexed source line for path inspection/navigation; path-only inspect starts at line 1. Omit with matchIndex, matchIndices or targets." })), matchIndex: Type.Optional(Type.Number({ description: "1-based retained match index for cursor-scoped inspect; replaces path and line." })), matchIndices: Type.Optional(Type.Array(Type.Integer({ minimum: 1 }), { minItems: 1, - maxItems: MAX_INSPECT_TARGETS, - description: "Inspect up to five visible match numbers together using the same cursor; mutually exclusive with matchIndex, path, line and targets." + maxItems: MAX_INSPECT_REQUEST_TARGETS, + description: `Inspect visible match numbers together using the same cursor; mutually exclusive with matchIndex, path, line and targets. One response inspects ${String(MAX_INSPECT_TARGETS)}; the rest are returned as an exact nextRequest.` })), targets: Type.Optional(Type.Array(Type.Object({ path: Type.String({ maxLength: MAX_PATH_CHARACTERS }), line: Type.Integer({ minimum: 1 }) }), { minItems: 1, - maxItems: MAX_INSPECT_TARGETS, - description: "Inspect known path/line locations together without a cursor. The complete batch shares one 16 KiB response budget." + maxItems: MAX_INSPECT_REQUEST_TARGETS, + description: `Inspect known path/line locations together without a cursor. One response inspects ${String(MAX_INSPECT_TARGETS)} within a shared 16 KiB budget; the rest are returned as an exact nextRequest.` })), cursor: Type.Optional(Type.String({ description: "Opaque cursor from a previous stable search snapshot." })), operationId: Type.Optional(Type.String({ diff --git a/src/python-roles.ts b/src/python-roles.ts new file mode 100644 index 00000000..3d92e715 --- /dev/null +++ b/src/python-roles.ts @@ -0,0 +1,247 @@ +import type { SourceDocument } from "./source-document.js"; +import type { SyntaxAnalysis, SyntaxRole } from "./syntax-types.js"; + +/** + * Bounded lexical role scanner for Python. + * + * It separates comments, string literals and code exactly, and marks `def`/`class` + * names, import statements and call-shaped callees. It does not parse Python: + * comments, strings and code are lexical facts ("syntax" certainty); a call is a + * name followed by `(` in code and therefore stays a "candidate". Offsets are + * UTF-16 indices into `document.text`, like every other role provider. + */ + +const KEYWORDS = new Set([ + "and", + "as", + "assert", + "async", + "await", + "case", + "del", + "elif", + "else", + "except", + "for", + "from", + "global", + "if", + "import", + "in", + "is", + "lambda", + "match", + "nonlocal", + "not", + "or", + "raise", + "return", + "while", + "with", + "yield", +]); +const STRING_PREFIX = /^(?:[rRbBuUfF]|[rR][bBfF]|[bBfF][rR])?$/u; +const IDENTIFIER_START = /[\p{ID_Start}_]/u; +const IDENTIFIER_PART = /[\p{ID_Continue}]/u; + +interface Span { + start: number; + end: number; +} + +interface Lexed { + comments: Span[]; + strings: Span[]; + /** `true` at each UTF-16 index that belongs to code (outside comments/strings). */ + code: Uint8Array; +} + +function lex(text: string): Lexed { + const comments: Span[] = []; + const strings: Span[] = []; + const code = new Uint8Array(text.length).fill(1); + const mark = (start: number, end: number): void => { + code.fill(0, start, end); + }; + let index = 0; + while (index < text.length) { + const character = text[index]; + if (character === "#") { + const newline = text.indexOf("\n", index); + const end = newline < 0 ? text.length : newline; + comments.push({ start: index, end }); + mark(index, end); + index = end; + continue; + } + if (character !== "'" && character !== '"') { + index += 1; + continue; + } + // A string prefix (r, b, f, u and their pairs) is part of the literal. + let prefixStart = index; + while (prefixStart > 0 && /[rRbBuUfF]/u.test(text[prefixStart - 1] ?? "")) prefixStart -= 1; + const prefix = text.slice(prefixStart, index); + const beforePrefix = text[prefixStart - 1] ?? ""; + const start = + STRING_PREFIX.test(prefix) && !IDENTIFIER_PART.test(beforePrefix) ? prefixStart : index; + const triple = text.slice(index, index + 3) === character.repeat(3); + const delimiter = triple ? character.repeat(3) : character; + let cursor = index + delimiter.length; + let end = text.length; + while (cursor < text.length) { + const current = text[cursor]; + // A backslash always protects the next character from ending the literal, + // including in raw strings (where the backslash itself is kept). + if (current === "\\") { + cursor += 2; + continue; + } + if (!triple && current === "\n") { + end = cursor; + break; + } + if (text.startsWith(delimiter, cursor)) { + end = cursor + delimiter.length; + break; + } + cursor += 1; + } + strings.push({ start, end }); + mark(start, end); + index = end; + } + return { comments, strings, code }; +} + +function codeRanges(code: Uint8Array): Span[] { + const ranges: Span[] = []; + let start = -1; + for (let index = 0; index <= code.length; index += 1) { + const inCode = index < code.length && code[index] === 1; + if (inCode && start < 0) start = index; + else if (!inCode && start >= 0) { + ranges.push({ start, end: index }); + start = -1; + } + } + return ranges; +} + +function identifierEnd(text: string, start: number): number { + let end = start; + while (end < text.length && IDENTIFIER_PART.test(text[end] ?? "")) end += 1; + return end; +} + +function skipSpaces(text: string, index: number, code: Uint8Array): number { + let cursor = index; + while (cursor < text.length && code[cursor] === 1 && /[ \t]/u.test(text[cursor] ?? "")) + cursor += 1; + return cursor; +} + +/** Index after the statement starting at `start`, following parenthesized continuation lines. */ +function statementEnd(text: string, start: number, code: Uint8Array): number { + let depth = 0; + for (let index = start; index < text.length; index += 1) { + if (code[index] !== 1) continue; + const character = text[index]; + if (character === "(" || character === "[" || character === "{") depth += 1; + else if (character === ")" || character === "]" || character === "}") + depth = Math.max(0, depth - 1); + else if (character === "\\" && text[index + 1] === "\n") index += 1; + else if (character === "\n" && depth === 0) return index; + } + return text.length; +} + +export function pythonRoleAnalysis(document: SourceDocument): SyntaxAnalysis { + const text = document.text; + const { comments, strings, code } = lex(text); + const roles: SyntaxRole[] = []; + const push = ( + start: number, + end: number, + role: SyntaxRole["role"], + certainty: SyntaxRole["certainty"], + subkind?: string, + ): void => { + if (start < end) + roles.push({ start, end, role, certainty, node: 0, ...(subkind ? { subkind } : {}) }); + }; + for (const span of comments) push(span.start, span.end, "comment", "syntax"); + for (const span of strings) push(span.start, span.end, "string", "syntax"); + for (const span of codeRanges(code)) push(span.start, span.end, "code", "syntax"); + + let index = 0; + let lineStart = true; + while (index < text.length) { + const character = text[index] ?? ""; + if (character === "\n") { + lineStart = true; + index += 1; + continue; + } + if (code[index] !== 1 || !IDENTIFIER_START.test(character)) { + if (!/[ \t]/u.test(character)) lineStart = false; + index += 1; + continue; + } + const previous = text[index - 1] ?? ""; + if (IDENTIFIER_PART.test(previous)) { + index += 1; + continue; + } + const end = identifierEnd(text, index); + const word = text.slice(index, end); + const atStatementStart = lineStart; + lineStart = false; + if (atStatementStart && (word === "import" || word === "from")) { + const statement = statementEnd(text, index, code); + const body = text.slice(index, statement); + if (word === "import" || /\bimport\b/u.test(body)) { + push(index, statement, "import", "syntax", word === "from" ? "from-import" : "import"); + index = statement; + continue; + } + } + if (word === "def" || word === "class") { + const nameStart = skipSpaces(text, end, code); + if (IDENTIFIER_START.test(text[nameStart] ?? "")) { + const nameEnd = identifierEnd(text, nameStart); + push(nameStart, nameEnd, "declaration", "syntax", word === "def" ? "function" : "class"); + index = nameEnd; + continue; + } + } + // Attribute chains such as `client.fetch(` are one callee. + let chainEnd = end; + while (text[chainEnd] === "." && IDENTIFIER_START.test(text[chainEnd + 1] ?? "")) { + chainEnd = identifierEnd(text, chainEnd + 1); + } + const open = skipSpaces(text, chainEnd, code); + const lastSegment = text.slice(text.lastIndexOf(".", chainEnd - 1) + 1, chainEnd); + const before = text.slice(Math.max(0, index - 4), index); + if ( + text[open] === "(" && + code[open] === 1 && + !KEYWORDS.has(word) && + !KEYWORDS.has(lastSegment) && + !/\bdef\s+$|\bclass\s+$/u.test(before) + ) { + push(index, open + 1, "call", "candidate", "lexical-call"); + } + index = chainEnd; + } + roles.sort((a, b) => a.start - b.start || a.end - b.end || a.role.localeCompare(b.role)); + return { + status: "ok", + nodes: [], + children: [], + symbols: [], + roles, + diagnostics: [], + limited: false, + }; +} diff --git a/src/redaction.ts b/src/redaction.ts index 9563a1b9..7aa3648e 100644 --- a/src/redaction.ts +++ b/src/redaction.ts @@ -7,6 +7,10 @@ const SENSITIVE_ASSIGNMENT = new RegExp( "gi", ); const SENSITIVE_TOKEN = /\b(?:sk|ghp|xox[baprs])[-_][A-Za-z0-9][A-Za-z0-9_-]{7,}\b/g; +// An unquoted value that starts as a call or index expression (`env.get(`, +// `os.environ[`) is source code that reads a secret, not the secret itself. +// Masking it hides the evidence a reader needs, such as which key is read. +const CODE_EXPRESSION = /^[A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)*\s*[([]/u; const TYPE_ONLY_VALUES = new Set([ "boolean", "number", @@ -27,9 +31,12 @@ function redactString(value: string): { value: string; count: number } { SENSITIVE_ASSIGNMENT, (match: string, prefix: string, rawValue: string) => { const unquoted = rawValue.replace(/^["']|["']$/g, "").toLowerCase(); - if (TYPE_ONLY_VALUES.has(unquoted)) return match; + if (TYPE_ONLY_VALUES.has(unquoted) || CODE_EXPRESSION.test(rawValue)) return match; count += 1; - return `${prefix}"[REDACTED]"`; + // Keep the source's own quoting: adding quotes to an unquoted value would + // misrepresent the file (for example, a bare `.env` value shown as quoted). + const quote = rawValue.startsWith('"') || rawValue.startsWith("'") ? rawValue[0] : ""; + return `${prefix}${quote}[REDACTED]${quote}`; }, ); redacted = redacted.replace(SENSITIVE_TOKEN, () => { diff --git a/src/request-aliases.ts b/src/request-aliases.ts new file mode 100644 index 00000000..19984d79 --- /dev/null +++ b/src/request-aliases.ts @@ -0,0 +1,48 @@ +import { SiftLightError } from "./errors.js"; +import type { SiftLightInput } from "./service.js"; + +/** + * Field names that callers commonly send as a mode. They are not modes: the field + * alone selects the multi-term search. The schema accepts them so the request can + * be served, and the result discloses the normalization. + */ +export const MODE_ALIASES = ["anyOf", "allOf"] as const; +export type ModeAlias = (typeof MODE_ALIASES)[number]; + +/** Public request shape before alias normalization. */ +export type SiftLightRequest = Omit & { + mode?: SiftLightInput["mode"] | ModeAlias; +}; + +export interface NormalizedRequest { + input: SiftLightInput; + notes: string[]; +} + +/** + * The single place where unambiguous request shapes are rewritten before the + * mode contract runs. Every rewrite preserves the requested scope and evidence + * semantics and is reported back as a note; anything ambiguous still fails. + */ +export function normalizeRequestAliases(request: SiftLightRequest): NormalizedRequest { + const notes: string[] = []; + const { mode, ...rest } = request; + let input: SiftLightInput = rest; + if (mode === "anyOf" || mode === "allOf") { + if (request[mode] === undefined) + throw new SiftLightError(`mode=${mode} is not a mode; pass ${mode}:[...terms] and omit mode`); + notes.push(`mode="${mode}" is not a mode; the ${mode} field alone selects this search.`); + } else if (mode === "summary" && request.anyOf !== undefined) { + notes.push( + "anyOf returns one analysis page whose header already carries per-file statistics; mode=summary was served by that page.", + ); + } else if (mode !== undefined) { + input = { ...rest, mode }; + } + if (input.sourceCursor !== undefined && input.line !== undefined) { + const { line: _line, ...withoutLine } = input; + input = withoutLine; + notes.push("line is ignored with sourceCursor; the continuation selects its own source range."); + } + return { input, notes }; +} diff --git a/src/request-contract-catalog.ts b/src/request-contract-catalog.ts index 4567d7db..2e33a689 100644 --- a/src/request-contract-catalog.ts +++ b/src/request-contract-catalog.ts @@ -62,7 +62,17 @@ export const MODE_FIELDS_BY_MODE: Record auto: ordinaryFields, summary: ordinaryFields, matches: ordinaryFields, - inspect: [...commonFields, "path", "line", "cursor", "matchIndex", "matchIndices", "targets"], + inspect: [ + ...commonFields, + "path", + "paths", + "line", + "cursor", + "matchIndex", + "matchIndices", + "targets", + "scope", + ], outline: [...commonFields, "path", "line", "symbol", "cursor", "matchIndex", "maxFilesToParse"], imports: [ ...commonFields, @@ -82,7 +92,16 @@ export const MODE_FIELDS_BY_MODE: Record "matchIndex", "maxFilesToParse", ], - files: [...commonFields, "query", ...sourceFilters, "modifiedAfter", "modifiedBefore"], + files: [ + ...commonFields, + "query", + ...sourceFilters, + "ignorePolicy", + "limit", + "scope", + "modifiedAfter", + "modifiedBefore", + ], structure: [...commonFields, "pattern", ...sourceFilters, "maxFilesToParse"], concept: [...commonFields, "query", ...sourceFilters, "maxFilesToParse"], hybrid: [...commonFields, "query", ...sourceFilters, "conceptLimit", "maxFilesToParse"], @@ -92,9 +111,7 @@ export const MODE_FIELDS_BY_MODE: Record cancel: ["mode", "operationId"], }; -export const SAFE_DROP_FIELDS: Partial> = { - files: ["scope"], -}; +export const SAFE_DROP_FIELDS: Partial> = {}; export const SUPPORTED_OUTLINE_EXTENSIONS = new Set( DEFAULT_LANGUAGE_CAPABILITIES.flatMap((descriptor) => @@ -149,9 +166,9 @@ export const REQUEST_FIELD_GUIDANCE: Partial> = { context: "context is an output-context budget and is never silently dropped", limit: "limit is an output/page budget and is never silently dropped", scope: - "scope applies to ordinary content search; mode=files rejects this field because its scope is fixed strict, and only redundant strict may be removed", + "scope applies to ordinary content search; mode=files is always strict, so it accepts only the redundant scope=strict", ignorePolicy: - "respect keeps repository ignore rules; include searches ignored files but always excludes .git internals and protected paths", + "respect keeps repository ignore rules and discloses ignored files; include also searches or lists ignored files but always excludes .git internals and protected paths", patterns: "audit accepts named exact-literal patterns and returns one closure receipt", maxFilesToParse: "concept/hybrid automatically batch the requested scope; this optional field sets an advanced hard file ceiling", diff --git a/src/request-contract.ts b/src/request-contract.ts index 13db31f1..62cfaf56 100644 --- a/src/request-contract.ts +++ b/src/request-contract.ts @@ -132,6 +132,21 @@ function plainFilePatternRecovery( ); } +/** `pattern: ".*"` in files mode means "every file": the exact repair omits it. */ +function matchAllFilePatternRecovery( + mode: SiftLightMode, + input: Record, + invalid: readonly string[], +): boolean { + return ( + mode === "files" && + invalid.includes("pattern") && + input.query === undefined && + typeof input.pattern === "string" && + /^(?:\.\*|\.\+|\*+)$/u.test(input.pattern.trim()) + ); +} + function safeNextRequest( input: Record, mode: SiftLightMode, @@ -140,7 +155,7 @@ function safeNextRequest( if (input.redact === true && containsSensitiveText(input)) return undefined; const safe = new Set(SAFE_DROP_FIELDS[mode] ?? []); const plainFilePattern = plainFilePatternRecovery(mode, input, invalid); - if (plainFilePattern) safe.add("pattern"); + if (plainFilePattern || matchAllFilePatternRecovery(mode, input, invalid)) safe.add("pattern"); if (invalid.some((field) => !safe.has(field))) return undefined; const next: Record = { ...input }; for (const field of invalid) delete next[field]; @@ -203,7 +218,9 @@ function fieldsError( action: "retry", reason: plainFilePatternRecovery(mode, input, invalid) ? "For filename discovery, move the plain text from pattern to query and copy nextRequest; all filters are preserved." - : `Remove only ${visibleFields} and copy the exact nextRequest; all other fields are preserved.`, + : matchAllFilePatternRecovery(mode, input, invalid) + ? "A match-all pattern lists every file; omitting query does that, so copy nextRequest; all filters are preserved." + : `Remove only ${visibleFields} and copy the exact nextRequest; all other fields are preserved.`, nextRequest, } : { @@ -458,7 +475,8 @@ export function validateRequestContract(input: SiftLightInput): void { const allowed = new Set(MODE_FIELDS_BY_MODE[mode]); if (mode === "inspect" && raw.sourceCursor !== undefined) { allowed.clear(); - for (const field of ["mode", "sourceCursor", "redact"] as const) allowed.add(field); + // path is checked against the continuation's own source; it cannot redirect it. + for (const field of ["mode", "sourceCursor", "redact", "path"] as const) allowed.add(field); } if ( (mode === "auto" || mode === "summary" || mode === "matches") && diff --git a/src/request.ts b/src/request.ts index 1af6e95d..6cfc4cb9 100644 --- a/src/request.ts +++ b/src/request.ts @@ -108,7 +108,9 @@ export function normalizeRequest(input: RawSearchInput): SearchRequest { throw new SiftLightError("ignorePolicy must be respect or include"); const pattern = input.pattern; if (pattern === undefined) { - throw new SiftLightError("pattern is required when cursor is not provided"); + throw new SiftLightError( + 'pattern is required when cursor is not provided; to list files use mode="files" (query optional), to read a known file use mode="inspect" with path', + ); } const path = input.path?.replace(/^@/, ""); diff --git a/src/runtime.ts b/src/runtime.ts index cbeaa09c..23b8db18 100644 --- a/src/runtime.ts +++ b/src/runtime.ts @@ -1,5 +1,5 @@ import type { SiftLightLocale } from "./config.js"; -import type { SiftLightInput, SiftLightSearchOptions, SiftLightService } from "./service.js"; +import type { SiftLightRequest, SiftLightSearchOptions, SiftLightService } from "./service.js"; import { type SessionSummarySnapshot, SessionSummary } from "./session-summary.js"; import type { ContextBudget, SiftLightResult } from "./types.js"; @@ -12,7 +12,7 @@ export class SiftLightRuntime { } async search( - input: SiftLightInput, + input: SiftLightRequest, cwd: string, signal?: AbortSignal, contextBudget?: ContextBudget, diff --git a/src/service.ts b/src/service.ts index 4510bb2c..77c66701 100644 --- a/src/service.ts +++ b/src/service.ts @@ -7,6 +7,18 @@ import { resolve } from "node:path"; import { CursorError, SiftLightError } from "./errors.js"; import { DISCOVERY_MODE_REQUIRED_ERROR } from "./discovery-errors.js"; import { validateRequestContract } from "./request-contract.js"; +import { normalizeRequestAliases, type SiftLightRequest } from "./request-aliases.js"; + +export type { SiftLightRequest } from "./request-aliases.js"; + +/** Normalizations are part of the evidence: show them in text and details. */ +function withRequestNotes(result: SiftLightResult, notes: readonly string[]): SiftLightResult { + if (notes.length === 0) return result; + return { + text: `${notes.map((note) => `[Request note: ${note}]`).join("\n")}\n${result.text}`, + details: { ...result.details, requestNotes: [...notes] }, + }; +} import { formatMatchMetadataPage, formatMatchPage, @@ -392,10 +404,20 @@ export class SiftLightService { } async search( - input: SiftLightInput, + request: SiftLightRequest, cwd: string, signal?: AbortSignal, options: SiftLightSearchOptions = {}, + ): Promise { + const { input, notes } = normalizeRequestAliases(request); + return withRequestNotes(await this.#searchNormalized(input, cwd, signal, options), notes); + } + + async #searchNormalized( + input: SiftLightInput, + cwd: string, + signal: AbortSignal | undefined, + options: SiftLightSearchOptions, ): Promise { validateRawSearchInput(input); validateRequestContract(input); diff --git a/src/session-summary.ts b/src/session-summary.ts index aaf81aa8..2585cef8 100644 --- a/src/session-summary.ts +++ b/src/session-summary.ts @@ -1,6 +1,6 @@ import type { SiftLightLocale } from "./config.js"; import { SIFT_LIGHT_VERSION } from "./package-version.js"; -import type { SiftLightInput } from "./service.js"; +import type { SiftLightRequest } from "./service.js"; import type { SiftLightResult } from "./types.js"; export const SESSION_STATUS_KEY = "sift-light_session"; @@ -13,7 +13,7 @@ export interface SessionSummarySnapshot { failedCalls: number; } -function isNewQuery(input: SiftLightInput): boolean { +function isNewQuery(input: SiftLightRequest): boolean { return ( input.cursor === undefined && input.sourceCursor === undefined && @@ -21,7 +21,7 @@ function isNewQuery(input: SiftLightInput): boolean { ); } -function wasAutomaticallyOrganized(input: SiftLightInput, result: SiftLightResult): boolean { +function wasAutomaticallyOrganized(input: SiftLightRequest, result: SiftLightResult): boolean { const autoMode = input.mode === undefined || input.mode === "auto"; return autoMode && input.limit === undefined && result.details.summaryFilesShown !== undefined; } @@ -59,7 +59,7 @@ export class SessionSummary { failedCalls: 0, }; - record(input: SiftLightInput, result: SiftLightResult): void { + record(input: SiftLightRequest, result: SiftLightResult): void { if (!isNewQuery(input)) return; this.#snapshot.queries += 1; if (result.details.status === "complete") this.#snapshot.completeQueries += 1; diff --git a/src/source-inspection.ts b/src/source-inspection.ts index 3b553a67..840ce928 100644 --- a/src/source-inspection.ts +++ b/src/source-inspection.ts @@ -33,6 +33,8 @@ export interface SourceInspectionTarget { /** Raw source byte offset; retained-match focus is line-relative. */ absoluteFocus?: number; focus?: number; + /** Structure already proven by the analysis that produced `range`. */ + structure?: StructureDetails; expectedRevision?: SourceRevision; unverified?: boolean; retry?: SiftLightInput; @@ -121,7 +123,20 @@ async function prepare( let range = target.range; let details: StructureDetails = { status: "no-symbol" }; const language = syntaxLanguage(document.path); - if (document.utf8 && language && language !== "go") { + if (target.range && target.structure) { + document.checkRange(target.range); + const lines = { + startLine: document.lineAt(target.range.start), + endLine: document.lineAt(Math.max(target.range.start, target.range.end - 1)), + }; + // Whole lines keep leading modifiers such as `export` or decorators readable. + range = document.lineRange(lines.startLine, lines.endLine); + details = { + ...target.structure, + range: lines, + ...(target.structure.symbol ? { symbol: { ...target.structure.symbol, range: lines } } : {}), + }; + } else if (document.utf8 && language && language !== "go") { const syntax = await access.syntax(document); details = { status: @@ -205,7 +220,9 @@ async function prepare( Math.min(document.lineStarts.length, target.line + 10), ); const boundary: Exclude = target.range - ? "requested-range" + ? target.structure + ? "syntax" + : "requested-range" : details.status === "available" && details.range ? "syntax" : "line-window"; @@ -692,8 +709,16 @@ export async function continueSource( cursor: string, access: SourceAccess, continuations: SourceContinuations, + expectedPath?: string, ): Promise { const state = continuations.resolve(cursor); + if ( + expectedPath !== undefined && + resolve(access.cwd, expectedPath.replace(/^@/, "")) !== resolve(access.cwd, state.source.path) + ) + throw new SiftLightError( + `sourceCursor continues ${JSON.stringify(state.source.path)}, not ${JSON.stringify(expectedPath)}; copy the returned nextRequest exactly`, + ); const document = await access.load(state.source.path, state.source); const page = sourcePage(document, state.remaining, MAX_RESULT_BYTES - 1400); const next = continuations.advance(cursor, page.fragment); diff --git a/src/tool-schema.ts b/src/tool-schema.ts index 58297e28..3013eb9d 100644 --- a/src/tool-schema.ts +++ b/src/tool-schema.ts @@ -10,6 +10,7 @@ import { } from "./analysis-limits.js"; import { MAX_CONTEXT_LINES, + MAX_INSPECT_REQUEST_TARGETS, MAX_INSPECT_TARGETS, MAX_PAGE_SIZE, MAX_SELECTED_PATHS, @@ -25,6 +26,7 @@ import { SIFT_LIGHT_MODES, fieldGuidance, } from "./request-contract.js"; +import { MODE_ALIASES } from "./request-aliases.js"; function stringEnum( values: Values, @@ -68,9 +70,9 @@ export const siftLightSchema = Type.Object({ ), allOf: Type.Optional( Type.Array(Type.String({ maxLength: MAX_PATH_CHARACTERS }), { - minItems: 2, + minItems: 1, maxItems: 3, - description: `${fieldGuidance("allOf")}. 2-3 distinct terms must occur in one file (default) or one function.`, + description: `${fieldGuidance("allOf")}. 1-3 distinct terms must occur in one file (default) or one function; one term lists the files or functions containing it.`, }), ), within: Type.Optional( @@ -95,7 +97,7 @@ export const siftLightSchema = Type.Object({ { minItems: 1, description: - "Filter each single-pattern occurrence by syntax role (JS/TS/TSX/Go). Roles may be candidates, especially Go call/conversion ambiguity. Cannot combine with allOf.", + "Filter each single-pattern occurrence by syntax role (JS/TS/TSX/Go parsed; Python lexical: comment, string, code, declaration, import, call candidates). Roles may be candidates, especially Go call/conversion and Python calls. Cannot combine with allOf.", }, ), ), @@ -151,7 +153,7 @@ export const siftLightSchema = Type.Object({ minItems: 1, maxItems: MAX_SELECTED_PATHS, description: - "Exact retained files to select together from a cursor. A new search accepts one path; split multiple roots into separate requests.", + "Exact retained files to select together from a cursor. A new search accepts one path; split multiple roots into separate requests. With mode=inspect and no cursor, opens each file from line 1 (same as targets with line 1).", }), ), glob: Type.Optional( @@ -193,7 +195,7 @@ export const siftLightSchema = Type.Object({ ignorePolicy: Type.Optional( stringEnum(["respect", "include"] as const, { description: - "respect (default) honors ignore rules and reports policy-filtered coverage when files are omitted. include searches ignored files while still excluding .git internals and protected paths.", + "respect (default) honors ignore rules and reports policy-filtered coverage when files are omitted, including mode=files. include searches or lists ignored files while still excluding .git internals and protected paths.", }), ), patterns: Type.Optional( @@ -264,19 +266,19 @@ export const siftLightSchema = Type.Object({ Type.Integer({ minimum: 1, maximum: MAX_PAGE_SIZE, - description: `${fieldGuidance("limit")}. Ordinary search only: explicit detail-page match limit (max 100). Normally omit to preserve automatic summarization; analysis and inspect modes reject it.`, + description: `${fieldGuidance("limit")}. Ordinary search: explicit detail-page match limit (max 100); mode=files: files per page (default 30). Normally omit to preserve automatic summarization; other analysis and inspect modes reject it.`, }), ), mode: Type.Optional( - stringEnum(SIFT_LIGHT_MODES, { - description: `Ordinary search defaults to auto; summary/matches request explicit pages. capabilities returns a compact names-only project inventory and per-language supported modes without loading providers. files uses query, structure uses an AST pattern, concept uses a required natural-language query, and hybrid uses one query for exact literal plus concept evidence in a single snapshot. validate rechecks saved search or analysis sources against their recorded origin. await waits for an existing long-running concept or hybrid operation without restarting it; cancel explicitly cancels one and waits for owned cleanup. Copy the returned nextRequest exactly and do not repeat the original query. Waiting is an operation state, not evidence. Validation details report the requested scope, comparison target, coverage, and freshness as current, stale, or unknown; partial coverage is retained during validation. inspect/outline/imports/tests retain their documented location selectors. Syntax results are static evidence; concept and related-test results remain candidates. ${MODE_CONTRACT_DESCRIPTION}`, + stringEnum([...SIFT_LIGHT_MODES, ...MODE_ALIASES], { + description: `Ordinary search defaults to auto; summary/matches request explicit pages. capabilities returns a compact names-only project inventory and per-language supported modes without loading providers. files uses query, structure uses an AST pattern, concept uses a required natural-language query, and hybrid uses one query for exact literal plus concept evidence in a single snapshot. validate rechecks saved search or analysis sources against their recorded origin. await waits for an existing long-running concept or hybrid operation without restarting it; cancel explicitly cancels one and waits for owned cleanup. Copy the returned nextRequest exactly and do not repeat the original query. Waiting is an operation state, not evidence. Validation details report the requested scope, comparison target, coverage, and freshness as current, stale, or unknown; partial coverage is retained during validation. inspect/outline/imports/tests retain their documented location selectors. Syntax results are static evidence; concept and related-test results remain candidates. ${MODE_CONTRACT_DESCRIPTION} anyOf/allOf are field names, not modes; as mode values they are accepted aliases for omitting mode, disclosed in a request note.`, }), ), line: Type.Optional( Type.Number({ description: - "1-indexed source line for path inspection/navigation. Omit with matchIndex, matchIndices or targets.", + "1-indexed source line for path inspection/navigation; path-only inspect starts at line 1. Omit with matchIndex, matchIndices or targets.", }), ), matchIndex: Type.Optional( @@ -288,9 +290,8 @@ export const siftLightSchema = Type.Object({ matchIndices: Type.Optional( Type.Array(Type.Integer({ minimum: 1 }), { minItems: 1, - maxItems: MAX_INSPECT_TARGETS, - description: - "Inspect up to five visible match numbers together using the same cursor; mutually exclusive with matchIndex, path, line and targets.", + maxItems: MAX_INSPECT_REQUEST_TARGETS, + description: `Inspect visible match numbers together using the same cursor; mutually exclusive with matchIndex, path, line and targets. One response inspects ${String(MAX_INSPECT_TARGETS)}; the rest are returned as an exact nextRequest.`, }), ), targets: Type.Optional( @@ -301,9 +302,8 @@ export const siftLightSchema = Type.Object({ }), { minItems: 1, - maxItems: MAX_INSPECT_TARGETS, - description: - "Inspect known path/line locations together without a cursor. The complete batch shares one 16 KiB response budget.", + maxItems: MAX_INSPECT_REQUEST_TARGETS, + description: `Inspect known path/line locations together without a cursor. One response inspects ${String(MAX_INSPECT_TARGETS)} within a shared 16 KiB budget; the rest are returned as an exact nextRequest.`, }, ), ), diff --git a/src/tui/layout.ts b/src/tui/layout.ts index 0038ceb3..1f3c98b7 100644 --- a/src/tui/layout.ts +++ b/src/tui/layout.ts @@ -1,6 +1,6 @@ import { truncateToWidth, visibleWidth } from "@earendil-works/pi-tui"; import type { SiftLightLocale } from "../config.js"; -import type { SiftLightInput } from "../service.js"; +import type { SiftLightRequest } from "../service.js"; import type { Dashboard } from "./dashboard.js"; import { candy, type SiftLightTheme } from "./palette.js"; @@ -254,7 +254,7 @@ export function renderDashboard( } export function renderSiftLightCallLines( - input: SiftLightInput, + input: SiftLightRequest, locale: SiftLightLocale, theme: SiftLightTheme, width: number, diff --git a/src/tui/renderers.ts b/src/tui/renderers.ts index e0456862..6d7a448e 100644 --- a/src/tui/renderers.ts +++ b/src/tui/renderers.ts @@ -1,6 +1,6 @@ import { type Component } from "@earendil-works/pi-tui"; import type { SiftLightLocale } from "../config.js"; -import type { SiftLightInput } from "../service.js"; +import type { SiftLightRequest } from "../service.js"; import type { SiftLightDetails } from "../types.js"; import { dashboard } from "./dashboard.js"; import { fit, renderDashboard, renderSiftLightCallLines, type SiftLightTheme } from "./layout.js"; @@ -47,7 +47,7 @@ function failure(text: string, locale: SiftLightLocale): string { } export function renderSiftLightCall( - input: SiftLightInput, + input: SiftLightRequest, locale: SiftLightLocale, theme: SiftLightTheme, ): Component { diff --git a/src/types.ts b/src/types.ts index c517faa9..f5adcd10 100644 --- a/src/types.ts +++ b/src/types.ts @@ -23,6 +23,8 @@ export const ESTIMATED_CHARACTERS_PER_TOKEN = 4; export const DEFAULT_SUMMARY_FILE_LIMIT = 30; export const MAX_SELECTED_PATHS = 20; export const MAX_INSPECT_TARGETS = 5; +/** A request may name more targets; they are served in pages of MAX_INSPECT_TARGETS. */ +export const MAX_INSPECT_REQUEST_TARGETS = 20; export const MAX_DISPLAYED_OCCURRENCES = 20; export const MAX_STORED_MATCHES = 50_000; export const MAX_STORED_OCCURRENCES = 200_000; @@ -310,6 +312,8 @@ export interface SiftLightDetails { source?: SourceExcerptDetails; inspections?: InspectBatchItemDetails[]; scope?: SearchScopeDetails; + /** Disclosed request normalizations, such as a field name sent as mode. */ + requestNotes?: string[]; redactedCount?: number; redactionRequested?: boolean; redactionApplied?: boolean;