diff --git a/.cursor/rules/use-local-code-intel.mdc b/.cursor/rules/use-local-code-intel.mdc index 4e8c9aa..8412358 100644 --- a/.cursor/rules/use-local-code-intel.mdc +++ b/.cursor/rules/use-local-code-intel.mdc @@ -9,8 +9,8 @@ MCP namespace: `user-local-code-intelligence`. The corpus is already embedded in ## Always do this first -1. `search_codebase` (pass `max_tokens` 800–1500). Only the query is embedded locally; stored vectors retrieve snippets. -2. `search_symbol` for a known name; `find_references` for occurrences. +1. For a new or broad coding task, `get_task_context`. Only the query is embedded locally; stored vectors retrieve snippets. +2. `search_symbol` for a known name; `search_codebase` for conceptual exploration; `find_references` for occurrences. 3. `get_file_context` with `start_line`/`end_line` for the range you will edit. 4. `get_repo_context` / `list_indexed_repos` / `index_status` for orientation. @@ -20,4 +20,4 @@ MCP namespace: `user-local-code-intelligence`. The corpus is already embedded in - Read whole files to "see how it works" - Ask the cloud model to embed or index code -If the workspace is a parent (e.g. Savor), `list_indexed_repos` and pass `repo`. If unindexed, run `code-intel setup --repo ` via Shell, then search again. +If retrieval reports low confidence, the index is stale, or the path is unindexed, targeted filesystem search is allowed. If the workspace is a parent (e.g. Savor), `list_indexed_repos` and pass `repo`. If unindexed, run `code-intel setup --repo ` via Shell, then search again. diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 080e849..8784465 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -7,3 +7,6 @@ - [ ] `npm run typecheck` - [ ] `npm run test:unit` - [ ] (optional) `npm test` with Ollama + `nomic-embed-text` if you touched indexing or MCP search +- [ ] Retrieval changes include before/after benchmark evidence +- [ ] CLI, MCP, privacy, or storage changes include documentation updates +- [ ] No credentials, private source, organization endpoints, or local index data are included diff --git a/CHANGELOG.md b/CHANGELOG.md new file mode 100644 index 0000000..13e0e6b --- /dev/null +++ b/CHANGELOG.md @@ -0,0 +1,53 @@ +# Changelog + +All notable changes to this project are documented here. The project follows [Semantic Versioning](https://semver.org). + +## [Unreleased] + +## [0.2.0] - 2026-09-21 + +### Added + +- `get_task_context`, a high-level MCP tool that returns a ranked, token-budgeted context package for a coding task. +- `code-intel context`, retrieval modes, confidence reporting, and `--explain` traces. +- Deterministic task intent, relationship expansion, diversity controls, and hard context budgets. +- Exact symbol and basename ranking floors, score-ordered files, and test/config intent handling. +- Lightweight chunk metadata for imports, exports, referenced identifiers, tests, and configuration files. +- TypeScript path-alias and Python module expansion. +- Labeled retrieval benchmark with Precision@5, coverage@5, Recall@10, MRR, NDCG, token reduction, latency, and auditable retrieved paths. +- Per-file watcher updates for small event batches, content-sample freshness detection, and persisted watcher errors in `index_status`. +- Actionable CLI and MCP recovery guidance plus confidence-based targeted filesystem fallback. + +### Changed + +- Cursor rules, skill, hooks, and MCP instructions now prefer `get_task_context` for broad tasks. +- Hybrid retrieval now combines semantic, keyword, symbol, path, structure, relationship, test, and recency signals. +- MCP starts an incremental catch-up when it opens an indexed workspace. +- Documentation now separates measured benchmark claims from estimates and describes the privacy boundary for each provider. + +### Fixed + +- Duplicate symbol chunk IDs no longer abort a LanceDB merge. +- Unchanged chunks reuse embeddings while refreshing metadata and line ranges. +- Large event bursts fall back to full incremental discovery instead of issuing many individual writes. +- Code-change queries that rank documentation first are treated as low confidence. + +## [0.1.1] - 2026-09-07 + +### Added + +- Guided onboarding, Ollama health checks, and Cursor integration. +- OpenAI-compatible embedding provider support and Git-aware multi-repository indexing. +- Parallel file indexing and IVF-PQ vector indexing for larger tables. + +## [0.1.0] - 2026-09-01 + +### Added + +- Initial public npm release. +- Structural chunking, local LanceDB storage, incremental indexing, hybrid search, MCP tools, and file watching. + +[Unreleased]: https://github.com/pranitmodi/code-intel/compare/v0.2.0...HEAD +[0.2.0]: https://github.com/pranitmodi/code-intel/compare/v0.1.1...v0.2.0 +[0.1.1]: https://github.com/pranitmodi/code-intel/releases/tag/v0.1.1 +[0.1.0]: https://github.com/pranitmodi/code-intel/releases/tag/v0.1.0 diff --git a/CODE_INTEL_NEXT_PHASE.md b/CODE_INTEL_NEXT_PHASE.md index 2d01eda..473aa34 100644 --- a/CODE_INTEL_NEXT_PHASE.md +++ b/CODE_INTEL_NEXT_PHASE.md @@ -1,5 +1,7 @@ # code-intel: Next-Generation Agent Context Architecture +> Historical implementation specification for the task-context work introduced in v0.2.0. Some acceptance items are now implemented and some remain future work. See the maintained [architecture](docs/ARCHITECTURE.md), [benchmarks](docs/BENCHMARKS.md), and [changelog](CHANGELOG.md) for current behavior. + ## Purpose This document is the implementation specification for evolving `code-intel` from a local semantic code-search server into a **persistent context layer for AI coding agents**. diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md new file mode 100644 index 0000000..c51f3d0 --- /dev/null +++ b/CODE_OF_CONDUCT.md @@ -0,0 +1,41 @@ +# Code of Conduct + +## Our pledge + +We pledge to make participation in the `code-intel` community a harassment-free experience for everyone, regardless of age, body size, disability, ethnicity, sex characteristics, gender identity and expression, experience level, education, socioeconomic status, nationality, personal appearance, race, caste, color, religion, sexual identity and orientation, or technology choices. + +We pledge to act and interact in ways that contribute to an open, welcoming, diverse, inclusive, and healthy community. + +## Expected behavior + +Examples of positive behavior include: + +- showing empathy and respect; +- giving and accepting constructive technical feedback; +- focusing discussion on evidence, reproducible behavior, and project goals; +- respecting privacy, especially when reports involve private source or credentials; +- accepting responsibility and learning from mistakes. + +Unacceptable behavior includes: + +- harassment, threats, insults, or discriminatory language; +- trolling, sustained disruption, or personal attacks; +- publishing another person's private information without permission; +- posting private source, credentials, or sensitive logs; +- conduct that would reasonably be considered inappropriate in a professional setting. + +## Enforcement + +Project maintainers are responsible for clarifying and enforcing these standards. They may edit, remove, or reject comments, commits, code, issues, and other contributions that violate this Code of Conduct, and may temporarily or permanently ban contributors for behavior they consider harmful. + +Report conduct concerns privately to the repository owner through the contact options on the [GitHub profile](https://github.com/pranitmodi). Do not use a public issue for a sensitive report. + +All reports will be reviewed promptly and fairly. Maintainers will respect the privacy and security of reporters. + +## Scope + +This Code of Conduct applies in project spaces and when an individual is officially representing the project in public spaces. + +## Attribution + +This Code of Conduct is adapted from the [Contributor Covenant](https://www.contributor-covenant.org), version 2.1. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index d1fd0e2..494a83c 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,6 +2,8 @@ Thanks for helping. The goal of this project is a **local-first** code index: embeddings and source stay on the contributor's machine, and AI agents query that index over MCP instead of re-scanning the tree. +Before changing a subsystem, read [Architecture](docs/ARCHITECTURE.md). Retrieval changes must follow the measurement rules in [Benchmarks](docs/BENCHMARKS.md). Participation is governed by the [Code of Conduct](CODE_OF_CONDUCT.md). + ## Prerequisites - Node.js 20+ @@ -30,6 +32,7 @@ Unit tests in `tests/unit` never need Ollama. Integration tests in `tests/integr ```bash npm run typecheck npm run test:unit # CI default +npm run benchmark:retrieval # labeled retrieval metrics (needs an index + embeddings) npm test # unit + integration (integration skipped without Ollama) npm run dev -- status # CLI from source, no build step ``` @@ -50,24 +53,51 @@ Indexes live under `~/.local-code-intelligence` (or `CODE_INTEL_DB_PATH`), not i `tests/integration/mcp.test.ts` spawns `code-intel mcp` over stdio with `@modelcontextprotocol/client` — the same transport Cursor uses. When you add or rename a tool, update that file's expected tool list. +## Retrieval and benchmark changes + +Retrieval quality is part of the product contract. A ranking change should include: + +- focused unit tests for the intended signal; +- `code-intel search "" --explain` or `code-intel context "" --explain` output for the affected case; +- before/after `npm run benchmark:retrieval` results; +- an explanation for any task-label or relevance-grade change. + +Do not improve metrics by broadening the context until it resembles a tree scan. The release floor is relevant-file coverage@5 ≥ 0.60, Recall@10 ≥ 0.75, MRR ≥ 0.70, and average per-task token reduction ≥ 60% on the repository dataset. + +When adding a benchmark task, use a realistic engineering request and defensible relevant files. Avoid tasks designed around the current ranker's implementation. + ## Publishing (maintainers) +Releases are made from a clean, tested `main` commit: + ```bash +npm run typecheck +npm run test:unit +npm run build +npm run benchmark:retrieval +npm audit --omit=dev +npm pack --dry-run npm login npm publish --access public ``` The package name is `@pranitmodi/code-intel` (scoped; unscoped `code-intel` is blocked by npm as too similar to `codeintel`). -`prepublishOnly` builds `dist/` and runs unit tests. The tarball includes `dist/` plus README and LICENSE; indexes under `~/.local-code-intelligence` are never published. +`prepublishOnly` builds `dist/` and runs unit tests. Inspect the tarball before publishing. It must contain only the built CLI/server and public documentation—never indexes, `.code-intel` config, credentials, private paths, benchmark secrets, or `~/.local-code-intelligence` data. + +Update [CHANGELOG.md](CHANGELOG.md), bump `package.json` and `package-lock.json` together, tag the exact published commit, verify it with `npm view`, and create a matching GitHub release. ## Pull requests -- Keep changes focused; match the existing module boundaries (`src/discovery`, `src/chunker`, `src/embeddings`, `src/vector-store`, `src/indexer`, `src/search`, `src/mcp`). +- Keep changes focused; match the existing module boundaries (`src/discovery`, `src/chunker`, `src/embeddings`, `src/vector-store`, `src/indexer`, `src/search`, `src/retrieval`, `src/benchmark`, `src/mcp`). - Do not commit `node_modules/`, `dist/`, or anything under `~/.local-code-intelligence`. - Do not add cloud embedding APIs as the default path; local Ollama is the contract. - Run `npm run typecheck` and `npm run test:unit` before opening a PR. +- Keep public examples provider-neutral. Do not commit organization-specific endpoints, usernames, paths, source, or benchmark credentials. +- Update documentation when changing CLI commands, MCP tools, privacy boundaries, storage, or watcher behavior. ## Security Secret-shaped files (`.env*`, keys, credentials) are excluded from indexing by default. Do not weaken those filters without a documented, opt-in config flag. + +Do not report vulnerabilities or credential exposure in a public issue. Follow [SECURITY.md](SECURITY.md) and use GitHub private vulnerability reporting. diff --git a/README.md b/README.md index d4d75a7..42fdd69 100644 --- a/README.md +++ b/README.md @@ -1,29 +1,47 @@ # code-intel -AI coding agents burn tokens re-scanning your repository every turn. **code-intel** indexes the tree once with local [Ollama](https://ollama.com) or an OpenAI-compatible embedding service, stores vectors on disk in [LanceDB](https://lancedb.github.io/lancedb/), and serves targeted snippets over the [Model Context Protocol](https://modelcontextprotocol.io) so Cursor, VS Code, Claude Code, Codex, or any MCP client can search without dumping the whole codebase into context. +[![npm](https://img.shields.io/npm/v/@pranitmodi/code-intel)](https://www.npmjs.com/package/@pranitmodi/code-intel) +[![CI](https://github.com/pranitmodi/code-intel/actions/workflows/ci.yml/badge.svg)](https://github.com/pranitmodi/code-intel/actions/workflows/ci.yml) +[![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE) +[![Node.js](https://img.shields.io/badge/Node.js-20%2B-339933)](package.json) -Ollama is the private, local default. When an OpenAI-compatible provider is configured, code chunks and search queries are sent to that provider for embedding; embeddings and the vector database remain on disk locally. Contributions welcome — see [CONTRIBUTING.md](CONTRIBUTING.md). MIT licensed. +**Persistent, local-first repository context for AI coding agents.** -A local-first semantic code indexing and retrieval server. It indexes a repository into a local vector database once, keeps that index incrementally in sync as files change, and exposes semantic + keyword + symbol search to AI coding agents. +Coding agents repeatedly list, search, and read the same repository. `code-intel` indexes that knowledge once, keeps it current as files change, and gives the agent a small, ranked context package over the [Model Context Protocol](https://modelcontextprotocol.io). -## What it does +The default path uses local [Ollama](https://ollama.com) embeddings and on-disk [LanceDB](https://lancedb.github.io/lancedb/). Cursor, VS Code, Claude Code, Codex, and other MCP clients can share the same index. -- Parses your repository with Tree-sitter and chunks it at structural boundaries (functions, classes, methods, interfaces, ...), not fixed character windows. -- Embeds each chunk via local **Ollama** (default) or a configured **OpenAI-compatible** endpoint and stores vectors + metadata in a local **LanceDB** database. -- Re-indexes incrementally: unchanged files/chunks are never re-embedded; only new or edited chunks are. -- Serves `search_codebase`, `search_symbol`, `get_file_context`, `get_repo_context`, `find_references`, `list_indexed_repos`, and `index_status` as MCP tools over stdio, so any MCP-capable IDE/agent can query the same persistent index instead of re-scanning or re-embedding the repository itself. +> **Measured on this repository:** task-context retrieval used **73.9% fewer estimated input tokens** than the workspace-scan baseline, with **0.96 Recall@10**, **0.84 MRR**, and **708 ms p95**. This is an eight-task TypeScript regression benchmark, not a universal cost guarantee. See [methodology and complete results](docs/BENCHMARKS.md). -## Architecture +## Why use it? -```text -Git Repo -> Discovery -> Parser/Chunker -> Hasher -> Embedding Provider -> LanceDB - | - MCP Server (stdio) <-----+ - | - Cursor / VS Code / Claude Code / Codex +- **Less repeated context:** return relevant symbols and files instead of dumping the tree. +- **Task-aware retrieval:** `get_task_context` combines semantic, keyword, symbol, path, and relationship signals. +- **Incremental by design:** only changed chunks are embedded; unchanged vectors are reused. +- **Local-first privacy:** Ollama is the default and no telemetry is enabled. +- **Agent-safe fallback:** stale or low-confidence retrieval permits targeted filesystem search. +- **Inspectable quality:** explain traces and a labeled benchmark expose ranking, misses, token budgets, and latency. + +## How it works + +```mermaid +flowchart LR + repo[Repository] + index[Incremental structural index] + db[(Local LanceDB)] + mcp[MCP context server] + agent[AI coding agent] + + repo -->|"parse, hash, embed changes"| index --> db + agent -->|"coding task"| mcp + mcp -->|"hybrid search and expansion"| db + db -->|"ranked chunks"| mcp + mcp -->|"token-budgeted context"| agent ``` -Each stage is an independent module (`src/discovery`, `src/parser`, `src/chunker`, `src/hashing`, `src/embeddings`, `src/vector-store`, `src/indexer`, `src/search`, `src/mcp`, `src/cli`) so the embedding provider or vector database can be swapped without touching the others. +The write path discovers files, creates Tree-sitter symbol chunks where supported, hashes content, and embeds only changes. The read path analyzes a task, retrieves candidates, expands imports/references/tests/configuration, removes redundancy, and packs the result under a hard token budget. + +Read the full [architecture](docs/ARCHITECTURE.md), [benchmark methodology](docs/BENCHMARKS.md), and [security model](SECURITY.md). ## Installation @@ -61,7 +79,7 @@ ollama pull nomic-embed-text # default embedding model, ~274MB ## OpenAI-compatible embeddings -Any OpenAI-compatible `/embeddings` service works: a company proxy, OpenAI, Azure, or a local gateway. +Any compatible `/embeddings` service can be used: a hosted provider, an organization-managed gateway, or a local service. Review that provider's source-code retention and training policies before indexing private repositories. ### One-command setup @@ -71,18 +89,18 @@ From the repository, run: code-intel wizard ``` -It first asks whether to use **local Ollama** or a **company OpenAI-compatible proxy**. If you choose the proxy, it prompts for API key (hidden in the terminal, not saved to disk), optional username, model, and endpoint, then validates, rebuilds the index, and wires Cursor. `code-intel corporate-setup` is the same flow already set to the company-proxy route. +It first asks whether to use **local Ollama** or an **OpenAI-compatible provider**. If you choose the provider, it prompts for API key (hidden in the terminal and not saved to YAML), optional username, model, and endpoint, then validates, rebuilds the index, and wires Cursor. `code-intel corporate-setup` is an alias that starts directly on the managed-provider route. -For a company-managed, non-interactive rollout: +For a non-interactive managed rollout: ```bash code-intel corporate-setup --non-interactive \ - --model Qwen3-Embedding-8B \ - --base-url https://llm-proxy-api.ai.eng.netapp.com \ + --model your-embedding-model \ + --base-url https://embeddings.example.com/v1 \ --embeddings-path /embeddings ``` -That path still requires `CODE_INTEL_EMBEDDING_API_KEY` in the environment (or `--api-key` for a one-off; it is still not written to YAML). Corporate TLS uses Node's system certificate store and never disables verification. Use a current Node release supporting `--use-system-ca`. On older/custom Node installations, set `NODE_EXTRA_CA_CERTS` to the company CA PEM file. +That path still requires `CODE_INTEL_EMBEDDING_API_KEY` in the environment (or `--api-key` for a one-off; it is not written to YAML). Managed TLS can use Node's system certificate store without disabling verification. Use a current Node release supporting `--use-system-ca`; on older/custom installations, set `NODE_EXTRA_CA_CERTS` to the organization CA PEM file. ### Manual configuration @@ -93,8 +111,8 @@ Model name, base URL, and path are all configurable. Set them once globally so e ```yaml embedding: provider: openai-compatible - model: Qwen3-Embedding-8B # or text-embedding-3-small, nomic-embed-text, ... - base_url: https://llm-proxy-api.ai.eng.netapp.com + model: your-embedding-model + base_url: https://embeddings.example.com/v1 embeddings_path: /embeddings # use /v1/embeddings if your API lives under /v1 batch_size: 32 timeout_ms: 60000 @@ -105,16 +123,16 @@ You can also set a per-repo `.code-intel/config.yaml`, environment variables, or ```bash export CODE_INTEL_EMBEDDING_PROVIDER=openai-compatible -export CODE_INTEL_EMBEDDING_MODEL='Qwen3-Embedding-8B' -export CODE_INTEL_EMBEDDING_BASE_URL='https://llm-proxy-api.ai.eng.netapp.com' +export CODE_INTEL_EMBEDDING_MODEL='your-embedding-model' +export CODE_INTEL_EMBEDDING_BASE_URL='https://embeddings.example.com/v1' export CODE_INTEL_EMBEDDING_PATH='/embeddings' export CODE_INTEL_EMBEDDING_API_KEY='your-key' export CODE_INTEL_EMBEDDING_USER='your-username' # only if the endpoint requires a user field code-intel doctor code-intel setup --embedding-provider openai-compatible \ - --embedding-model Qwen3-Embedding-8B \ - --embedding-base-url https://llm-proxy-api.ai.eng.netapp.com + --embedding-model your-embedding-model \ + --embedding-base-url https://embeddings.example.com/v1 ``` Put the same `CODE_INTEL_EMBEDDING_*` variables in the MCP server environment so the IDE process can embed search queries. Keys are never read from YAML. If the repository was already indexed with another model, run `code-intel rebuild`. @@ -149,18 +167,23 @@ code-intel index full/incremental index code-intel watch incremental re-index on file changes code-intel repos list every locally indexed repository code-intel search "" hybrid semantic+keyword+symbol search +code-intel search "" --explain score breakdown, diversity, token budget +code-intel context "" assemble a task-oriented context package +code-intel context "" --mode minimal --max-tokens 8000 --explain code-intel symbol exact/fuzzy symbol lookup code-intel file [--start N --end M] exact source content code-intel status repo/index status code-intel doctor diagnose embedding provider/model/database health code-intel doctor --fix same, and pull a missing Ollama model -code-intel rebuild wipe and fully re-index +code-intel rebuild wipe and fully re-index (backfills extra_metadata) code-intel clean remove the local index (not your source) code-intel cursor-install merge ~/.cursor/mcp.json and write the user rule code-intel mcp start the MCP server over stdio (watch on by default) code-intel mcp --no-watch start the MCP server without file watchers code-intel savings estimate token/$ savings vs tree scans code-intel savings --benchmark re-run the Grep vs search A/B on indexed repos +code-intel benchmark labeled retrieval quality vs workspace-scan baseline +code-intel benchmark --format json --task ``` Every command accepts `--repo ` to target a repository other than the current directory — this is what makes MCP configuration below work regardless of the IDE's spawn working directory. @@ -183,7 +206,18 @@ Embedding knobs (also available as YAML / env) can be passed on any command: code-intel savings --benchmark --rate 3 --turns 15 ``` -`--benchmark` runs that A/B on every indexed repo (needs the configured embedding provider and `rg`). Live MCP searches and blocked workspace Grep/Glob (from `cursor-install` hooks) append to `~/.local-code-intelligence/usage.jsonl`. Re-run `code-intel savings` after a real coding session to see session totals. Dollar figures use `--rate` as dollars per million **input** tokens (default `$3`, a Sonnet-class list price). `--turns` compounds a dump that would have stayed in the chat. +`--benchmark` runs that A/B on every indexed repo (needs the configured embedding provider and `rg`). Live MCP searches and blocked workspace Grep/Glob (from `cursor-install` hooks) append to `~/.local-code-intelligence/usage.jsonl`. Re-run `code-intel savings` after a real coding session to see session totals. + +Dollar figures use `--rate` as dollars per million **input** tokens. `--turns` models a dump remaining in later turns. Both are illustrative: provider caching, client behavior, pricing, and context management determine the actual bill. + +The separate labeled retrieval benchmark checks whether smaller context is still relevant: + +```bash +code-intel benchmark +code-intel benchmark --format json --task known-search-codebase +``` + +See [Benchmarks](docs/BENCHMARKS.md) for the baseline, formulas, complete measured result, dataset ceiling, limitations, and reproduction steps. ## MCP configuration @@ -206,7 +240,7 @@ Reload/trust the server when prompted, then ask Copilot Chat to use the tools (o ### Cursor -The reliable setup after `npm install -g code-intel` (or `npm link` from a checkout): +The reliable setup after `npm install -g @pranitmodi/code-intel` (or `npm link` from a checkout): ```bash code-intel cursor-install @@ -215,7 +249,7 @@ code-intel cursor-install That merges `~/.cursor/mcp.json` (it will not drop other MCP servers), writes an always-on user rule and skill, and installs hooks that: - Inject the indexed-repo list at session start -- Block workspace-wide Grep/Glob and Task `explore` so the agent has to hit `search_codebase` first +- Block workspace-wide Grep/Glob and Task `explore` so the agent has to hit `get_task_context` / `search_codebase` first (targeted filesystem search is allowed after low-confidence retrieval) Then reload MCP in Cursor (Settings → MCP). @@ -223,7 +257,7 @@ The checked-in example is [examples/mcp/cursor.mcp.json](examples/mcp/cursor.mcp Note the different top-level key (`mcpServers` vs VS Code's `servers`) — this is a real difference between the two clients' config formats, not a typo. -Both IDEs will then list `search_codebase`, `search_symbol`, `get_file_context`, `get_repo_context`, `find_references`, `list_indexed_repos`, and `index_status` as available tools, backed by the index you already built with `code-intel setup` — the agent never re-scans or re-embeds your repository itself. +Both IDEs will then list `get_task_context`, `search_codebase`, `search_symbol`, `get_file_context`, `get_repo_context`, `find_references`, `list_indexed_repos`, and `index_status`. Retrieval uses the persistent index first; targeted filesystem search remains available when the index is stale, missing, or low confidence. ## Database location @@ -231,11 +265,13 @@ Default: `~/.local-code-intelligence/repos//{db,metadata,state.json,log ## Privacy behavior -- With the default Ollama provider, source code, chunks, embeddings, and the vector database stay on this machine. +- With the default local Ollama host, source code, chunks, embeddings, and the vector database stay on this machine. A remote Ollama host is a remote provider. - With an OpenAI-compatible provider, chunks and semantic search queries are sent to the configured endpoint; embeddings and LanceDB remain local. - No telemetry or third-party cloud API is enabled by default. - `.env*`, private keys, and other secret-shaped files are excluded by default (`security.allow_sensitive_files: false`); files with likely secret *values* are skipped even if their name would otherwise be allowed. +Indexes contain source text. MCP configuration may contain provider credentials. Read the [security policy](SECURITY.md) before indexing private code. + ## Configuration `.code-intel/config.yaml` (repo-level) or `~/.local-code-intelligence/config.yaml` (global), merged over built-in defaults, then overridden by environment variables, then CLI flags. Environment variables: `CODE_INTEL_EMBEDDING_PROVIDER`, `CODE_INTEL_EMBEDDING_MODEL`, `CODE_INTEL_EMBEDDING_HOST`, `CODE_INTEL_EMBEDDING_BASE_URL`, `CODE_INTEL_EMBEDDING_PATH`, `CODE_INTEL_EMBEDDING_BATCH_SIZE`, `CODE_INTEL_EMBEDDING_TIMEOUT_MS`, `CODE_INTEL_USE_SYSTEM_CA`, `CODE_INTEL_EMBEDDING_API_KEY`, `CODE_INTEL_EMBEDDING_USER`, `CODE_INTEL_DB_PATH`, `CODE_INTEL_WATCH`, `CODE_INTEL_INDEX_CONCURRENCY`. @@ -263,6 +299,22 @@ search: vector_weight: 0.7 keyword_weight: 0.2 symbol_weight: 0.1 + path_weight: 0.05 + structural_weight: 0.05 + dependency_weight: 0.08 + reference_weight: 0.08 + test_weight: 0.03 + recency_weight: 0.02 + max_chunks_per_file: 4 + max_chunks_per_symbol: 2 +retrieval: + seed_results: 8 + max_expansion_hops: 2 + max_context_chunks: 20 + max_context_tokens: 12000 + confidence_threshold: 0.15 + retrieval_required: true + allow_fallback_after_failed_retrieval: true security: allow_sensitive_files: false ignore: [] @@ -274,7 +326,7 @@ ignore: [] - "Model not found" — run `ollama pull ` for whatever `embedding.model` is configured. - Proxy authentication failures — set `CODE_INTEL_EMBEDDING_API_KEY` (and `CODE_INTEL_EMBEDDING_USER` if required) in both your shell and the MCP server `env` block. - MCP tools not appearing — reload the IDE's MCP servers list; check the IDE's MCP output/log panel for the spawned process's stderr. -- Switching embedding models requires `code-intel rebuild` (a different model produces vectors in a different space). +- Switching embedding models requires `code-intel rebuild` (a different model produces vectors in a different space). Rebuild is also the way to backfill `extra_metadata` (imports/exports/test/config flags) on an index created before those fields were populated. Incremental `code-intel index` fills metadata on files that change. ## Performance considerations @@ -283,7 +335,7 @@ ignore: [] - Tables with 256+ chunks get an IVF-PQ ANN index (L2); smaller indexes keep brute-force kNN. - A single-writer PID lock file prevents two `index`/`watch` processes from corrupting the same repo's index concurrently. The `mcp` command reads the index and, by default, incrementally writes when watched files change. -## Development +## Contributing ```bash npm run typecheck @@ -291,9 +343,14 @@ npm test npm run dev -- # run the CLI from source via tsx, no build step ``` -See [CONTRIBUTING.md](CONTRIBUTING.md) for the full contributor workflow. `tests/fixtures/test-repo` plus `tests/integration/acceptance.test.ts` and `tests/integration/mcp.test.ts` exercise the full spec acceptance scenario end-to-end against a real local Ollama model (skipped automatically if Ollama/the model isn't available). +Contributions are welcome. Start with [CONTRIBUTING.md](CONTRIBUTING.md), follow the [Code of Conduct](CODE_OF_CONDUCT.md), and report vulnerabilities through [SECURITY.md](SECURITY.md). `tests/fixtures/test-repo` plus the integration suite exercise the MCP path end to end against local Ollama when available. -## Known gaps +## Current limitations - Structural (Tree-sitter) chunking currently covers TypeScript/TSX, JavaScript, Python, Go, and Bash; other languages from the spec's list still get correctly tagged and indexed, just via generic text chunking rather than symbol-level chunks. - `find_references` is a textual occurrence scan, not full semantic reference resolution. +- Import expansion handles relative JavaScript/TypeScript paths, root TypeScript aliases, and Python modules; it is not a compiler or package manager. +- The file watcher runs with the MCP or `code-intel watch` process, not as a system daemon. +- Published retrieval metrics currently come from eight labeled tasks in this repository and do not guarantee equal quality on every language or monorepo. + +See the [architecture limitations](docs/ARCHITECTURE.md#current-limitations), [changelog](CHANGELOG.md), and [open issues](https://github.com/pranitmodi/code-intel/issues). diff --git a/SECURITY.md b/SECURITY.md new file mode 100644 index 0000000..a1f2077 --- /dev/null +++ b/SECURITY.md @@ -0,0 +1,82 @@ +# Security policy + +## Supported versions + +Security fixes are applied to the latest published npm version. Upgrade before reporting an issue that may already be fixed. + +## Reporting a vulnerability + +Do not open a public issue for a vulnerability, leaked credential, or private source exposure. + +Use [GitHub private vulnerability reporting](https://github.com/pranitmodi/code-intel/security/advisories/new). Include: + +- affected version or commit; +- operating system and Node.js version; +- embedding provider type, without credentials; +- reproduction steps; +- expected impact; +- whether private source or secrets may have left the machine. + +You should receive an acknowledgement within seven days. Please allow reasonable time for investigation and a coordinated fix before public disclosure. + +## Security model + +`code-intel` reads source files, sends text to an embedding provider, and stores source chunks and vectors in a local LanceDB index. + +### Default Ollama mode + +- Source chunks and queries are sent only to the configured Ollama host. +- The default host is local. +- Source chunks, vectors, metadata, usage events, and repository paths are stored under `~/.local-code-intelligence`. +- No telemetry is enabled by default. + +Treat a non-local Ollama host as a remote service. + +### OpenAI-compatible mode + +- Source chunks are sent to the configured embeddings endpoint during indexing. +- Search queries are sent to that endpoint during semantic retrieval. +- Embeddings and LanceDB remain local. +- API keys and optional user identities are read from environment variables. They are not read from YAML or written to the index. + +Review the endpoint's retention, training, and access policies before indexing private source. + +## Source filtering + +By default, discovery excludes: + +- `.env*` and common credential files; +- private keys and certificates; +- dependency, build, VCS, and cache directories; +- binary files; +- files containing likely secret values; +- files above the indexer safety cap. + +These controls reduce risk but are not a substitute for repository access control or secret scanning. False negatives are possible. Do not index source that the configured embedding provider is not allowed to receive. + +## Local files to protect + +Protect these as developer data: + +```text +~/.cursor/mcp.json +~/.local-code-intelligence/config.yaml +~/.local-code-intelligence/registry.json +~/.local-code-intelligence/usage.jsonl +~/.local-code-intelligence/repos/ +``` + +MCP configuration may contain embedding credentials. Repository indexes contain source text. Do not upload either directory in bug reports. + +## Operational guidance + +- Prefer Ollama for private, local-only operation. +- Use environment variables or a secret manager for provider credentials. +- Never commit `.code-intel/config.yaml` if it contains organization-specific endpoints or identities. +- Rotate credentials immediately if they appear in terminal output, logs, screenshots, issues, or chat transcripts. +- Run `code-intel clean --repo ` before decommissioning a workstation or transferring a repository index. +- Keep Node.js and dependencies current, and review `npm audit` before releases. + +## Out of scope + +Reports requiring physical access to an unlocked developer machine, social engineering, or denial of service through intentionally enormous trusted repositories may be closed unless they demonstrate a concrete boundary bypass. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..72f1a93 --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,189 @@ +# Architecture + +`code-intel` is a persistent, local-first context layer for coding agents. It separates the expensive write path—discovering, parsing, and embedding source—from the interactive read path that retrieves a small context package for each task. + +## System overview + +```mermaid +flowchart LR + subgraph workspace [Developer workspace] + source[Source files] + editor[IDE or coding agent] + end + + subgraph indexing [Incremental indexing] + discovery[Discovery and ignore rules] + parser[Structural parser] + chunker[Chunker and content hashes] + embeddings[Embedding provider] + end + + subgraph storage [Local storage] + registry[Repository registry] + vectors[LanceDB chunks and vectors] + state[Index and watch state] + end + + subgraph retrieval [Context retrieval] + mcp[MCP server] + intent[Task intent] + search[Hybrid search] + expansion[Relationship expansion] + selection[Ranking and token budget] + end + + source --> discovery --> parser --> chunker + chunker --> embeddings --> vectors + chunker --> state + discovery --> registry + + editor --> mcp --> intent --> search + vectors --> search --> expansion --> selection + selection --> mcp --> editor +``` + +The index lives under `~/.local-code-intelligence` by default, outside the repository. Source text is stored with each indexed chunk so retrieval does not need to scan the working tree. `get_file_context` remains authoritative for exact current file content. + +## Incremental indexing + +```mermaid +sequenceDiagram + participant FS as Working tree + participant Watch as File watcher + participant Indexer as Indexer + participant Embed as Embedding provider + participant DB as LanceDB + + FS->>Watch: create, update, or delete + Watch->>Watch: debounce event burst + alt Small event batch + Watch->>Indexer: changed paths only + else Startup or large event burst + Watch->>Indexer: full discovery with hash diff + end + Indexer->>DB: load existing file and chunk hashes + Indexer->>Indexer: parse and hash changed chunks + Indexer->>Embed: embed only new or changed chunks + Embed-->>Indexer: normalized vectors + Indexer->>DB: upsert chunks and remove stale rows + Indexer->>DB: update index, freshness, and watch state +``` + +Important properties: + +- A SHA-256 file hash decides whether a file changed. +- Normalized chunk hashes reuse embeddings when surrounding line numbers move. +- Content-identical renames update paths without re-embedding. +- Small watcher batches update individual paths; bursts above the threshold use full discovery. +- A single-writer PID lock prevents concurrent index corruption. +- A stable content sample complements file-count checks so edits can mark the index stale even when no files were added or removed. +- Watch failures and lock skips are persisted and exposed by `index_status`. + +The watcher belongs to the MCP or `code-intel watch` process. If that process is stopped, changes are caught up by the next MCP startup or explicit `code-intel index`. + +## Chunking and metadata + +Tree-sitter creates symbol-aware chunks for: + +- TypeScript and TSX +- JavaScript +- Python +- Go +- Bash + +Other detected languages use bounded text windows. Every chunk records file path, language, symbol identity when available, line range, content hashes, embedding, and lightweight metadata such as imports, exports, referenced identifiers, test status, and configuration status. + +Files larger than the safety cap, binary files, ignored paths, and likely secrets are skipped. Oversized symbols are split before embedding. + +## Retrieval pipeline + +```mermaid +flowchart TB + task[Task from agent] + analyze[Extract concepts, symbols, files, and intent] + candidates[Vector, keyword, symbol, and path candidates] + score[Inspectable weighted score] + seed[High-confidence seed selection] + relationships[Import, reference, test, and config expansion] + diversity[Per-file and per-symbol diversity] + budget[Hard token budget] + context[Context package] + fallback[Targeted filesystem fallback] + + task --> analyze --> candidates --> score --> seed + seed --> relationships --> diversity --> budget --> context + score -->|"stale or low confidence"| fallback +``` + +The final score combines semantic similarity, full-text relevance, exact symbol matching, path matching, structural information, relationship signals, test intent, and recency. Exact symbols and exact basenames receive floors so a definition is not buried under vaguely similar vector hits. + +`get_task_context` then: + +1. Generates and merges candidates. +2. Selects high-scoring seeds. +3. Expands relative imports, TypeScript path aliases, Python modules, references, tests, and configuration files. +4. Limits expansion-only files near the top. +5. Applies per-file and per-symbol caps. +6. Packs chunks under the requested token budget. +7. Returns files in score order with confidence and retrieval statistics. + +If the index is stale, results are empty, confidence is below the threshold, or a code-change query ranks documentation first, the Cursor integration temporarily permits targeted filesystem search. + +## MCP boundary + +The MCP server exposes: + +- `get_task_context` for broad coding tasks +- `search_symbol` for known identifiers +- `search_codebase` for conceptual exploration +- `find_references` for textual occurrence lookup +- `get_file_context` for authoritative source ranges +- `get_repo_context`, `list_indexed_repos`, and `index_status` for orientation + +The same index can serve Cursor, VS Code, Claude Code, Codex, or another MCP client. Each tool accepts an optional repository path, ID, or basename, allowing one parent workspace to address multiple indexed child repositories. + +## Provider and privacy boundary + +The embedding interface has two implementations: + +- **Ollama**: source chunks, queries, vectors, and database stay on the machine. +- **OpenAI-compatible**: source chunks and semantic queries are sent to the configured endpoint; vectors and LanceDB stay local. + +Credentials are read from environment variables only. They are never read from repository YAML or stored in the index. See [SECURITY.md](../SECURITY.md) for the security model and reporting process. + +## Storage layout + +```text +~/.local-code-intelligence/ +├── config.yaml +├── registry.json +├── usage.jsonl +└── repos/ + └── / + ├── db/ + ├── metadata/ + ├── state.json + ├── progress.json + ├── watch-status.json + └── logs/ +``` + +The repository ID is derived from its canonical absolute path. Index state records the embedding model and dimensions; changing models requires `code-intel rebuild`. + +## Current limitations + +- Structural symbols are unavailable for languages without a configured Tree-sitter grammar. +- `find_references` is textual occurrence matching, not compiler-grade semantic resolution. +- Relationship metadata is intentionally lightweight rather than a complete code graph. +- TypeScript aliases are loaded from root `tsconfig.json` or `jsconfig.json`; project references and framework-specific resolvers may need targeted filesystem fallback. +- Python import expansion is path-based and does not execute package resolution. +- Query quality benchmarks currently cover a small labeled TypeScript repository; they do not prove equal results on every language or monorepo. +- The watcher is process-scoped, not a system daemon. + +## Extension points + +- Add structural languages in `src/parser/languageRegistry.ts`. +- Add embedding providers behind `EmbeddingProvider`. +- Add ranking signals in `src/retrieval/score.ts`. +- Add relationship resolvers in `src/retrieval/taskContext.ts`. +- Add MCP tools in `src/mcp/server.ts`, while keeping high-level context retrieval the preferred interface. diff --git a/docs/BENCHMARKS.md b/docs/BENCHMARKS.md new file mode 100644 index 0000000..052e59a --- /dev/null +++ b/docs/BENCHMARKS.md @@ -0,0 +1,168 @@ +# Benchmarks + +This document explains what `code-intel` measures, how to reproduce it, and what the results do—and do not—prove. + +## Latest pre-release result + +Measured on September 21, 2026 against eight labeled tasks in the `code-intel` TypeScript repository: + +| Metric | Result | +| --- | ---: | +| Average per-task token reduction | 73.9% | +| Precision@5 | 0.33 | +| Precision@5 dataset ceiling | 0.40 | +| Relevant-file coverage@5 | 0.85 | +| Recall@10 | 0.96 | +| Mean reciprocal rank | 0.84 | +| NDCG | 0.72 | +| Operation p50 | 67 ms | +| Operation p95 | 708 ms | + +The run used `Qwen3-Embedding-8B` through an OpenAI-compatible embedding provider on a macOS development machine. It ran against the final pre-release source and documentation index on September 21, 2026. + +### Per-task audit + +| Task | Token reduction | Coverage@5 | Recall@10 | MRR | Missing relevant files | +| --- | ---: | ---: | ---: | ---: | --- | +| Known `searchCodebase` symbol | 85.8% | 1.00 | 1.00 | 1.00 | None | +| Conceptual hybrid search | 96.0% | 1.00 | 1.00 | 1.00 | None | +| Task-context feature | 85.3% | 0.50 | 1.00 | 1.00 | None; one relevant file ranked sixth | +| Stale-index bug | 55.5% | 1.00 | 1.00 | 0.20 | None; relevant file ranked fifth | +| Search-ranking refactor | 47.2% | 1.00 | 1.00 | 1.00 | None | +| Cross-cutting MCP instructions | 76.1% | 0.33 | 0.67 | 0.50 | `src/cursor/mcpInstructions.ts` | +| Tree-scan policy tests | 51.0% | 1.00 | 1.00 | 1.00 | None | +| Embedding configuration | 94.1% | 1.00 | 1.00 | 1.00 | None | + +## What is compared + +For each labeled task, the benchmark compares three paths: + +1. **Workspace scan baseline**: list files, run `rg` using a keyword derived from the task, and read matching source. +2. **Semantic search**: retrieve up to ten hybrid-search chunks. +3. **Task context**: run `get_task_context`, expand relationships, rank files, and apply an 8,000-token cap. + +```mermaid +flowchart LR + task[Same labeled task] + + subgraph baseline [Filesystem baseline] + list[List repository] + grep[Search with rg] + read[Read matching files] + list --> grep --> read + end + + subgraph indexed [Indexed retrieval] + query[Embed query] + search[Hybrid search] + expand[Relationship expansion] + pack[Token-budgeted context] + query --> search --> expand --> pack + end + + task --> list + task --> query + read --> compare[Compare relevance, tokens, and latency] + pack --> compare +``` + +Source-bearing payloads are counted using the same estimator used by the CLI: approximately four characters per token. This is deterministic and convenient, but it is not a provider-specific tokenizer. + +## Metric definitions + +- **Token reduction**: percentage decrease from the workspace-scan payload to the task-context payload. The aggregate is the mean of per-task percentages. +- **Precision@5**: relevant files among the first five returned files, divided by five. +- **Precision@5 ceiling**: maximum possible Precision@5 for the dataset given the number of labeled relevant files. Most tasks label only one to three files, so the aggregate ceiling is 0.40—not 1.00. +- **Relevant-file coverage@5**: fraction of labeled relevant files found in the first five. +- **Recall@10**: fraction of labeled relevant files found in the first ten. +- **MRR**: reciprocal rank of the first relevant file, averaged across tasks. +- **NDCG**: ranking quality using graded file relevance. +- **p50/p95**: retrieval latency percentiles across scan, semantic, and task-context operations recorded by the runner. + +## Interpreting token savings + +The measured reduction is **model input avoided**, not a direct reading of a Cursor, OpenAI, Anthropic, or cloud invoice. + +For an illustrative input price `R` dollars per million tokens: + +```text +estimated dollars avoided = avoided input tokens × R / 1,000,000 +``` + +If an unnecessary repository dump remains in a conversation for later turns, the same text may be charged or processed repeatedly. `code-intel savings --turns N` shows that compounding scenario, but it is an estimate and depends on the client, model, caching policy, and conversation behavior. + +Do not interpret a 73.9% payload reduction as a guaranteed 73.9% invoice reduction. + +## Reproducing the benchmark + +Prerequisites: + +- a fresh index for this repository; +- the configured embedding provider available; +- `rg` on `PATH`; +- no uncommitted source changes that are absent from the index. + +```bash +code-intel status +code-intel index +npm run benchmark:retrieval +``` + +Run one labeled task: + +```bash +npm run dev -- benchmark --format json --task known-search-codebase +``` + +Inspect one retrieval: + +```bash +npm run dev -- context "Where is searchCodebase implemented?" --explain +``` + +The JSON report includes relevant files, semantic files, retrieved files, missing relevant files, token counts, quality metrics, and latency for every task. + +## Dataset scope + +The current dataset includes: + +- known-symbol lookup; +- conceptual hybrid-search explanation; +- feature work around task context; +- stale-index diagnosis; +- ranking refactoring; +- cross-cutting MCP instruction changes; +- tree-scan policy tests; +- embedding configuration. + +The dataset is useful for regression detection inside this project. It is not yet representative of: + +- large polyglot monorepos; +- Java, Rust, C++, Swift, or mobile projects; +- compiler-grade references; +- real coding-agent task completion; +- every embedding model or hardware profile. + +## Honest failure cases + +The current system can still miss or mis-rank: + +- conceptual files that share no identifier with the task; +- overloaded short symbols such as `run` or `get`; +- source in languages that use generic text chunks; +- framework aliases not represented in root TypeScript configuration; +- dependencies resolved dynamically at runtime; +- edits made while no watcher is running. + +Low-confidence or stale retrieval should open targeted filesystem fallback rather than pretending the context package is complete. + +## Adding benchmark tasks + +Tasks live in `src/benchmark/datasets/codeIntelTasks.ts`. A useful task must have: + +- a realistic coding prompt; +- one or more defensible relevant files; +- graded relevance when some files are secondary; +- a category that improves coverage rather than duplicating an existing prompt. + +Benchmark changes should include the before/after JSON, explain traces for regressions, and a justification for any label changes. diff --git a/examples/cursor/local-code-intel.SKILL.md b/examples/cursor/local-code-intel.SKILL.md index 03cbd61..2756e97 100644 --- a/examples/cursor/local-code-intel.SKILL.md +++ b/examples/cursor/local-code-intel.SKILL.md @@ -17,9 +17,11 @@ Corpus embeddings already live in local LanceDB. Ollama (`nomic-embed-text`) emb ## Required order -1. `search_codebase` with `max_tokens` 800–1500 (optional `repo` for Savor children) -2. `search_symbol` / `find_references` when you have a name -3. `get_file_context` with a line range for the hit you will change -4. Grep/Glob/Read only after those miss, and only on a specific file or subdirectory +1. For a new or broad coding task, use `get_task_context` (optional `repo` for Savor children). +2. For a known symbol, use `search_symbol`. +3. For conceptual exploration, use `search_codebase` with `max_tokens` 800–1500. +4. For call-site analysis, use `find_references`. +5. Use `get_file_context` with a line range for the hit you will change. +6. Grep/Glob/Read only after those miss, after a low-confidence retrieval, or when the index is stale/unindexed — and only against a specific file or subdirectory. If `index_status` says unindexed, run `code-intel setup --repo ` via Shell, then search again. diff --git a/instructions.md b/instructions.md index fb1b29f..5b50db0 100644 --- a/instructions.md +++ b/instructions.md @@ -1,5 +1,7 @@ # Build a Local Persistent Code Intelligence Server +> Historical design specification. This document records the original implementation brief and may describe targets rather than current behavior. For maintained documentation, use the [README](README.md), [architecture](docs/ARCHITECTURE.md), and [benchmarks](docs/BENCHMARKS.md). + I want you to build a **local-first semantic code indexing and retrieval system** for software repositories. The goal is to create a reusable local service that indexes an entire codebase into a **local vector database**, continuously keeps that index synchronized with the repository as files change, and exposes semantic code search to AI coding agents through **MCP**. diff --git a/package-lock.json b/package-lock.json index da8cab8..430e600 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,15 +1,15 @@ { "name": "@pranitmodi/code-intel", - "version": "0.1.1", + "version": "0.2.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@pranitmodi/code-intel", - "version": "0.1.1", + "version": "0.2.0", "license": "MIT", "dependencies": { - "@lancedb/lancedb": "^0.38.0", + "@lancedb/lancedb": "^0.39.0", "@modelcontextprotocol/server": "^2.0.0", "@parcel/watcher": "^2.6.0", "@tree-sitter-grammars/tree-sitter-yaml": "^0.7.1", @@ -41,7 +41,7 @@ "vitest": "^4.1.11" }, "engines": { - "node": ">=20" + "node": ">=20.9" } }, "node_modules/@emnapi/runtime": { @@ -497,32 +497,50 @@ } }, "node_modules/@huggingface/jinja": { - "version": "0.3.4", - "resolved": "https://registry.npmjs.org/@huggingface/jinja/-/jinja-0.3.4.tgz", - "integrity": "sha512-kFFQWJiWwvxezKQnvH1X7GjsECcMljFx+UZK9hx6P26aVHwwidJVTB0ptLfRVZQvVkOGHoMmTGvo4nT0X9hHOA==", + "version": "0.5.10", + "resolved": "https://registry.npmjs.org/@huggingface/jinja/-/jinja-0.5.10.tgz", + "integrity": "sha512-SgS1D1bglQ94ceD4ZCL6eayUDy9uV1xuyk61OjgSxU5GJh7upqZCKII0JoTYMc6wWsg3JUgaPRX0ElQzXPk3Cw==", "license": "MIT", "optional": true, "engines": { "node": ">=18" } }, + "node_modules/@huggingface/tokenizers": { + "version": "0.2.0", + "resolved": "https://registry.npmjs.org/@huggingface/tokenizers/-/tokenizers-0.2.0.tgz", + "integrity": "sha512-LidMHe1FpcSYH4vcSjXooda34pC0M8m1gmDOv7SQi6Iv+ib1zX7J3sHBCC9/3FPCkn2k1b55FneVqMRxrr99pg==", + "license": "Apache-2.0", + "optional": true + }, "node_modules/@huggingface/transformers": { - "version": "3.0.2", - "resolved": "https://registry.npmjs.org/@huggingface/transformers/-/transformers-3.0.2.tgz", - "integrity": "sha512-lTyS81eQazMea5UCehDGFMfdcNRZyei7XQLH5X6j4AhA/18Ka0+5qPgMxUxuZLU4xkv60aY2KNz9Yzthv6WVJg==", + "version": "4.3.0", + "resolved": "https://registry.npmjs.org/@huggingface/transformers/-/transformers-4.3.0.tgz", + "integrity": "sha512-fL1A/WUZwouPrOlYxU5dzIwD2T5J781JiB2jDR8bFe5DwCj0Gfudq+NEXCMno49kQgajHA7xQkrRLJlqG1veEA==", "license": "Apache-2.0", "optional": true, "dependencies": { - "@huggingface/jinja": "^0.3.0", - "onnxruntime-node": "1.19.2", - "onnxruntime-web": "1.21.0-dev.20241024-d9ca84ef96", - "sharp": "^0.33.5" + "@huggingface/jinja": "^0.5.10", + "@huggingface/tokenizers": "^0.2.0", + "onnxruntime-node": "1.30.0", + "onnxruntime-web": "1.31.0-dev.20260914-8d85527a0", + "sharp": "^0.35.4" + } + }, + "node_modules/@img/colour": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/@img/colour/-/colour-1.1.0.tgz", + "integrity": "sha512-Td76q7j57o/tLVdgS746cYARfSyxk8iEfRxewL9h4OMzYhbW4TAcppl0mT4eyqXddh6L/jwoM75mo7ixa/pCeQ==", + "license": "MIT", + "optional": true, + "engines": { + "node": ">=18" } }, "node_modules/@img/sharp-darwin-arm64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-darwin-arm64/-/sharp-darwin-arm64-0.33.5.tgz", - "integrity": "sha512-UT4p+iz/2H4twwAoLCqfA9UH5pI6DggwKEGuaPy7nCVQ8ZsiY5PIcrRvD1DzuY3qYL07NtIQcWnBSY/heikIFQ==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-darwin-arm64/-/sharp-darwin-arm64-0.35.4.tgz", + "integrity": "sha512-Uhfl4V4lhP2nbUVF9+hyH1+luj86f1gUFeo8ALYxFoULoU+G87D43BfeMP8XHsk9boxAnCY/bf2EHwhA7MuGsA==", "cpu": [ "arm64" ], @@ -532,19 +550,19 @@ "darwin" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-darwin-arm64": "1.0.4" + "@img/sharp-libvips-darwin-arm64": "1.3.3" } }, "node_modules/@img/sharp-darwin-x64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-darwin-x64/-/sharp-darwin-x64-0.33.5.tgz", - "integrity": "sha512-fyHac4jIc1ANYGRDxtiqelIbdWkIuQaI84Mv45KvGRRxSAa7o7d1ZKAOBaYbnepLC1WqxfpimdeWfvqqSGwR2Q==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-darwin-x64/-/sharp-darwin-x64-0.35.4.tgz", + "integrity": "sha512-hWniXY3bG5qKpkKrAwPe4y+VTPmf086YQAnkxWh7uA1YrlRouWGa0M0Mxj3ZjnXFkv7/TD1bTy9lGUK26vRvWw==", "cpu": [ "x64" ], @@ -554,19 +572,38 @@ "darwin" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-darwin-x64": "1.0.4" + "@img/sharp-libvips-darwin-x64": "1.3.3" + } + }, + "node_modules/@img/sharp-freebsd-wasm32": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-freebsd-wasm32/-/sharp-freebsd-wasm32-0.35.4.tgz", + "integrity": "sha512-lIsKw/BU+kjB4eZjxrYrZmwOJYi3Ajrv66iAlBmUPyKc3HpnloevB1g3wxGD9P/5BbQ1brBGl65VRRrCvQDEqA==", + "license": "Apache-2.0", + "optional": true, + "os": [ + "freebsd" + ], + "dependencies": { + "@img/sharp-wasm32": "0.35.4" + }, + "engines": { + "node": ">=20.9.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" } }, "node_modules/@img/sharp-libvips-darwin-arm64": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-darwin-arm64/-/sharp-libvips-darwin-arm64-1.0.4.tgz", - "integrity": "sha512-XblONe153h0O2zuFfTAbQYAX2JhYmDHeWikp1LM9Hul9gVPjFY427k6dFEcOL72O01QxQsWi761svJ/ev9xEDg==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-darwin-arm64/-/sharp-libvips-darwin-arm64-1.3.3.tgz", + "integrity": "sha512-suTBPTDGrI9WodccaDdwZItTSaBYASlBk1NSfElSHrUfzu3szG6lvIF58+WiFvnfzuK8ZBFS5zE00PxqxnRiPg==", "cpu": [ "arm64" ], @@ -580,9 +617,9 @@ } }, "node_modules/@img/sharp-libvips-darwin-x64": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-darwin-x64/-/sharp-libvips-darwin-x64-1.0.4.tgz", - "integrity": "sha512-xnGR8YuZYfJGmWPvmlunFaWJsb9T/AO2ykoP3Fz/0X5XV2aoYBPkX6xqCQvUTKKiLddarLaxpzNe+b1hjeWHAQ==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-darwin-x64/-/sharp-libvips-darwin-x64-1.3.3.tgz", + "integrity": "sha512-FVJZ5mITMobmXIz/hPDTw0EintTW5H3WfrxwLqEqjiIihlu+hVRyGrFQ60xl0Lxn7Bt3zdpevPaQi0HEzqz9fw==", "cpu": [ "x64" ], @@ -596,12 +633,15 @@ } }, "node_modules/@img/sharp-libvips-linux-arm": { - "version": "1.0.5", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-arm/-/sharp-libvips-linux-arm-1.0.5.tgz", - "integrity": "sha512-gvcC4ACAOPRNATg/ov8/MnbxFDJqf/pDePbBnuBDcjsI8PssmjoKMAz4LtLaVi+OnSb5FK/yIOamqDwGmXW32g==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-arm/-/sharp-libvips-linux-arm-1.3.3.tgz", + "integrity": "sha512-3rbU4vqXXc3hY/OiXdl52xZvT0F1yEngWfvqudtPJg/KkyiaQw2DRsFrNzpmLvfavbwOq3qXn36GP8obHRULQA==", "cpu": [ "arm" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -612,12 +652,53 @@ } }, "node_modules/@img/sharp-libvips-linux-arm64": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-arm64/-/sharp-libvips-linux-arm64-1.0.4.tgz", - "integrity": "sha512-9B+taZ8DlyyqzZQnoeIvDVR/2F4EbMepXMc/NdVbkzsJbzkUjhXv/70GQJ7tdLA4YJgNP25zukcxpX2/SueNrA==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-arm64/-/sharp-libvips-linux-arm64-1.3.3.tgz", + "integrity": "sha512-0DaL0A6Xu6sQSQFwe4iVCrKWU2cCTItnRsYsCdxAMm9NF6twAA9BKnoqy4hqz4+azQ0JHuA26qiUKsf1XJ/v5A==", "cpu": [ "arm64" ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-ppc64": { + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-ppc64/-/sharp-libvips-linux-ppc64-1.3.3.tgz", + "integrity": "sha512-cdn1OvUBwsXhbC0zSzJnNzf5MZ/mTrobawDvNXBTxe8VtqKAm0sRuEY2Evzovb/w9JMk4TvRxqt1mekSuJz64w==", + "cpu": [ + "ppc64" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-riscv64": { + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-riscv64/-/sharp-libvips-linux-riscv64-1.3.3.tgz", + "integrity": "sha512-HjPVx7yKz+0lqdhDlTw1tt90wamBoxhiXpvl1XZpJLiHH4RCJ5yDTqH+VlYPv2fwFs89JFw4c1IexYOcQUi4IQ==", + "cpu": [ + "riscv64" + ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -628,12 +709,15 @@ } }, "node_modules/@img/sharp-libvips-linux-s390x": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-s390x/-/sharp-libvips-linux-s390x-1.0.4.tgz", - "integrity": "sha512-u7Wz6ntiSSgGSGcjZ55im6uvTrOxSIS8/dgoVMoiGE9I6JAfU50yH5BoDlYA1tcuGS7g/QNtetJnxA6QEsCVTA==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-s390x/-/sharp-libvips-linux-s390x-1.3.3.tgz", + "integrity": "sha512-neWLh+3yCNThxnfy3c4BbVBeGgt9aftno+XbT56iK28RgeDs3UOFWviLWlUu0bArYVYJaFDK+RRohbicUNCm8Q==", "cpu": [ "s390x" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -644,12 +728,15 @@ } }, "node_modules/@img/sharp-libvips-linux-x64": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-x64/-/sharp-libvips-linux-x64-1.0.4.tgz", - "integrity": "sha512-MmWmQ3iPFZr0Iev+BAgVMb3ZyC4KeFc3jFxnNbEPas60e1cIfevbtuyf9nDGIzOaW9PdnDciJm+wFFaTlj5xYw==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-x64/-/sharp-libvips-linux-x64-1.3.3.tgz", + "integrity": "sha512-4vKmvAst9nrowcqquKFAyZJUDolUaIp8uRiN0mWFguJ1IplC9/pitXtlnnlU4aa/eJw3J7i67V+pwUL+wZGdsA==", "cpu": [ "x64" ], + "libc": [ + "glibc" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -660,12 +747,15 @@ } }, "node_modules/@img/sharp-libvips-linuxmusl-arm64": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linuxmusl-arm64/-/sharp-libvips-linuxmusl-arm64-1.0.4.tgz", - "integrity": "sha512-9Ti+BbTYDcsbp4wfYib8Ctm1ilkugkA/uscUn6UXK1ldpC1JjiXbLfFZtRlBhjPZ5o1NCLiDbg8fhUPKStHoTA==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linuxmusl-arm64/-/sharp-libvips-linuxmusl-arm64-1.3.3.tgz", + "integrity": "sha512-Y9kQaLMuNoB0bPYOOdcZMaseNrFpPodIWWMrx+CZyydf2xn68j9WYc6sWWRrDwNkzCQjKYfc68L7jKjGlHMibw==", "cpu": [ "arm64" ], + "libc": [ + "musl" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -676,12 +766,15 @@ } }, "node_modules/@img/sharp-libvips-linuxmusl-x64": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linuxmusl-x64/-/sharp-libvips-linuxmusl-x64-1.0.4.tgz", - "integrity": "sha512-viYN1KX9m+/hGkJtvYYp+CCLgnJXwiQB39damAO7WMdKWlIhmYTfHjwSbQeUK/20vY154mwezd9HflVFM1wVSw==", + "version": "1.3.3", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linuxmusl-x64/-/sharp-libvips-linuxmusl-x64-1.3.3.tgz", + "integrity": "sha512-fj8Mv0HHfD1Rr+4I68+3agJynxDWtBFgicTbSOb9Bke6pIwzGcJ+RX/yHjmiEGFMCavY/dxvem7MyNaJF+wDiw==", "cpu": [ "x64" ], + "libc": [ + "musl" + ], "license": "LGPL-3.0-or-later", "optional": true, "os": [ @@ -692,162 +785,246 @@ } }, "node_modules/@img/sharp-linux-arm": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-linux-arm/-/sharp-linux-arm-0.33.5.tgz", - "integrity": "sha512-JTS1eldqZbJxjvKaAkxhZmBqPRGmxgu+qFKSInv8moZ2AmT5Yib3EQ1c6gp493HvrvV8QgdOXdyaIBrhvFhBMQ==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-arm/-/sharp-linux-arm-0.35.4.tgz", + "integrity": "sha512-7OAS8gI0EReKGVN2HssHlM6umJgxF5VI3xN0p9FA91p/YO+ou5hiNghLdZ5BEHztwaaK5+bLKRf8x/o2L2nk9A==", "cpu": [ "arm" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ "linux" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-linux-arm": "1.0.5" + "@img/sharp-libvips-linux-arm": "1.3.3" } }, "node_modules/@img/sharp-linux-arm64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-linux-arm64/-/sharp-linux-arm64-0.33.5.tgz", - "integrity": "sha512-JMVv+AMRyGOHtO1RFBiJy/MBsgz0x4AWrT6QoEVVTyh1E39TrCUpTRI7mx9VksGX4awWASxqCYLCV4wBZHAYxA==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-arm64/-/sharp-linux-arm64-0.35.4.tgz", + "integrity": "sha512-De4jpEnAU8Hd5oT0j1G3uL4ZvTuipVMn7YC6vPaJhy6/7EwEae0SVAoBrUMYQbkLGDm85taVWwuPc1a44LTzCQ==", "cpu": [ "arm64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ "linux" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-linux-arm64": "1.0.4" + "@img/sharp-libvips-linux-arm64": "1.3.3" + } + }, + "node_modules/@img/sharp-linux-ppc64": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-ppc64/-/sharp-linux-ppc64-0.35.4.tgz", + "integrity": "sha512-2oYZJeIl4kCcMGk4ouZVjnkCtFrpQFlNEtJ6GbxzhHQchwH0NH/qEb9ykmOl29dqwMq+JhFdZn+1ak2FKhI9fQ==", + "cpu": [ + "ppc64" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=20.9.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-ppc64": "1.3.3" + } + }, + "node_modules/@img/sharp-linux-riscv64": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-riscv64/-/sharp-linux-riscv64-0.35.4.tgz", + "integrity": "sha512-cPbNChoRURAWdebDIHSenxRpgEdy7JkPydSnUxRm9VvKD7m0/xVaR/8Fzlu81pk5nHEvHH87UZUA7cTtwnbJSA==", + "cpu": [ + "riscv64" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">=20.9.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-riscv64": "1.3.3" } }, "node_modules/@img/sharp-linux-s390x": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-linux-s390x/-/sharp-linux-s390x-0.33.5.tgz", - "integrity": "sha512-y/5PCd+mP4CA/sPDKl2961b+C9d+vPAveS33s6Z3zfASk2j5upL6fXVPZi7ztePZ5CuH+1kW8JtvxgbuXHRa4Q==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-s390x/-/sharp-linux-s390x-0.35.4.tgz", + "integrity": "sha512-RY0JFY8Fd6RonCBtHz+DvadaPkXDSI1AUn6yWL9TipqkZ1vY8w8evqdgyDFnkm4/K1ve1TvZiaePP5oSd4+WVQ==", "cpu": [ "s390x" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ "linux" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-linux-s390x": "1.0.4" + "@img/sharp-libvips-linux-s390x": "1.3.3" } }, "node_modules/@img/sharp-linux-x64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-linux-x64/-/sharp-linux-x64-0.33.5.tgz", - "integrity": "sha512-opC+Ok5pRNAzuvq1AG0ar+1owsu842/Ab+4qvU879ippJBHvyY5n2mxF1izXqkPYlGuP/M556uh53jRLJmzTWA==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-x64/-/sharp-linux-x64-0.35.4.tgz", + "integrity": "sha512-9qvvEAuk8k89TfWUoX2htWjbAMX8p+NxCppjpcg5k6xMsjhBQPTsoIh36h9Qde4WRuGpJeYnOjdosDn/cnv+OA==", "cpu": [ "x64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ "linux" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-linux-x64": "1.0.4" + "@img/sharp-libvips-linux-x64": "1.3.3" } }, "node_modules/@img/sharp-linuxmusl-arm64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-linuxmusl-arm64/-/sharp-linuxmusl-arm64-0.33.5.tgz", - "integrity": "sha512-XrHMZwGQGvJg2V/oRSUfSAfjfPxO+4DkiRh6p2AFjLQztWUuY/o8Mq0eMQVIY7HJ1CDQUJlxGGZRw1a5bqmd1g==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linuxmusl-arm64/-/sharp-linuxmusl-arm64-0.35.4.tgz", + "integrity": "sha512-KB5jxpfWQTr0nc3xdHtWChdbifHrBGsd2SM62Eyxrl8afikm+f5qGBU75SJIZBT/S1MC8XyacdlXBMSWq6OURA==", "cpu": [ "arm64" ], + "libc": [ + "musl" + ], "license": "Apache-2.0", "optional": true, "os": [ "linux" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-linuxmusl-arm64": "1.0.4" + "@img/sharp-libvips-linuxmusl-arm64": "1.3.3" } }, "node_modules/@img/sharp-linuxmusl-x64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-linuxmusl-x64/-/sharp-linuxmusl-x64-0.33.5.tgz", - "integrity": "sha512-WT+d/cgqKkkKySYmqoZ8y3pxx7lx9vVejxW/W4DOFMYVSkErR+w7mf2u8m/y4+xHe7yY9DAXQMWQhpnMuFfScw==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-linuxmusl-x64/-/sharp-linuxmusl-x64-0.35.4.tgz", + "integrity": "sha512-f+eZJZIQNEEd26RPSW+76chwOf1XtA2Y/O+5ocVyLliHkeih3e+jhLVBdNTd2rS3IbNXK8+ug93Vf5ZXtF5Lxg==", "cpu": [ "x64" ], + "libc": [ + "musl" + ], "license": "Apache-2.0", "optional": true, "os": [ "linux" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-libvips-linuxmusl-x64": "1.0.4" + "@img/sharp-libvips-linuxmusl-x64": "1.3.3" } }, "node_modules/@img/sharp-wasm32": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-wasm32/-/sharp-wasm32-0.33.5.tgz", - "integrity": "sha512-ykUW4LVGaMcU9lu9thv85CbRMAwfeadCJHRsg2GmeRa/cJxsVY9Rbd57JcMxBkKHag5U/x7TSBpScF4U8ElVzg==", + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-wasm32/-/sharp-wasm32-0.35.4.tgz", + "integrity": "sha512-zQnl4Kwp7Q6NHsENtU2T/00Zi+w3AQNwz3+UaTyVBy2FpXrzXzGjndpK61onhZjRtRpQXxCTeqw19bVyXOh7jA==", + "license": "Apache-2.0 AND LGPL-3.0-or-later AND MIT", + "optional": true, + "dependencies": { + "@emnapi/runtime": "^1.11.3" + }, + "engines": { + "node": ">=20.9.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-webcontainers-wasm32": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-webcontainers-wasm32/-/sharp-webcontainers-wasm32-0.35.4.tgz", + "integrity": "sha512-ESfNkywmCfPNyaZjxooddJQiQ+l/nTpGEOGthxiLnIHXC/CmcBixnfwUleX9mCz9ovrUUvKMap/pm8RYbzfwaA==", "cpu": [ "wasm32" ], - "license": "Apache-2.0 AND LGPL-3.0-or-later AND MIT", + "license": "Apache-2.0", "optional": true, "dependencies": { - "@emnapi/runtime": "^1.2.0" + "@img/sharp-wasm32": "0.35.4" }, "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" } }, - "node_modules/@img/sharp-win32-ia32": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-win32-ia32/-/sharp-win32-ia32-0.33.5.tgz", - "integrity": "sha512-T36PblLaTwuVJ/zw/LaH0PdZkRz5rd3SmMHX8GSmR7vtNSP5Z6bQkExdSK7xGWyxLw4sUknBuugTelgw2faBbQ==", + "node_modules/@img/sharp-win32-arm64": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-win32-arm64/-/sharp-win32-arm64-0.35.4.tgz", + "integrity": "sha512-iNdlBX9gLVvqe2I3uIJSIKTq6wckP/DYxZtcqxm09x5Gi24DnFBmPAWZmr60ZyYMG0xlzo6goG3670ar+RXvRw==", "cpu": [ - "ia32" + "arm64" ], "license": "Apache-2.0 AND LGPL-3.0-or-later", "optional": true, @@ -855,18 +1032,18 @@ "win32" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" } }, - "node_modules/@img/sharp-win32-x64": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/@img/sharp-win32-x64/-/sharp-win32-x64-0.33.5.tgz", - "integrity": "sha512-MpY/o8/8kj+EcnxwvrP4aTJSWw/aZ7JIGR4aBeZkZw5B7/Jn+tY9/VNwtcoGmdT7GfggGIU4kygOMSbYnOrAbg==", + "node_modules/@img/sharp-win32-ia32": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-win32-ia32/-/sharp-win32-ia32-0.35.4.tgz", + "integrity": "sha512-kqRsbaa5CS6KHlpxnN7WhE6vAAugXyZButpRdvDWetlv6Qv4N9WTcrWzF7tXfB9T7MsoadqdI8hmwLq6UlLvtw==", "cpu": [ - "x64" + "ia32" ], "license": "Apache-2.0 AND LGPL-3.0-or-later", "optional": true, @@ -874,23 +1051,29 @@ "win32" ], "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": "^20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" } }, - "node_modules/@isaacs/fs-minipass": { - "version": "4.0.1", - "resolved": "https://registry.npmjs.org/@isaacs/fs-minipass/-/fs-minipass-4.0.1.tgz", - "integrity": "sha512-wgm9Ehl2jpeqP3zw/7mo3kRHFp5MEDhqAdwy1fTGkHAwnkGOVsgpvQhL8B5n1qlb01jV3n/bI0ZfZp5lWA1k4w==", - "license": "ISC", + "node_modules/@img/sharp-win32-x64": { + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/@img/sharp-win32-x64/-/sharp-win32-x64-0.35.4.tgz", + "integrity": "sha512-XtmnYhBcrORsJ4XJngyzr/EWP0hRZLAZRFaApdKuviyqF78+ylxh2y06ZmtULAMOnObJ3ucpN0AcwSWnMowTRg==", + "cpu": [ + "x64" + ], + "license": "Apache-2.0 AND LGPL-3.0-or-later", "optional": true, - "dependencies": { - "minipass": "^7.0.4" - }, + "os": [ + "win32" + ], "engines": { - "node": ">=18.0.0" + "node": ">=20.9.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" } }, "node_modules/@jridgewell/sourcemap-codec": { @@ -901,9 +1084,9 @@ "license": "MIT" }, "node_modules/@lancedb/lancedb": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb/-/lancedb-0.38.0.tgz", - "integrity": "sha512-10+Cy0P8GyGg7d6l6AmuruEfFt6sxdCZNFCmhcJyUofvopxCb1yIiFNXsfzjO8Iyh2hijTEuyKTzIi4LGlH6Kw==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb/-/lancedb-0.39.0.tgz", + "integrity": "sha512-mcHi7/G2knZVj7OUya1vcMAjbHEqcait31cXINdUzQrru+ulo9Rks35T5XQpHQXI/P+6o10azATBtNa0u2tDCA==", "cpu": [ "x64", "arm64" @@ -923,13 +1106,13 @@ }, "optionalDependencies": { "@huggingface/transformers": "3.0.2", - "@lancedb/lancedb-darwin-arm64": "0.38.0", - "@lancedb/lancedb-linux-arm64-gnu": "0.38.0", - "@lancedb/lancedb-linux-arm64-musl": "0.38.0", - "@lancedb/lancedb-linux-x64-gnu": "0.38.0", - "@lancedb/lancedb-linux-x64-musl": "0.38.0", - "@lancedb/lancedb-win32-arm64-msvc": "0.38.0", - "@lancedb/lancedb-win32-x64-msvc": "0.38.0", + "@lancedb/lancedb-darwin-arm64": "0.39.0", + "@lancedb/lancedb-linux-arm64-gnu": "0.39.0", + "@lancedb/lancedb-linux-arm64-musl": "0.39.0", + "@lancedb/lancedb-linux-x64-gnu": "0.39.0", + "@lancedb/lancedb-linux-x64-musl": "0.39.0", + "@lancedb/lancedb-win32-arm64-msvc": "0.39.0", + "@lancedb/lancedb-win32-x64-msvc": "0.39.0", "openai": "4.29.2" }, "peerDependencies": { @@ -943,9 +1126,9 @@ } }, "node_modules/@lancedb/lancedb-darwin-arm64": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-darwin-arm64/-/lancedb-darwin-arm64-0.38.0.tgz", - "integrity": "sha512-AOSqmezmLGjv7SjZfDdzYznebdihApjPklhIJJ8cFfMsqYqc1pJtRAM8zWKotbiwLOg31ttIcGIoWbv/0dUJSQ==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-darwin-arm64/-/lancedb-darwin-arm64-0.39.0.tgz", + "integrity": "sha512-8MtalZRx8Ks/vED1jydXLuSlgJKmAtY2Cn9nuZo6RpgNtDxbB6dhdoF3gWhhbsRUOiETzHkmPukR53yfk8A/LQ==", "cpu": [ "arm64" ], @@ -959,12 +1142,15 @@ } }, "node_modules/@lancedb/lancedb-linux-arm64-gnu": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-arm64-gnu/-/lancedb-linux-arm64-gnu-0.38.0.tgz", - "integrity": "sha512-Lgw64fSC5+jRpEpYtPnOOOIq7ODc+uXnAbdzAoofqmzjizfT2s8MoFVq1I/oXHxgKbEQLyeM3IrRqp7ur2+ivA==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-arm64-gnu/-/lancedb-linux-arm64-gnu-0.39.0.tgz", + "integrity": "sha512-WMoIED9D4XScRVgB0Da1DTW0PHP0apzUUYbcA0XYJgZ/kzHhFBigCtiOksMIVTT31SRXLsyX00AYRys3VyrEmQ==", "cpu": [ "arm64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -975,12 +1161,15 @@ } }, "node_modules/@lancedb/lancedb-linux-arm64-musl": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-arm64-musl/-/lancedb-linux-arm64-musl-0.38.0.tgz", - "integrity": "sha512-g1TWe8QxCTU2YwSTs8LqJVnK+kmws6cSxq04xo8ap6RefybUq67ik9DUNR6H0oqkPuM5HBfPNoYvZvCIWeuFAg==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-arm64-musl/-/lancedb-linux-arm64-musl-0.39.0.tgz", + "integrity": "sha512-46K+pDQ88prIpYjx+mRpxJoy1kaeZJneJAOGSTlEneB7pA155wNggr7VwXm+cQeyjp2uuS0BzgWAiN4RGD9Bgg==", "cpu": [ "arm64" ], + "libc": [ + "musl" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -991,12 +1180,15 @@ } }, "node_modules/@lancedb/lancedb-linux-x64-gnu": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-x64-gnu/-/lancedb-linux-x64-gnu-0.38.0.tgz", - "integrity": "sha512-RdlMdP4JjMvK1Ro7esrhtLsSwdO0jrZnD+XNEZswQz00apvPF/F3+nLcY94IVH44Tho8jBP9qgVc4y+GimIHKQ==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-x64-gnu/-/lancedb-linux-x64-gnu-0.39.0.tgz", + "integrity": "sha512-CJz2Dm2rPSFQSGWIllWt4PYgz2JsdAlz1fNzm8AFkxIhQwO30/kxtbRqzs4rqptQOtGUxlaIIeSipsANov1cKA==", "cpu": [ "x64" ], + "libc": [ + "glibc" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1007,12 +1199,15 @@ } }, "node_modules/@lancedb/lancedb-linux-x64-musl": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-x64-musl/-/lancedb-linux-x64-musl-0.38.0.tgz", - "integrity": "sha512-5lbuUJijTuypzRrNHNGEa7iYe2T22sgTZBRmf3S/zis0sFiT3pxPvEEOu8nhF3GstPbJElBweFPW6Yzfpa4o3Q==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-linux-x64-musl/-/lancedb-linux-x64-musl-0.39.0.tgz", + "integrity": "sha512-S6oCxSEnwD6L7lmcoxnLSYJMVUMx57dPYKMwv1MtY/Bv4+VIgmCg8u0HTtXzJrrHnowtuhjtTqptu/7O41ZXsw==", "cpu": [ "x64" ], + "libc": [ + "musl" + ], "license": "Apache-2.0", "optional": true, "os": [ @@ -1023,9 +1218,9 @@ } }, "node_modules/@lancedb/lancedb-win32-arm64-msvc": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-win32-arm64-msvc/-/lancedb-win32-arm64-msvc-0.38.0.tgz", - "integrity": "sha512-ObG2yXxZxo3mEJp7whFh1A4rspJW8DxMGRnLvjPs9EvCoyqjOXaIgrsqL+FnsZ4g3HAGYYOmUhusTYcD0Nu54g==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-win32-arm64-msvc/-/lancedb-win32-arm64-msvc-0.39.0.tgz", + "integrity": "sha512-ZGVJEPVXnl1dWDs+LHR4AW0x9XmCBjch8eHz/iJL1q+niAbaDaBx66QAt4DMBp3zxMxCS8oDB6nwUm6/UEpFsg==", "cpu": [ "arm64" ], @@ -1039,9 +1234,9 @@ } }, "node_modules/@lancedb/lancedb-win32-x64-msvc": { - "version": "0.38.0", - "resolved": "https://registry.npmjs.org/@lancedb/lancedb-win32-x64-msvc/-/lancedb-win32-x64-msvc-0.38.0.tgz", - "integrity": "sha512-DrzY0/XSyf7z8+xiSXQWhwREaNXLsxvRqbrdWLr4cisOoZe80PoEWQiwqKNIgHRyHY9alXPoyXj9BV60pkvanw==", + "version": "0.39.0", + "resolved": "https://registry.npmjs.org/@lancedb/lancedb-win32-x64-msvc/-/lancedb-win32-x64-msvc-0.39.0.tgz", + "integrity": "sha512-HE8aD11DSElESGYBaTuy9Pvcapg1vRIK2qAsZVGPGN1AVUOI3j304UdVJiHQqY+cSwPs0DvBv9rkbsewY9XAOQ==", "cpu": [ "x64" ], @@ -1954,6 +2149,16 @@ "node": ">=6.5" } }, + "node_modules/adm-zip": { + "version": "0.6.1", + "resolved": "https://registry.npmjs.org/adm-zip/-/adm-zip-0.6.1.tgz", + "integrity": "sha512-Xwrja8nx9e5o2N1my4DsKCeKpdrnACyr1wtbPxBDgGzKzKyE9kRtBFA8mWldI+RVlD7CBZNWY/wQ2+ydwOR6kQ==", + "license": "MIT", + "optional": true, + "engines": { + "node": ">=14.0" + } + }, "node_modules/agentkeepalive": { "version": "4.6.0", "resolved": "https://registry.npmjs.org/agentkeepalive/-/agentkeepalive-4.6.0.tgz", @@ -2114,30 +2319,6 @@ "node": "*" } }, - "node_modules/chownr": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz", - "integrity": "sha512-+IxzY9BZOQd/XuYPRmrvEVjF/nqj5kgT4kEq7VofrDoM1MxoRjEWkrCC3EtLi59TVawxTAn+orJwFQcrqEN1+g==", - "license": "BlueOak-1.0.0", - "optional": true, - "engines": { - "node": ">=18" - } - }, - "node_modules/color": { - "version": "4.2.3", - "resolved": "https://registry.npmjs.org/color/-/color-4.2.3.tgz", - "integrity": "sha512-1rXeuUUiGGrykh+CeBdu5Ie7OJwinCgQY0bc7GCRxy5xVHy+moaqkpL/jqQq0MtQOeYcrqEz4abc5f0KtU7W4A==", - "license": "MIT", - "optional": true, - "dependencies": { - "color-convert": "^2.0.1", - "color-string": "^1.9.0" - }, - "engines": { - "node": ">=12.5.0" - } - }, "node_modules/color-convert": { "version": "2.0.1", "resolved": "https://registry.npmjs.org/color-convert/-/color-convert-2.0.1.tgz", @@ -2156,17 +2337,6 @@ "integrity": "sha512-dOy+3AuW3a2wNbZHIuMZpTcgjGuLU/uBL/ubcZF9OXbDo8ff4O8yVp5Bf0efS8uEoYo5q4Fx7dY9OgQGXgAsQA==", "license": "MIT" }, - "node_modules/color-string": { - "version": "1.9.1", - "resolved": "https://registry.npmjs.org/color-string/-/color-string-1.9.1.tgz", - "integrity": "sha512-shrVawQFojnZv6xM40anx4CkoDP+fZsw/ZerEMsW/pyzsRbElpsL/DBVW7q3ExxwusdNXI3lXpuhEZkzs8p5Eg==", - "license": "MIT", - "optional": true, - "dependencies": { - "color-name": "^1.0.0", - "simple-swizzle": "^0.2.2" - } - }, "node_modules/combined-stream": { "version": "1.0.8", "resolved": "https://registry.npmjs.org/combined-stream/-/combined-stream-1.0.8.tgz", @@ -2269,6 +2439,42 @@ "node": "*" } }, + "node_modules/define-data-property": { + "version": "1.1.4", + "resolved": "https://registry.npmjs.org/define-data-property/-/define-data-property-1.1.4.tgz", + "integrity": "sha512-rBMvIzlpA8v6E+SJZoo++HAYqsLrkg7MSfIinMPFhmkorw7X+dOXVJQs+QT69zGkzMyfDnIMN2Wid1+NbL3T+A==", + "license": "MIT", + "optional": true, + "dependencies": { + "es-define-property": "^1.0.0", + "es-errors": "^1.3.0", + "gopd": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/define-properties": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/define-properties/-/define-properties-1.2.1.tgz", + "integrity": "sha512-8QmQKqEASLd5nx0U1B1okLElbUuuttJ/AnYmRXbbbGDWh6uS208EjD4Xqq/I9wK7u0v6O08XhTWnt5XtEbR6Dg==", + "license": "MIT", + "optional": true, + "dependencies": { + "define-data-property": "^1.0.1", + "has-property-descriptors": "^1.0.0", + "object-keys": "^1.1.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, "node_modules/delayed-stream": { "version": "1.0.0", "resolved": "https://registry.npmjs.org/delayed-stream/-/delayed-stream-1.0.0.tgz", @@ -2412,6 +2618,19 @@ "@esbuild/win32-x64": "0.28.2" } }, + "node_modules/escape-string-regexp": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-4.0.0.tgz", + "integrity": "sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA==", + "license": "MIT", + "optional": true, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/estree-walker": { "version": "3.0.3", "resolved": "https://registry.npmjs.org/estree-walker/-/estree-walker-3.0.3.tgz", @@ -2613,6 +2832,39 @@ "node": ">= 0.4" } }, + "node_modules/global-agent": { + "version": "4.1.3", + "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-4.1.3.tgz", + "integrity": "sha512-KUJEViiuFT3I97t+GYMikLPJS2Lfo/S2F+DQuBWzuzaMPnvt5yyZePzArx36fBzpGTxZjIpDbXLeySLgh+k76g==", + "license": "BSD-3-Clause", + "optional": true, + "dependencies": { + "globalthis": "^1.0.2", + "matcher": "^4.0.0", + "semver": "^7.3.5", + "serialize-error": "^8.1.0" + }, + "engines": { + "node": ">=10.0" + } + }, + "node_modules/globalthis": { + "version": "1.0.4", + "resolved": "https://registry.npmjs.org/globalthis/-/globalthis-1.0.4.tgz", + "integrity": "sha512-DpLKbNU4WylpxJykQujfCcwYWiV/Jhm50Goo0wrVILAv5jOr9d+H+UR3PhSCD2rCCEIg0uc+G+muBTwD54JhDQ==", + "license": "MIT", + "optional": true, + "dependencies": { + "define-properties": "^1.2.1", + "gopd": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, "node_modules/gopd": { "version": "1.2.0", "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz", @@ -2642,6 +2894,19 @@ "node": ">=8" } }, + "node_modules/has-property-descriptors": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/has-property-descriptors/-/has-property-descriptors-1.0.2.tgz", + "integrity": "sha512-55JNKuIW+vq4Ke1BjOTjM2YctQIvCT7GFzHwmfZPGo5wnrgkid0YQtnAleFSqumZm4az3n2BS+erby5ipJdgrg==", + "license": "MIT", + "optional": true, + "dependencies": { + "es-define-property": "^1.0.0" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, "node_modules/has-symbols": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/has-symbols/-/has-symbols-1.1.0.tgz", @@ -2703,13 +2968,6 @@ "node": ">= 4" } }, - "node_modules/is-arrayish": { - "version": "0.3.4", - "resolved": "https://registry.npmjs.org/is-arrayish/-/is-arrayish-0.3.4.tgz", - "integrity": "sha512-m6UrgzFVUYawGBh1dUsWR5M2Clqic9RVXC/9f8ceNlv2IcO9j9J/z8UoCLPqtsPBFNzEpfR3xftohbfqDx8EQA==", - "license": "MIT", - "optional": true - }, "node_modules/is-buffer": { "version": "1.1.6", "resolved": "https://registry.npmjs.org/is-buffer/-/is-buffer-1.1.6.tgz", @@ -3069,6 +3327,22 @@ "@jridgewell/sourcemap-codec": "^1.5.5" } }, + "node_modules/matcher": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/matcher/-/matcher-4.0.0.tgz", + "integrity": "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ==", + "license": "MIT", + "optional": true, + "dependencies": { + "escape-string-regexp": "^4.0.0" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/math-intrinsics": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/math-intrinsics/-/math-intrinsics-1.1.0.tgz", @@ -3114,29 +3388,6 @@ "node": ">= 0.6" } }, - "node_modules/minipass": { - "version": "7.1.3", - "resolved": "https://registry.npmjs.org/minipass/-/minipass-7.1.3.tgz", - "integrity": "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A==", - "license": "BlueOak-1.0.0", - "optional": true, - "engines": { - "node": ">=16 || 14 >=14.17" - } - }, - "node_modules/minizlib": { - "version": "3.1.0", - "resolved": "https://registry.npmjs.org/minizlib/-/minizlib-3.1.0.tgz", - "integrity": "sha512-KZxYo1BUkWD2TVFLr0MQoM8vUUigWD3LlD83a/75BqC+4qE0Hb1Vo5v1FgcfaNXvfXzr+5EhQ6ing/CaBijTlw==", - "license": "MIT", - "optional": true, - "dependencies": { - "minipass": "^7.1.2" - }, - "engines": { - "node": ">= 18" - } - }, "node_modules/ms": { "version": "2.1.3", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", @@ -3222,6 +3473,16 @@ "node-gyp-build-test": "build-test.js" } }, + "node_modules/object-keys": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/object-keys/-/object-keys-1.1.1.tgz", + "integrity": "sha512-NuAESUOUMrlIXOfHKzD6bpPu3tYt3xvjNdRIQ+FeT0lNb4K8WR70CaDxhuNguS2XG+GjkyMwOzsN5ZktImfhLA==", + "license": "MIT", + "optional": true, + "engines": { + "node": ">= 0.4" + } + }, "node_modules/obug": { "version": "2.1.4", "resolved": "https://registry.npmjs.org/obug/-/obug-2.1.4.tgz", @@ -3246,16 +3507,16 @@ } }, "node_modules/onnxruntime-common": { - "version": "1.19.2", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.19.2.tgz", - "integrity": "sha512-a4R7wYEVFbZBlp0BfhpbFWqe4opCor3KM+5Wm22Az3NGDcQMiU2hfG/0MfnBs+1ZrlSGmlgWeMcXQkDk1UFb8Q==", + "version": "1.30.0", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.30.0.tgz", + "integrity": "sha512-7fdVWjAID1dVhH/G8qK3APARunV4VkBFoCQAP7qp4Wkab0mrorvmc+sqiT+mKXOzDqdjN5j+/Z9nb4gzNPWcyA==", "license": "MIT", "optional": true }, "node_modules/onnxruntime-node": { - "version": "1.19.2", - "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.19.2.tgz", - "integrity": "sha512-9eHMP/HKbbeUcqte1JYzaaRC8JPn7ojWeCeoyShO86TOR97OCyIyAIOGX3V95ErjslVhJRXY8Em/caIUc0hm1Q==", + "version": "1.30.0", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.30.0.tgz", + "integrity": "sha512-twhs1C2C/BFkz1yc5OY0KIU2GUq6DURO7hD4bx5Q2Qy3nAMJwRXW8xU3NVczE29VA9lolLOYepoD8fjTGOfIqw==", "hasInstallScript": true, "license": "MIT", "optional": true, @@ -3265,36 +3526,37 @@ "linux" ], "dependencies": { - "onnxruntime-common": "1.19.2", - "tar": "^7.0.1" + "adm-zip": "^0.6.0", + "global-agent": "^4.1.3", + "onnxruntime-common": "1.30.0" } }, "node_modules/onnxruntime-web": { - "version": "1.21.0-dev.20241024-d9ca84ef96", - "resolved": "https://registry.npmjs.org/onnxruntime-web/-/onnxruntime-web-1.21.0-dev.20241024-d9ca84ef96.tgz", - "integrity": "sha512-ANSQfMALvCviN3Y4tvTViKofKToV1WUb2r2VjZVCi3uUBPaK15oNJyIxhsNyEckBr/Num3JmSXlkHOD8HfVzSQ==", + "version": "1.31.0-dev.20260914-8d85527a0", + "resolved": "https://registry.npmjs.org/onnxruntime-web/-/onnxruntime-web-1.31.0-dev.20260914-8d85527a0.tgz", + "integrity": "sha512-Iy7rtadoBgxS/LLvDr3QW38DB1PNXRnr0GJMcL0TAt7c9qjgVQl83UlGCVyeAnK2InpmW8Uc3PL8XIuqtDeF6g==", "license": "MIT", "optional": true, "dependencies": { - "flatbuffers": "^1.12.0", + "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", "long": "^5.2.3", - "onnxruntime-common": "1.20.0-dev.20241016-2b8fc5529b", + "onnxruntime-common": "1.31.0-dev.20260911-2a43ec07e", "platform": "^1.3.6", "protobufjs": "^7.2.4" } }, "node_modules/onnxruntime-web/node_modules/flatbuffers": { - "version": "1.12.0", - "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-1.12.0.tgz", - "integrity": "sha512-c7CZADjRcl6j0PlvFy0ZqXQ67qSEZfrVPynmnL+2zPc+NtMvrF8Y0QceMo7QqnSPc7+uWjUIAbvCQ5WIKlMVdQ==", - "license": "SEE LICENSE IN LICENSE.txt", + "version": "25.9.23", + "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-25.9.23.tgz", + "integrity": "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ==", + "license": "Apache-2.0", "optional": true }, "node_modules/onnxruntime-web/node_modules/onnxruntime-common": { - "version": "1.20.0-dev.20241016-2b8fc5529b", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.20.0-dev.20241016-2b8fc5529b.tgz", - "integrity": "sha512-KZK8b6zCYGZFjd4ANze0pqBnqnFTS3GIVeclQpa2qseDpXrCQJfkWBixRcrZShNhm3LpFOZ8qJYFC5/qsJK9WQ==", + "version": "1.31.0-dev.20260911-2a43ec07e", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.31.0-dev.20260911-2a43ec07e.tgz", + "integrity": "sha512-gBuF6U32YErKIAt+yD7DeGBpjRxhJ1uJwako6ygjokSZULDU6Pd+KWComX8Tumk1hYVHRxjPj7mdMnD0ryAPOw==", "license": "MIT", "optional": true }, @@ -3495,44 +3757,70 @@ "node": ">=10" } }, + "node_modules/serialize-error": { + "version": "8.1.0", + "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-8.1.0.tgz", + "integrity": "sha512-3NnuWfM6vBYoy5gZFvHiYsVbafvI9vZv/+jlIigFn4oP4zjNPK3LhcY0xSCgeb1a5L8jO71Mit9LlNoi2UfDDQ==", + "license": "MIT", + "optional": true, + "dependencies": { + "type-fest": "^0.20.2" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/sharp": { - "version": "0.33.5", - "resolved": "https://registry.npmjs.org/sharp/-/sharp-0.33.5.tgz", - "integrity": "sha512-haPVm1EkS9pgvHrQ/F3Xy+hgcuMV0Wm9vfIBSiwZ05k+xgb0PkBQpGsAA/oWdDobNaZTH5ppvHtzCFbnSEwHVw==", - "hasInstallScript": true, + "version": "0.35.4", + "resolved": "https://registry.npmjs.org/sharp/-/sharp-0.35.4.tgz", + "integrity": "sha512-n++8XWcj+jCOr2IOl7h8LbKnGBDY4aPbmprMONBNFdn0ImXqpGVv5zliDs0V9HbmbCQLpbuo2ej9rAoOQTvMDA==", "license": "Apache-2.0", "optional": true, "dependencies": { - "color": "^4.2.3", - "detect-libc": "^2.0.3", - "semver": "^7.6.3" + "@img/colour": "^1.1.0", + "detect-libc": "^2.1.2", + "semver": "^7.8.5" }, "engines": { - "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + "node": ">=20.9.0" }, "funding": { "url": "https://opencollective.com/libvips" }, "optionalDependencies": { - "@img/sharp-darwin-arm64": "0.33.5", - "@img/sharp-darwin-x64": "0.33.5", - "@img/sharp-libvips-darwin-arm64": "1.0.4", - "@img/sharp-libvips-darwin-x64": "1.0.4", - "@img/sharp-libvips-linux-arm": "1.0.5", - "@img/sharp-libvips-linux-arm64": "1.0.4", - "@img/sharp-libvips-linux-s390x": "1.0.4", - "@img/sharp-libvips-linux-x64": "1.0.4", - "@img/sharp-libvips-linuxmusl-arm64": "1.0.4", - "@img/sharp-libvips-linuxmusl-x64": "1.0.4", - "@img/sharp-linux-arm": "0.33.5", - "@img/sharp-linux-arm64": "0.33.5", - "@img/sharp-linux-s390x": "0.33.5", - "@img/sharp-linux-x64": "0.33.5", - "@img/sharp-linuxmusl-arm64": "0.33.5", - "@img/sharp-linuxmusl-x64": "0.33.5", - "@img/sharp-wasm32": "0.33.5", - "@img/sharp-win32-ia32": "0.33.5", - "@img/sharp-win32-x64": "0.33.5" + "@img/sharp-darwin-arm64": "0.35.4", + "@img/sharp-darwin-x64": "0.35.4", + "@img/sharp-freebsd-wasm32": "0.35.4", + "@img/sharp-libvips-darwin-arm64": "1.3.3", + "@img/sharp-libvips-darwin-x64": "1.3.3", + "@img/sharp-libvips-linux-arm": "1.3.3", + "@img/sharp-libvips-linux-arm64": "1.3.3", + "@img/sharp-libvips-linux-ppc64": "1.3.3", + "@img/sharp-libvips-linux-riscv64": "1.3.3", + "@img/sharp-libvips-linux-s390x": "1.3.3", + "@img/sharp-libvips-linux-x64": "1.3.3", + "@img/sharp-libvips-linuxmusl-arm64": "1.3.3", + "@img/sharp-libvips-linuxmusl-x64": "1.3.3", + "@img/sharp-linux-arm": "0.35.4", + "@img/sharp-linux-arm64": "0.35.4", + "@img/sharp-linux-ppc64": "0.35.4", + "@img/sharp-linux-riscv64": "0.35.4", + "@img/sharp-linux-s390x": "0.35.4", + "@img/sharp-linux-x64": "0.35.4", + "@img/sharp-linuxmusl-arm64": "0.35.4", + "@img/sharp-linuxmusl-x64": "0.35.4", + "@img/sharp-webcontainers-wasm32": "0.35.4", + "@img/sharp-win32-arm64": "0.35.4", + "@img/sharp-win32-ia32": "0.35.4", + "@img/sharp-win32-x64": "0.35.4" + }, + "peerDependenciesMeta": { + "@types/node": { + "optional": true + } } }, "node_modules/shebang-command": { @@ -3565,16 +3853,6 @@ "dev": true, "license": "ISC" }, - "node_modules/simple-swizzle": { - "version": "0.2.4", - "resolved": "https://registry.npmjs.org/simple-swizzle/-/simple-swizzle-0.2.4.tgz", - "integrity": "sha512-nAu1WFPQSMNr2Zn9PGSZK9AGn4t/y97lEm+MXTtUDwfP0ksAIX4nO+6ruD9Jwut4C49SB1Ws+fbXsm/yScWOHw==", - "license": "MIT", - "optional": true, - "dependencies": { - "is-arrayish": "^0.3.1" - } - }, "node_modules/source-map-js": { "version": "1.2.1", "resolved": "https://registry.npmjs.org/source-map-js/-/source-map-js-1.2.1.tgz", @@ -3633,23 +3911,6 @@ "node": ">=12.17" } }, - "node_modules/tar": { - "version": "7.5.22", - "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.22.tgz", - "integrity": "sha512-MFO/QzvtAOmJbkhOaCTvbGcFN9L9b+JunIsDwaKljSOdcLMea3NJ1k9Usz/rjdfSXTq4dfzfeS7W4p4YOAAHeA==", - "license": "BlueOak-1.0.0", - "optional": true, - "dependencies": { - "@isaacs/fs-minipass": "^4.0.0", - "chownr": "^3.0.0", - "minipass": "^7.1.2", - "minizlib": "^3.1.0", - "yallist": "^5.0.0" - }, - "engines": { - "node": ">=18" - } - }, "node_modules/tinybench": { "version": "2.9.0", "resolved": "https://registry.npmjs.org/tinybench/-/tinybench-2.9.0.tgz", @@ -3973,6 +4234,19 @@ "fsevents": "~2.3.3" } }, + "node_modules/type-fest": { + "version": "0.20.2", + "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.20.2.tgz", + "integrity": "sha512-Ne+eE4r0/iWnpAxD852z3A+N0Bt5RN//NjJwRd2VFHEmrywxf5vsZlh4R6lixl6B+wz/8d+maTSAkN1FIkI3LQ==", + "license": "(MIT OR CC0-1.0)", + "optional": true, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/typescript": { "version": "5.9.3", "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", @@ -4252,16 +4526,6 @@ "node": ">=12.17" } }, - "node_modules/yallist": { - "version": "5.0.0", - "resolved": "https://registry.npmjs.org/yallist/-/yallist-5.0.0.tgz", - "integrity": "sha512-YgvUTfwqyc7UXVMrB+SImsVYSmTS8X/tSrtdNZMImM+n7+QTriRXyXim0mBrTXNeqzVF0KWGgHPeiyViFFrNDw==", - "license": "BlueOak-1.0.0", - "optional": true, - "engines": { - "node": ">=18" - } - }, "node_modules/zod": { "version": "4.5.4", "resolved": "https://registry.npmjs.org/zod/-/zod-4.5.4.tgz", diff --git a/package.json b/package.json index 57b8710..01ddd0b 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "@pranitmodi/code-intel", - "version": "0.1.1", - "description": "Local-first semantic code indexing and retrieval server, exposed to AI agents via MCP.", + "version": "0.2.0", + "description": "Local-first repository context layer for AI coding agents, exposed through MCP.", "type": "module", "license": "MIT", "publishConfig": { @@ -14,7 +14,10 @@ "embeddings", "cursor", "code-search", - "local-first" + "local-first", + "semantic-search", + "context-retrieval", + "coding-agents" ], "repository": { "type": "git", @@ -25,7 +28,7 @@ }, "homepage": "https://github.com/pranitmodi/code-intel#readme", "engines": { - "node": ">=20" + "node": ">=20.9" }, "bin": { "code-intel": "dist/cli/index.js" @@ -33,7 +36,11 @@ "files": [ "dist", "README.md", - "LICENSE" + "LICENSE", + "CHANGELOG.md", + "SECURITY.md", + "CODE_OF_CONDUCT.md", + "docs" ], "scripts": { "build": "tsc -p tsconfig.json", @@ -44,10 +51,11 @@ "typecheck": "tsc --noEmit", "link": "npm run build && npm link", "onboard": "node dist/cli/index.js onboard", + "benchmark:retrieval": "tsx src/cli/index.ts benchmark --format json", "prepublishOnly": "npm run build && npm run test:unit" }, "dependencies": { - "@lancedb/lancedb": "^0.38.0", + "@lancedb/lancedb": "^0.39.0", "@modelcontextprotocol/server": "^2.0.0", "@parcel/watcher": "^2.6.0", "@tree-sitter-grammars/tree-sitter-yaml": "^0.7.1", @@ -67,6 +75,9 @@ "web-tree-sitter": "^0.27.0", "zod": "^4.5.4" }, + "overrides": { + "@huggingface/transformers": "4.3.0" + }, "devDependencies": { "@modelcontextprotocol/client": "^2.0.0", "@types/js-yaml": "^4.0.9", diff --git a/src/benchmark/datasets/codeIntelTasks.ts b/src/benchmark/datasets/codeIntelTasks.ts new file mode 100644 index 0000000..8fe6a40 --- /dev/null +++ b/src/benchmark/datasets/codeIntelTasks.ts @@ -0,0 +1,109 @@ +export interface BenchmarkTask { + id: string; + prompt: string; + category: + | 'known-symbol' + | 'conceptual' + | 'feature' + | 'bug' + | 'refactor' + | 'cross-cutting' + | 'tests' + | 'configuration'; + relevantFiles: string[]; + relevantSymbols?: string[]; + /** 0 irrelevant … 3 essential, keyed by file path. */ + relevanceLevels?: Record; +} + +export const CODE_INTEL_TASKS: BenchmarkTask[] = [ + { + id: 'known-search-codebase', + category: 'known-symbol', + prompt: 'Where is searchCodebase implemented?', + relevantFiles: ['src/search/searchCodebase.ts'], + relevantSymbols: ['searchCodebase'], + relevanceLevels: { + 'src/search/searchCodebase.ts': 3, + 'src/retrieval/hybrid.ts': 2, + 'src/mcp/server.ts': 1 + } + }, + { + id: 'conceptual-hybrid-search', + category: 'conceptual', + prompt: 'How does hybrid semantic and keyword ranking work?', + relevantFiles: ['src/search/searchCodebase.ts', 'src/retrieval/hybrid.ts', 'src/retrieval/score.ts'], + relevantSymbols: ['hybridSearch', 'combineScore'], + relevanceLevels: { + 'src/retrieval/hybrid.ts': 3, + 'src/retrieval/score.ts': 3, + 'src/search/searchCodebase.ts': 2 + } + }, + { + id: 'feature-task-context', + category: 'feature', + prompt: 'Add rate limiting around get_task_context and update the tests.', + relevantFiles: ['src/retrieval/taskContext.ts', 'src/mcp/server.ts'], + relevantSymbols: ['getTaskContext'], + relevanceLevels: { + 'src/retrieval/taskContext.ts': 3, + 'src/mcp/server.ts': 2, + 'src/retrieval/intent.ts': 1 + } + }, + { + id: 'bug-stale-index', + category: 'bug', + prompt: 'Fix stale index detection when discoverable file counts drift.', + relevantFiles: ['src/indexer/status.ts'], + relevantSymbols: ['getIndexStatus'], + relevanceLevels: { 'src/indexer/status.ts': 3, 'src/indexer/Indexer.ts': 1 } + }, + { + id: 'refactor-search-ranking', + category: 'refactor', + prompt: 'Move ranking weights out of searchCodebase into a dedicated scorer.', + relevantFiles: ['src/retrieval/score.ts', 'src/search/searchCodebase.ts'], + relevantSymbols: ['combineScore'], + relevanceLevels: { + 'src/retrieval/score.ts': 3, + 'src/search/searchCodebase.ts': 2, + 'src/config/types.ts': 1 + } + }, + { + id: 'cross-cutting-mcp-instructions', + category: 'cross-cutting', + prompt: 'Add request IDs to MCP tool responses and the Cursor skill instructions.', + relevantFiles: ['src/mcp/server.ts', 'src/cursor/mcpInstructions.ts', 'src/cursor/skill.ts'], + relevanceLevels: { + 'src/mcp/server.ts': 3, + 'src/cursor/mcpInstructions.ts': 2, + 'src/cursor/skill.ts': 2 + } + }, + { + id: 'tests-tree-scan', + category: 'tests', + prompt: 'Find the tests that cover workspace-wide Grep and Glob denial.', + relevantFiles: ['tests/unit/tree-scan-policy.test.ts', 'src/cursor/treeScanPolicy.ts'], + relevantSymbols: ['shouldDenyTreeScan'], + relevanceLevels: { + 'tests/unit/tree-scan-policy.test.ts': 3, + 'src/cursor/treeScanPolicy.ts': 2 + } + }, + { + id: 'config-embedding', + category: 'configuration', + prompt: 'Where is the default embedding model and search weight configuration?', + relevantFiles: ['src/config/defaults.ts', 'src/config/types.ts'], + relevanceLevels: { + 'src/config/defaults.ts': 3, + 'src/config/types.ts': 2, + 'src/config/load.ts': 1 + } + } +]; diff --git a/src/benchmark/metrics.ts b/src/benchmark/metrics.ts new file mode 100644 index 0000000..c66d177 --- /dev/null +++ b/src/benchmark/metrics.ts @@ -0,0 +1,50 @@ +export function precisionAtK(retrieved: string[], relevant: Set, k: number): number { + if (k <= 0) return 0; + const top = retrieved.slice(0, k); + if (top.length === 0) return 0; + const hits = top.filter((item) => relevant.has(item)).length; + return hits / top.length; +} + +export function recallAtK(retrieved: string[], relevant: Set, k: number): number { + if (relevant.size === 0) return 0; + const top = retrieved.slice(0, k); + const hits = top.filter((item) => relevant.has(item)).length; + return hits / relevant.size; +} + +export function meanReciprocalRank(retrieved: string[], relevant: Set): number { + const index = retrieved.findIndex((item) => relevant.has(item)); + if (index < 0) return 0; + return 1 / (index + 1); +} + +export function dcg(gains: number[]): number { + return gains.reduce((sum, gain, i) => sum + gain / Math.log2(i + 2), 0); +} + +export function ndcgAtK( + retrieved: string[], + relevanceLevels: Record, + k: number +): number { + const gains = retrieved.slice(0, k).map((item) => relevanceLevels[item] ?? 0); + const ideal = Object.values(relevanceLevels) + .sort((a, b) => b - a) + .slice(0, k); + const denom = dcg(ideal); + if (denom === 0) return 0; + return dcg(gains) / denom; +} + +export function percentile(values: number[], p: number): number { + if (values.length === 0) return 0; + const sorted = [...values].sort((a, b) => a - b); + const idx = Math.min(sorted.length - 1, Math.max(0, Math.ceil((p / 100) * sorted.length) - 1)); + return sorted[idx] ?? 0; +} + +export function tokenReductionPercent(baselineTokens: number, systemTokens: number): number { + if (baselineTokens <= 0) return 0; + return ((baselineTokens - systemTokens) / baselineTokens) * 100; +} diff --git a/src/benchmark/runner.ts b/src/benchmark/runner.ts new file mode 100644 index 0000000..512faa2 --- /dev/null +++ b/src/benchmark/runner.ts @@ -0,0 +1,192 @@ +import type { LoadConfigOptions } from '../config/load.js'; +import { createContext } from '../context.js'; +import { getTaskContext } from '../retrieval/taskContext.js'; +import { searchCodebase } from '../search/searchCodebase.js'; +import { estimateTokensFromText } from '../utils/tokens.js'; +import { CODE_INTEL_TASKS, type BenchmarkTask } from './datasets/codeIntelTasks.js'; +import { + meanReciprocalRank, + ndcgAtK, + percentile, + precisionAtK, + recallAtK, + tokenReductionPercent +} from './metrics.js'; +import { keywordFromPrompt, workspaceScan } from './workspaceScan.js'; + +export interface RetrievalBenchmarkOptions { + repoRoot: string; + taskId?: string; + loadOptions?: Omit; +} + +export interface RetrievalBenchmarkReport { + version: 1; + repository: string; + timestamp: string; + tasks: Array<{ + id: string; + relevantFiles: string[]; + semanticFiles: string[]; + retrievedFiles: string[]; + missingRelevantFiles: string[]; + baselineTokens: number; + semanticTokens: number; + taskContextTokens: number; + tokenReductionVsScan: number; + precisionAt5: number; + precisionAt5Ceiling: number; + relevantCoverageAt5: number; + recallAt10: number; + mrr: number; + ndcg: number; + scanLatencyMs: number; + semanticLatencyMs: number; + taskContextLatencyMs: number; + }>; + aggregate: { + tokenReduction: number; + precisionAt5: number; + precisionAt5Ceiling: number; + relevantCoverageAt5: number; + recallAt10: number; + mrr: number; + ndcg: number; + p50LatencyMs: number; + p95LatencyMs: number; + p99LatencyMs: number; + }; + text: string; +} + +function relevantSet(task: BenchmarkTask): Set { + return new Set(task.relevantFiles); +} + +function levels(task: BenchmarkTask): Record { + if (task.relevanceLevels) return task.relevanceLevels; + return Object.fromEntries(task.relevantFiles.map((file) => [file, 3])); +} + +export async function runRetrievalBenchmark( + options: RetrievalBenchmarkOptions +): Promise { + const app = await createContext(options.repoRoot, options.loadOptions); + const tasks = options.taskId + ? CODE_INTEL_TASKS.filter((task) => task.id === options.taskId) + : CODE_INTEL_TASKS; + if (tasks.length === 0) { + throw new Error(`No benchmark task matches "${options.taskId}".`); + } + + const rows: RetrievalBenchmarkReport['tasks'] = []; + const latencies: number[] = []; + + for (const task of tasks) { + const relevant = relevantSet(task); + const relLevels = levels(task); + const scan = workspaceScan(options.repoRoot, keywordFromPrompt(task.prompt)); + + const semanticStarted = Date.now(); + const semantic = await searchCodebase( + task.prompt, + app.vectorStore, + app.embeddingProvider, + app.config.search, + { limit: 10, maxTokens: 4000 } + ); + const semanticLatencyMs = Date.now() - semanticStarted; + const semanticFiles = semantic.map((item) => item.file); + const semanticTokens = estimateTokensFromText(JSON.stringify({ results: semantic })); + + const taskStarted = Date.now(); + const pkg = await getTaskContext(task.prompt, app, { maxTokens: 8000 }); + const taskContextLatencyMs = Date.now() - taskStarted; + const taskFiles = pkg.files.map((file) => file.path); + // MCP intentionally strips the internal trace; benchmark the payload an agent actually receives. + const { files, trace: _trace, ...pkgWithoutSource } = pkg; + // Count the source-bearing payload delivered to the agent, while keeping source out of the report rows. + const taskTokens = estimateTokensFromText(JSON.stringify({ ...pkgWithoutSource, files })); + + latencies.push(scan.latencyMs, semanticLatencyMs, taskContextLatencyMs); + rows.push({ + id: task.id, + relevantFiles: task.relevantFiles, + semanticFiles: [...new Set(semanticFiles)], + retrievedFiles: taskFiles, + missingRelevantFiles: task.relevantFiles.filter((file) => !taskFiles.includes(file)), + baselineTokens: scan.estimatedTokens, + semanticTokens, + taskContextTokens: taskTokens, + tokenReductionVsScan: tokenReductionPercent(scan.estimatedTokens, taskTokens), + precisionAt5: precisionAtK(taskFiles, relevant, 5), + precisionAt5Ceiling: Math.min(5, relevant.size) / 5, + relevantCoverageAt5: recallAtK(taskFiles, relevant, 5), + recallAt10: recallAtK(taskFiles, relevant, 10), + mrr: meanReciprocalRank(taskFiles.length > 0 ? taskFiles : semanticFiles, relevant), + ndcg: ndcgAtK(taskFiles, relLevels, 10), + scanLatencyMs: scan.latencyMs, + semanticLatencyMs, + taskContextLatencyMs + }); + } + + const avg = (pick: (row: (typeof rows)[number]) => number) => + rows.reduce((sum, row) => sum + pick(row), 0) / rows.length; + + const aggregate = { + tokenReduction: avg((row) => row.tokenReductionVsScan) / 100, + precisionAt5: avg((row) => row.precisionAt5), + precisionAt5Ceiling: avg((row) => row.precisionAt5Ceiling), + relevantCoverageAt5: avg((row) => row.relevantCoverageAt5), + recallAt10: avg((row) => row.recallAt10), + mrr: avg((row) => row.mrr), + ndcg: avg((row) => row.ndcg), + p50LatencyMs: percentile(latencies, 50), + p95LatencyMs: percentile(latencies, 95), + p99LatencyMs: percentile(latencies, 99) + }; + + const header = 'Task Baseline Code-Intel Reduction'; + const body = rows + .map((row) => { + const id = row.id.padEnd(28); + const base = String(Math.round(row.baselineTokens)).padStart(10); + const ci = String(Math.round(row.taskContextTokens)).padStart(12); + const red = `${row.tokenReductionVsScan.toFixed(1)}%`.padStart(12); + return `${id}${base}${ci}${red}`; + }) + .join('\n'); + + const text = [ + 'code-intel benchmark', + '', + header, + '-'.repeat(64), + body, + '', + `Average token reduction: ${(aggregate.tokenReduction * 100).toFixed(1)}%`, + '', + 'Retrieval', + '-'.repeat(64), + `Precision@5 ${aggregate.precisionAt5.toFixed(2)} (dataset ceiling ${aggregate.precisionAt5Ceiling.toFixed(2)})`, + `Relevant coverage@5 ${aggregate.relevantCoverageAt5.toFixed(2)}`, + `Recall@10 ${aggregate.recallAt10.toFixed(2)}`, + `MRR ${aggregate.mrr.toFixed(2)}`, + `NDCG ${aggregate.ndcg.toFixed(2)}`, + '', + 'Latency', + '-'.repeat(64), + `p50 ${Math.round(aggregate.p50LatencyMs)}ms`, + `p95 ${Math.round(aggregate.p95LatencyMs)}ms` + ].join('\n'); + + return { + version: 1, + repository: options.repoRoot, + timestamp: new Date().toISOString(), + tasks: rows, + aggregate, + text + }; +} diff --git a/src/benchmark/workspaceScan.ts b/src/benchmark/workspaceScan.ts new file mode 100644 index 0000000..c7f339b --- /dev/null +++ b/src/benchmark/workspaceScan.ts @@ -0,0 +1,97 @@ +import { spawnSync } from 'node:child_process'; +import { statSync } from 'node:fs'; +import { estimateTokensFromChars } from '../utils/tokens.js'; + +const RG_GLOBS = ['!node_modules/**', '!.git/**', '!dist/**', '!build/**']; +const MAX_READ_FILES = 12; + +export interface WorkspaceScanResult { + filesRead: number; + linesRead: number; + estimatedTokens: number; + latencyMs: number; + files: string[]; +} + +function rgBin(): string { + return process.env.CODE_INTEL_RG ?? 'rg'; +} + +function assertRgSucceeded( + result: { error?: Error; status: number | null; stderr?: Buffer | string | null }, + operation: string, + allowNoMatches = false +): void { + if (result.error) { + throw new Error( + `Workspace-scan baseline could not run ripgrep (${rgBin()}): ${result.error.message}. ` + + 'Install ripgrep or set CODE_INTEL_RG to its executable path.' + ); + } + if (result.status === 0 || (allowNoMatches && result.status === 1)) return; + const stderr = + typeof result.stderr === 'string' + ? result.stderr.trim() + : Buffer.from(result.stderr ?? Buffer.alloc(0)).toString('utf8').trim(); + throw new Error( + `Workspace-scan baseline failed during ${operation} (ripgrep exit ${result.status ?? 'unknown'})` + + `${stderr ? `: ${stderr}` : '.'}` + ); +} + +function globArgs(repo: string, extra: string[]): string[] { + const globs = RG_GLOBS.flatMap((g) => ['--glob', g]); + return [...globs, ...extra, repo]; +} + +export function workspaceScan(repo: string, keyword: string): WorkspaceScanResult { + const started = Date.now(); + const listed = spawnSync(rgBin(), globArgs(repo, ['--files']), { + encoding: 'buffer', + maxBuffer: 32 * 1024 * 1024 + }); + assertRgSucceeded(listed, 'file listing'); + const globTokens = estimateTokensFromChars((listed.stdout ?? Buffer.alloc(0)).length); + + const dump = spawnSync(rgBin(), globArgs(repo, ['-n', '-C', '2', '--max-count', '200', '--', keyword]), { + encoding: 'buffer', + maxBuffer: 32 * 1024 * 1024 + }); + assertRgSucceeded(dump, 'content search', true); + const filesOut = spawnSync(rgBin(), globArgs(repo, ['-l', '--max-count', '200', '--', keyword]), { + encoding: 'utf8', + maxBuffer: 32 * 1024 * 1024 + }); + assertRgSucceeded(filesOut, 'matching-file search', true); + const files = (filesOut.stdout ?? '') + .split('\n') + .map((line) => line.trim()) + .filter(Boolean); + + let readBytes = 0; + let linesRead = 0; + for (const file of files.slice(0, MAX_READ_FILES)) { + try { + const st = statSync(file); + readBytes += st.size; + linesRead += Math.max(1, Math.ceil(st.size / 40)); + } catch { + /* skip */ + } + } + + const grepDumpTokens = estimateTokensFromChars((dump.stdout ?? Buffer.alloc(0)).length); + const readTopFilesTokens = estimateTokensFromChars(readBytes); + return { + filesRead: Math.min(files.length, MAX_READ_FILES), + linesRead, + estimatedTokens: globTokens + grepDumpTokens + readTopFilesTokens, + latencyMs: Date.now() - started, + files: files.slice(0, MAX_READ_FILES) + }; +} + +export function keywordFromPrompt(prompt: string): string { + const ident = prompt.match(/\b[A-Za-z][A-Za-z0-9]{3,}\b/g); + return ident?.[ident.length - 1] ?? prompt.split(/\s+/)[0] ?? prompt; +} diff --git a/src/chunker/chunkMetadata.ts b/src/chunker/chunkMetadata.ts new file mode 100644 index 0000000..73d68ec --- /dev/null +++ b/src/chunker/chunkMetadata.ts @@ -0,0 +1,167 @@ +export interface ChunkExtraMetadata { + imports: string[]; + exports: string[]; + referencedSymbols: string[]; + isTest: boolean; + isConfig: boolean; +} + +const CONFIG_BASENAMES = new Set([ + 'package.json', + 'package-lock.json', + 'pnpm-lock.yaml', + 'yarn.lock', + 'tsconfig.json', + 'jsconfig.json', + 'pyproject.toml', + 'cargo.toml', + 'go.mod', + 'go.sum', + 'dockerfile', + 'docker-compose.yml', + 'docker-compose.yaml', + 'makefile', + '.eslintrc', + '.eslintrc.js', + '.eslintrc.cjs', + '.prettierrc', + 'vite.config.ts', + 'vite.config.js', + 'webpack.config.js', + 'vitest.config.ts', + 'jest.config.js' +]); + +const TS_JS_IMPORT = + /(?:import|export)\s+(?:type\s+)?(?:[\s\S]*?\s+from\s+)?['"]([^'"]+)['"]|require\(\s*['"]([^'"]+)['"]\s*\)|import\(\s*['"]([^'"]+)['"]\s*\)/g; +const TS_JS_EXPORT_NAME = + /export\s+(?:default\s+)?(?:async\s+)?(?:function|class|const|let|var|type|interface|enum)\s+([A-Za-z_$][\w$]*)/g; +const PYTHON_IMPORT = /(?:from\s+([\w.]+)\s+import|import\s+([\w.]+))/g; +const IDENTIFIER = /\b[A-Z][A-Za-z0-9]+|[a-z][a-zA-Z0-9]{3,}|[a-z]+_[a-z0-9_]+\b/g; +const KEYWORDS = new Set([ + 'function', + 'class', + 'const', + 'return', + 'import', + 'export', + 'from', + 'async', + 'await', + 'interface', + 'type', + 'true', + 'false', + 'null', + 'undefined', + 'this', + 'super', + 'default', + 'extends', + 'implements' +]); + +export function isTestPath(filePath: string): boolean { + const normalized = filePath.replaceAll('\\', '/').toLowerCase(); + return ( + normalized.includes('.test.') || + normalized.includes('.spec.') || + normalized.includes('_test.') || + normalized.includes('/__tests__/') || + normalized.includes('/tests/') || + normalized.endsWith('_test.go') || + /\/test\/[^/]+$/.test(normalized) + ); +} + +export function isConfigPath(filePath: string): boolean { + const normalized = filePath.replaceAll('\\', '/'); + const base = normalized.split('/').pop()?.toLowerCase() ?? ''; + if (CONFIG_BASENAMES.has(base)) return true; + if (base.endsWith('.config.ts') || base.endsWith('.config.js') || base.endsWith('.config.mjs')) { + return true; + } + if (base.startsWith('.') && (base.endsWith('rc') || base.endsWith('rc.json'))) return true; + return false; +} + +export function extractImports(content: string, language: string): string[] { + const found = new Set(); + if (language === 'python') { + for (const match of content.matchAll(PYTHON_IMPORT)) { + const value = match[1] ?? match[2]; + if (value) found.add(value); + } + return [...found]; + } + for (const match of content.matchAll(TS_JS_IMPORT)) { + const value = match[1] ?? match[2] ?? match[3]; + if (value) found.add(value); + } + return [...found]; +} + +export function extractExports(content: string, language: string): string[] { + const found = new Set(); + if (language === 'python') { + for (const match of content.matchAll(/^__all__\s*=\s*\[([^\]]+)\]/m)) { + for (const name of match[1]?.match(/['"]([^'"]+)['"]/g) ?? []) { + found.add(name.replaceAll(/['"]/g, '')); + } + } + return [...found]; + } + for (const match of content.matchAll(TS_JS_EXPORT_NAME)) { + if (match[1]) found.add(match[1]); + } + return [...found]; +} + +export function extractReferencedSymbols(content: string): string[] { + const found = new Set(); + for (const match of content.matchAll(IDENTIFIER)) { + const token = match[0]; + if (!KEYWORDS.has(token) && token.length >= 2) found.add(token); + if (found.size >= 40) break; + } + return [...found]; +} + +export function buildChunkExtraMetadata( + filePath: string, + fileContent: string, + chunkContent: string, + language: string +): ChunkExtraMetadata { + return { + imports: extractImports(fileContent, language), + exports: extractExports(fileContent, language), + referencedSymbols: extractReferencedSymbols(chunkContent), + isTest: isTestPath(filePath), + isConfig: isConfigPath(filePath) + }; +} + +export function serializeChunkExtraMetadata(meta: ChunkExtraMetadata): string { + return JSON.stringify(meta); +} + +export function parseChunkExtraMetadata(raw: string | null | undefined): ChunkExtraMetadata { + if (!raw) { + return { imports: [], exports: [], referencedSymbols: [], isTest: false, isConfig: false }; + } + try { + const parsed = JSON.parse(raw) as Partial; + return { + imports: Array.isArray(parsed.imports) ? parsed.imports.map(String) : [], + exports: Array.isArray(parsed.exports) ? parsed.exports.map(String) : [], + referencedSymbols: Array.isArray(parsed.referencedSymbols) + ? parsed.referencedSymbols.map(String) + : [], + isTest: Boolean(parsed.isTest), + isConfig: Boolean(parsed.isConfig) + }; + } catch { + return { imports: [], exports: [], referencedSymbols: [], isTest: false, isConfig: false }; + } +} diff --git a/src/cli/doctor.ts b/src/cli/doctor.ts index 2708801..6a9b807 100644 --- a/src/cli/doctor.ts +++ b/src/cli/doctor.ts @@ -4,6 +4,7 @@ import type { CodeIntelConfig } from '../config/types.js'; import { resolveRepoPaths } from '../config/paths.js'; import { computeRepoId } from '../utils/repo-id.js'; import { createEmbeddingProvider } from '../embeddings/createEmbeddingProvider.js'; +import { formatCliFailure } from './formatCliFailure.js'; export function ollamaHasModel(models: Array<{ name: string }>, model: string): boolean { return models.some((entry) => entry.name === model || entry.name.startsWith(`${model}:`)); @@ -44,15 +45,19 @@ export async function runDoctor(options: { console.log(`[OK] Model "${config.embedding.model}" is available`); } else { healthy = false; - const hint = fix - ? `could not pull "${config.embedding.model}"` - : `run \`code-intel doctor --fix\` or \`ollama pull ${config.embedding.model}\``; - console.log(`[FAIL] Model "${config.embedding.model}" not found — ${hint}`); + console.log( + formatCliFailure( + new Error( + fix + ? `Ollama model "${config.embedding.model}" is not available locally (pull failed).` + : `Ollama model "${config.embedding.model}" is not available locally.` + ) + ) + ); } } catch (error) { healthy = false; - console.log(`[FAIL] Ollama not reachable at ${config.embedding.host} — is \`ollama serve\` running?`); - console.log(` ${error instanceof Error ? error.message : String(error)}`); + console.log(formatCliFailure(error)); } } else { try { @@ -62,8 +67,7 @@ export async function runDoctor(options: { console.log(`[OK] Model "${config.embedding.model}" returned ${dimensions}-dimension vectors`); } catch (error) { healthy = false; - console.log(`[FAIL] OpenAI-compatible embedding provider is not ready`); - console.log(` ${error instanceof Error ? error.message : String(error)}`); + console.log(formatCliFailure(error)); } } @@ -77,8 +81,8 @@ export async function runDoctor(options: { console.log(`[OK] Index directory is writable (${paths.indexDir})`); } catch (error) { healthy = false; - console.log(`[FAIL] Index directory is not writable (${paths.indexDir})`); - console.log(` ${error instanceof Error ? error.message : String(error)}`); + const detail = error instanceof Error ? error.message : String(error); + console.log(formatCliFailure(new Error(`Index directory is not writable (${paths.indexDir}): ${detail}`))); } console.log( diff --git a/src/cli/formatCliFailure.ts b/src/cli/formatCliFailure.ts new file mode 100644 index 0000000..15d0b9e --- /dev/null +++ b/src/cli/formatCliFailure.ts @@ -0,0 +1,126 @@ +import { IndexerLockedError } from '../indexer/lock.js'; +import { OllamaModelNotFoundError, OllamaNotReachableError } from '../embeddings/OllamaEmbeddingProvider.js'; +import { + EmbeddingConfigurationError, + OpenAICompatibleEmbeddingError +} from '../embeddings/OpenAICompatibleEmbeddingProvider.js'; + +function stepsFor(error: unknown, message: string): string[] { + if ( + error instanceof EmbeddingConfigurationError && + /CODE_INTEL_EMBEDDING_API_KEY is required/i.test(message) + ) { + return [ + 'Export the key in this shell (Cursor MCP env is not inherited by your terminal):', + ' export CODE_INTEL_EMBEDDING_API_KEY=…', + ' export CODE_INTEL_EMBEDDING_USER=… # only if the company proxy requires a user', + 'Then re-run the command.', + 'Or run `code-intel wizard` / `code-intel corporate-setup` to enter the key for this session (it is not written to YAML).', + 'Or switch to local Ollama: `code-intel index --embedding-provider ollama --embedding-model nomic-embed-text` (run `code-intel rebuild` if the index was built with a different model).', + 'Confirm with `code-intel doctor`.' + ]; + } + + if (error instanceof EmbeddingConfigurationError && /base_url is required/i.test(message)) { + return [ + 'Set embedding.base_url in `.code-intel/config.yaml` or `CODE_INTEL_EMBEDDING_BASE_URL`.', + 'Or pass `--embedding-base-url https://your-proxy.example`.', + 'Or switch to Ollama: `--embedding-provider ollama`.', + 'Confirm with `code-intel doctor`.' + ]; + } + + if (error instanceof OpenAICompatibleEmbeddingError) { + if (error.status === 401 || error.status === 403 || /rejected the credentials/i.test(message)) { + return [ + 'Check `CODE_INTEL_EMBEDDING_API_KEY` (and `CODE_INTEL_EMBEDDING_USER` if the proxy requires it).', + 'Keys are never read from YAML — they must be in the environment of this process.', + 'Run `code-intel doctor` after exporting the variables.' + ]; + } + if (error.status === 404 || /was not found \(404\)/i.test(message)) { + return [ + 'Check `embedding.base_url`, `embedding.embeddings_path`, and `embedding.model` in `.code-intel/config.yaml`.', + 'Override with `--embedding-base-url` / `--embedding-model` if needed.', + 'Run `code-intel doctor` to probe the endpoint.' + ]; + } + if (/Cannot reach embedding proxy/i.test(message)) { + return [ + 'Check network / VPN access to the embedding proxy.', + 'If this is a corporate proxy, set `embedding.use_system_ca: true` and use a Node that supports `--use-system-ca`.', + 'Run `code-intel doctor` to retry the probe.' + ]; + } + if (error.status === 408 || error.status === 503 || /Timeout|temporarily unavailable|timed out/i.test(message)) { + return [ + 'Retry in a minute — the proxy may be overloaded or timing out.', + 'If this keeps happening, raise `embedding.timeout_ms` in `.code-intel/config.yaml`.', + 'Or switch to local Ollama: `--embedding-provider ollama --embedding-model nomic-embed-text` (then `code-intel rebuild`).', + 'Confirm with `code-intel doctor`.' + ]; + } + } + + if (error instanceof OllamaNotReachableError || /Cannot reach Ollama/i.test(message)) { + return [ + 'Start Ollama: `ollama serve`.', + 'Or point at a running host: `--embedding-host http://127.0.0.1:11434`.', + 'If you meant the company proxy, this repo’s `.code-intel/config.yaml` should keep `provider: openai-compatible` and you must export `CODE_INTEL_EMBEDDING_API_KEY`.', + 'Run `code-intel doctor`.' + ]; + } + + if (error instanceof OllamaModelNotFoundError || /is not available locally/i.test(message)) { + return [ + 'Pull the model: `ollama pull ` or `code-intel doctor --fix`.', + 'Or set `embedding.model` / `--embedding-model` to a model you already have.' + ]; + } + + if (error instanceof IndexerLockedError || /already indexing this repository/i.test(message)) { + return [ + 'Wait for the other `code-intel index`, `watch`, or MCP watcher to finish.', + 'If that process is dead, delete the `.lock` file under the repo’s index directory (`code-intel status` prints Index location) and retry.' + ]; + } + + if (/index was built with embedding model/i.test(message)) { + return [ + 'Run `code-intel rebuild` to re-embed with the currently configured model.', + 'Or pass `--embedding-model` matching the model shown in `code-intel status`.' + ]; + } + + if (/Index directory is not writable/i.test(message) || /EACCES|EROFS|EPERM/i.test(message)) { + return [ + 'Check permissions on the index directory (`code-intel status` prints Index location).', + 'If the path is on a read-only volume, set `database.path` in `.code-intel/config.yaml` to a writable location.', + 'Retry after fixing ownership (`chown`) or disk space.' + ]; + } + + if (/not indexed|No indexed repository matches|code-intel setup --repo/i.test(message)) { + return [ + 'Index this repo: `code-intel setup --repo ` (or `code-intel index` if setup already ran).', + 'If embeddings fail, export `CODE_INTEL_EMBEDDING_API_KEY` first (keys are never stored in YAML).', + 'Confirm with `code-intel doctor` and `code-intel status`.' + ]; + } + + return ['Run `code-intel doctor` for a full diagnosis of the embedding provider and index.']; +} + +/** Recovery steps for a CLI or MCP failure (no FAIL prefix). */ +export function recoverySteps(error: unknown): string[] { + const message = error instanceof Error ? error.message : String(error); + return stepsFor(error, message); +} + +/** Multi-line CLI failure: diagnosis plus concrete recovery steps. */ +export function formatCliFailure(error: unknown): string { + const message = error instanceof Error ? error.message : String(error); + const steps = recoverySteps(error); + const body = steps.map((step, index) => ` ${index + 1}. ${step}`).join('\n'); + return `[FAIL] ${message}\n\nWhat you can do:\n${body}`; +} diff --git a/src/cli/index.ts b/src/cli/index.ts index a9d26b9..cfdd9c3 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -26,20 +26,28 @@ import { isIndexInsideRepo, scaffoldRepo } from '../indexer/scaffold.js'; import { getIndexStatus, listIndexedReposWithStale } from '../indexer/status.js'; import { Indexer, type IndexSummary } from '../indexer/Indexer.js'; import { watchRepo } from '../indexer/watch.js'; -import { searchCodebase } from '../search/searchCodebase.js'; +import { searchCodebaseDetailed } from '../search/searchCodebase.js'; import { searchSymbol } from '../search/searchSymbol.js'; import { getFileContext } from '../search/getFileContext.js'; import { startMcpServer } from '../mcp/server.js'; import { formatBytes } from '../utils/dirSize.js'; import { runDoctor } from './doctor.js'; +import { formatCliFailure } from './formatCliFailure.js'; import { runSavingsBenchmark } from '../usage/benchmark.js'; import { formatSavingsReport, summarizeUsage } from '../usage/report.js'; import { readBenchmark, readUsageEvents } from '../usage/store.js'; +import { formatSearchExplain } from '../retrieval/explain.js'; +import { formatContextPackage, getTaskContext } from '../retrieval/taskContext.js'; +import { runRetrievalBenchmark } from '../benchmark/runner.js'; +const packageVersion = ( + JSON.parse(readFileSync(new URL('../../package.json', import.meta.url), 'utf-8')) as { version: string } +).version; const program = new Command(); program .name('code-intel') .description('Local-first semantic code indexing and retrieval, exposed to AI agents via MCP.') + .version(packageVersion) .option('--repo ', 'repository root (defaults to the current directory) — use this when a host spawns the process with an unrelated cwd') .option('--embedding-provider ', 'ollama or openai-compatible') .option('--embedding-model ', 'embedding model id (e.g. nomic-embed-text or Qwen3-Embedding-8B)') @@ -506,22 +514,38 @@ program .description('Semantic + keyword + symbol hybrid search') .option('-l, --limit ', 'max results', (v) => Number.parseInt(v, 10)) .option('--json', 'print raw JSON instead of a table') - .action(async (query: string, options: { limit?: number; json?: boolean }) => { + .option('--explain', 'print score breakdown, diversity drops, and token budget') + .action(async (query: string, options: { limit?: number; json?: boolean; explain?: boolean }) => { const context = await createContext(resolveRepoRoot(), cliLoadOptions()); - const results = await searchCodebase(query, context.vectorStore, context.embeddingProvider, context.config.search, { - limit: options.limit - }); + const detailed = await searchCodebaseDetailed( + query, + context.vectorStore, + context.embeddingProvider, + context.config.search, + { limit: options.limit } + ); if (options.json) { - console.log(JSON.stringify({ results }, null, 2)); + console.log( + JSON.stringify( + options.explain ? { results: detailed.results, trace: detailed.trace } : { results: detailed.results }, + null, + 2 + ) + ); return; } - if (results.length === 0) { + if (options.explain) { + console.log(formatSearchExplain(detailed.trace, detailed.selected)); + console.log(''); + } + + if (detailed.results.length === 0) { console.log('No results.'); return; } - results.forEach((result, index) => { + detailed.results.forEach((result, index) => { console.log(`${index + 1}. ${result.file}`); if (result.symbol) console.log(` ${result.symbol}`); console.log(` lines ${result.startLine}-${result.endLine}`); @@ -530,6 +554,58 @@ program }); }); +program + .command('context ') + .description('Assemble a minimal task-oriented context package') + .option('--max-tokens ', 'hard token budget', (v) => Number.parseInt(v, 10)) + .option('--mode ', 'minimal | normal | deep') + .option('--explain', 'include retrieval trace') + .option('--json', 'print JSON') + .action( + async ( + task: string, + options: { maxTokens?: number; mode?: string; explain?: boolean; json?: boolean } + ) => { + const repoRoot = resolveRepoRoot(); + const context = await createContext(repoRoot, cliLoadOptions()); + const status = await getIndexStatus(repoRoot); + const mode = + options.mode === 'minimal' || options.mode === 'normal' || options.mode === 'deep' + ? options.mode + : undefined; + const pkg = await getTaskContext(task, context, { + maxTokens: options.maxTokens, + mode, + stale: status.stale + }); + if (options.json) { + const payload = options.explain ? pkg : { ...pkg, trace: undefined }; + console.log(JSON.stringify(payload, null, 2)); + return; + } + console.log(formatContextPackage(pkg, Boolean(options.explain))); + } + ); + +program + .command('benchmark') + .description('Compare workspace-scan, semantic-search, and task-context retrieval') + .option('--format ', 'text or json', 'text') + .option('--task ', 'run a single labeled task') + .action(async (options: { format?: string; task?: string }) => { + const report = await runRetrievalBenchmark({ + repoRoot: resolveRepoRoot(), + taskId: options.task, + loadOptions: cliLoadOptions() + }); + if (options.format === 'json') { + const { text: _text, ...json } = report; + console.log(JSON.stringify(json, null, 2)); + return; + } + console.log(report.text); + }); + program .command('symbol ') .description('Exact/fuzzy symbol lookup, independent of embeddings') @@ -607,10 +683,15 @@ program ); } - if (status.stale && status.filesDiscoverable != null) { - console.log( - `\nStale index: ${status.filesDiscoverable} files currently discoverable vs ${status.filesIndexed} at last index — run \`code-intel index\`.` - ); + if (status.stale) { + const countHint = + status.filesDiscoverable != null + ? `${status.filesDiscoverable} files currently discoverable vs ${status.filesIndexed} at last index` + : 'working-tree content no longer matches the stored sample'; + console.log(`\nStale index: ${countHint} — run \`code-intel index\`.`); + } + if (status.watch?.lastError) { + console.log(`\nWatch: last error at ${status.watch.lastErrorAt ?? 'unknown'}: ${status.watch.lastError}`); } }); @@ -640,7 +721,7 @@ program program .command('rebuild') - .description('Remove the local index and rebuild it from scratch') + .description('Remove the local index and rebuild it from scratch (also backfills extra_metadata)') .action(async () => { const repoRoot = resolveRepoRoot(); const config = loadConfig(cliLoadOptions()); @@ -701,6 +782,6 @@ program }); void program.parseAsync(process.argv).catch((error: unknown) => { - console.error(`[FAIL] ${error instanceof Error ? error.message : String(error)}`); + console.error(formatCliFailure(error)); process.exitCode = 1; }); diff --git a/src/config/defaults.ts b/src/config/defaults.ts index f21f59c..84271f5 100644 --- a/src/config/defaults.ts +++ b/src/config/defaults.ts @@ -25,7 +25,30 @@ export const DEFAULT_CONFIG: CodeIntelConfig = { defaultLimit: 10, vectorWeight: 0.7, keywordWeight: 0.2, - symbolWeight: 0.1 + symbolWeight: 0.1, + pathWeight: 0.05, + structuralWeight: 0.05, + dependencyWeight: 0.08, + referenceWeight: 0.08, + testWeight: 0.03, + recencyWeight: 0.02, + maxChunksPerFile: 4, + maxChunksPerSymbol: 2 + }, + retrieval: { + seedResults: 8, + maxExpansionHops: 2, + maxContextChunks: 20, + maxContextTokens: 12_000, + confidenceThreshold: 0.15, + retrievalRequired: true, + allowFallbackAfterFailedRetrieval: true + }, + benchmark: { + maxTokenRegressionPercent: 10, + minPrecisionAt5: 0.75, + minRecallAt10: 0.8, + maxP95LatencyMs: 1000 }, security: { allowSensitiveFiles: false diff --git a/src/config/load.ts b/src/config/load.ts index 7758efb..4bf8a06 100644 --- a/src/config/load.ts +++ b/src/config/load.ts @@ -75,7 +75,33 @@ function applyRawConfig(base: CodeIntelConfig, raw: RawConfigFile | undefined): defaultLimit: raw.search?.default_limit ?? base.search.defaultLimit, vectorWeight: raw.search?.vector_weight ?? base.search.vectorWeight, keywordWeight: raw.search?.keyword_weight ?? base.search.keywordWeight, - symbolWeight: raw.search?.symbol_weight ?? base.search.symbolWeight + symbolWeight: raw.search?.symbol_weight ?? base.search.symbolWeight, + pathWeight: raw.search?.path_weight ?? base.search.pathWeight, + structuralWeight: raw.search?.structural_weight ?? base.search.structuralWeight, + dependencyWeight: raw.search?.dependency_weight ?? base.search.dependencyWeight, + referenceWeight: raw.search?.reference_weight ?? base.search.referenceWeight, + testWeight: raw.search?.test_weight ?? base.search.testWeight, + recencyWeight: raw.search?.recency_weight ?? base.search.recencyWeight, + maxChunksPerFile: raw.search?.max_chunks_per_file ?? base.search.maxChunksPerFile, + maxChunksPerSymbol: raw.search?.max_chunks_per_symbol ?? base.search.maxChunksPerSymbol + }, + retrieval: { + seedResults: raw.retrieval?.seed_results ?? base.retrieval.seedResults, + maxExpansionHops: raw.retrieval?.max_expansion_hops ?? base.retrieval.maxExpansionHops, + maxContextChunks: raw.retrieval?.max_context_chunks ?? base.retrieval.maxContextChunks, + maxContextTokens: raw.retrieval?.max_context_tokens ?? base.retrieval.maxContextTokens, + confidenceThreshold: raw.retrieval?.confidence_threshold ?? base.retrieval.confidenceThreshold, + retrievalRequired: raw.retrieval?.retrieval_required ?? base.retrieval.retrievalRequired, + allowFallbackAfterFailedRetrieval: + raw.retrieval?.allow_fallback_after_failed_retrieval ?? + base.retrieval.allowFallbackAfterFailedRetrieval + }, + benchmark: { + maxTokenRegressionPercent: + raw.benchmark?.max_token_regression_percent ?? base.benchmark.maxTokenRegressionPercent, + minPrecisionAt5: raw.benchmark?.min_precision_at_5 ?? base.benchmark.minPrecisionAt5, + minRecallAt10: raw.benchmark?.min_recall_at_10 ?? base.benchmark.minRecallAt10, + maxP95LatencyMs: raw.benchmark?.max_p95_latency_ms ?? base.benchmark.maxP95LatencyMs }, security: { allowSensitiveFiles: raw.security?.allow_sensitive_files ?? base.security.allowSensitiveFiles diff --git a/src/config/paths.ts b/src/config/paths.ts index d2e83b9..4464923 100644 --- a/src/config/paths.ts +++ b/src/config/paths.ts @@ -8,6 +8,7 @@ export interface RepoPaths { metadataDir: string; stateFile: string; progressFile: string; + watchStatusFile: string; logsDir: string; lockFile: string; } @@ -21,6 +22,7 @@ export function resolveRepoPaths(config: CodeIntelConfig, repoRoot: string, repo metadataDir: join(indexDir, 'metadata'), stateFile: join(indexDir, 'state.json'), progressFile: join(indexDir, 'progress.json'), + watchStatusFile: join(indexDir, 'watch-status.json'), logsDir: join(indexDir, 'logs'), lockFile: join(indexDir, '.lock') }; diff --git a/src/config/types.ts b/src/config/types.ts index 9d26158..269ac06 100644 --- a/src/config/types.ts +++ b/src/config/types.ts @@ -39,6 +39,31 @@ export interface SearchConfig { vectorWeight: number; keywordWeight: number; symbolWeight: number; + pathWeight: number; + structuralWeight: number; + dependencyWeight: number; + referenceWeight: number; + testWeight: number; + recencyWeight: number; + maxChunksPerFile: number; + maxChunksPerSymbol: number; +} + +export interface RetrievalConfig { + seedResults: number; + maxExpansionHops: number; + maxContextChunks: number; + maxContextTokens: number; + confidenceThreshold: number; + retrievalRequired: boolean; + allowFallbackAfterFailedRetrieval: boolean; +} + +export interface BenchmarkConfig { + maxTokenRegressionPercent: number; + minPrecisionAt5: number; + minRecallAt10: number; + maxP95LatencyMs: number; } export interface SecurityConfig { @@ -50,6 +75,8 @@ export interface CodeIntelConfig { database: DatabaseConfig; indexing: IndexingConfig; search: SearchConfig; + retrieval: RetrievalConfig; + benchmark: BenchmarkConfig; security: SecurityConfig; /** Additional user-provided ignore patterns, on top of the built-in defaults. */ ignore: string[]; @@ -82,6 +109,29 @@ export interface RawConfigFile { vector_weight: number; keyword_weight: number; symbol_weight: number; + path_weight: number; + structural_weight: number; + dependency_weight: number; + reference_weight: number; + test_weight: number; + recency_weight: number; + max_chunks_per_file: number; + max_chunks_per_symbol: number; + }>; + retrieval?: Partial<{ + seed_results: number; + max_expansion_hops: number; + max_context_chunks: number; + max_context_tokens: number; + confidence_threshold: number; + retrieval_required: boolean; + allow_fallback_after_failed_retrieval: boolean; + }>; + benchmark?: Partial<{ + max_token_regression_percent: number; + min_precision_at_5: number; + min_recall_at_10: number; + max_p95_latency_ms: number; }>; security?: Partial<{ allow_sensitive_files: boolean; diff --git a/src/cursor/mcpInstructions.ts b/src/cursor/mcpInstructions.ts index 312386a..81016fa 100644 --- a/src/cursor/mcpInstructions.ts +++ b/src/cursor/mcpInstructions.ts @@ -1,11 +1,14 @@ export const MCP_SERVER_INSTRUCTIONS = `This server is the primary code-discovery path for Cursor on this machine. -Corpus embeddings already live in local LanceDB (Ollama nomic-embed-text). search_codebase embeds ONLY the user's query locally, then retrieves stored chunks — never re-index or re-embed the tree, and never send vectors to the chat model. +Corpus embeddings already live in local LanceDB (Ollama nomic-embed-text). Only the search query is embedded locally; stored vectors retrieve snippets — never re-index or re-embed the tree, and never send vectors to the chat model. Required order: -1. search_codebase / search_symbol / find_references / get_repo_context -2. get_file_context with a line range for the hit you will edit -3. Grep/Glob/Read only if those tools miss or the path is unindexed +1. For a new or broad coding task, use get_task_context. +2. For a known symbol, use search_symbol. +3. For conceptual exploration, use search_codebase. +4. For call-site analysis, use find_references. +5. Use get_file_context for exact source ranges. +6. Avoid workspace-wide Grep/Glob unless retrieval returned low confidence, the index is stale/unindexed, or those tools genuinely cannot answer. If the workspace is a parent folder, call list_indexed_repos and pass repo (path, id, or name). If nothing is indexed, tell the user to run: code-intel setup --repo . `; diff --git a/src/cursor/preferHookMain.ts b/src/cursor/preferHookMain.ts index 22bd608..d336ceb 100644 --- a/src/cursor/preferHookMain.ts +++ b/src/cursor/preferHookMain.ts @@ -1,5 +1,7 @@ import { readFileSync } from 'node:fs'; +import { loadConfig } from '../config/load.js'; import { recordDeniedScan } from '../usage/record.js'; +import { isFilesystemFallbackOpen } from '../retrieval/fallback.js'; import { shouldDenyTreeScan, TREE_SCAN_DENY_MESSAGE, type TreeScanInput } from './treeScanPolicy.js'; function readStdin(): TreeScanInput { @@ -12,7 +14,11 @@ function readStdin(): TreeScanInput { } const input = readStdin(); -if (shouldDenyTreeScan(input)) { +const config = loadConfig(); +const allowFallback = + config.retrieval.allowFallbackAfterFailedRetrieval && isFilesystemFallbackOpen(config.database.path); + +if (shouldDenyTreeScan(input, { allowFallback })) { recordDeniedScan(input); process.stdout.write( JSON.stringify({ diff --git a/src/cursor/skill.ts b/src/cursor/skill.ts index 4359ae7..4e008de 100644 --- a/src/cursor/skill.ts +++ b/src/cursor/skill.ts @@ -1,26 +1,30 @@ -export const LOCAL_CODE_INTEL_SKILL = `--- -name: local-code-intel -description: >- - Routes codebase search, symbol lookup, and repo orientation through the - local-code-intelligence MCP (LanceDB + local Ollama embeddings). Use whenever - finding code, exploring a repository, indexing, or answering where/how - something works. Never Grep/Glob/Task-explore first and never re-embed the corpus. ---- - -# Local code intelligence - -Corpus embeddings already live in local LanceDB. Ollama (\`nomic-embed-text\`) embeds **only the search query**. Retrieved snippets go to the chat model — not vectors, not the whole tree. - -## MCP namespace - -\`user-local-code-intelligence\` - -## Required order - -1. \`search_codebase\` with \`max_tokens\` 800–1500 (optional \`repo\` for Savor children) -2. \`search_symbol\` / \`find_references\` when you have a name -3. \`get_file_context\` with a line range for the hit you will change -4. Grep/Glob/Read only after those miss, and only on a specific file or subdirectory - -If \`index_status\` says unindexed, run \`code-intel setup --repo \` via Shell, then search again. -`; +export const LOCAL_CODE_INTEL_SKILL = [ + '---', + 'name: local-code-intel', + 'description: >-', + ' Routes codebase search, symbol lookup, and repo orientation through the', + ' local-code-intelligence MCP (LanceDB + local Ollama embeddings). Use whenever', + ' finding code, exploring a repository, indexing, or answering where/how', + ' something works. Never Grep/Glob/Task-explore first and never re-embed the corpus.', + '---', + '', + '# Local code intelligence', + '', + 'Corpus embeddings already live in local LanceDB. Ollama (`nomic-embed-text`) embeds **only the search query**. Retrieved snippets go to the chat model — not vectors, not the whole tree.', + '', + '## MCP namespace', + '', + '`user-local-code-intelligence`', + '', + '## Required order', + '', + '1. For a new or broad coding task, use `get_task_context` (optional `repo` for Savor children).', + '2. For a known symbol, use `search_symbol`.', + '3. For conceptual exploration, use `search_codebase` with `max_tokens` 800–1500.', + '4. For call-site analysis, use `find_references`.', + '5. Use `get_file_context` with a line range for the hit you will change.', + '6. Grep/Glob/Read only after those miss, after a low-confidence retrieval, or when the index is stale/unindexed — and only against a specific file or subdirectory.', + '', + 'If `index_status` says unindexed, run `code-intel setup --repo ` via Shell, then search again.', + '' +].join('\n'); diff --git a/src/cursor/treeScanPolicy.ts b/src/cursor/treeScanPolicy.ts index e200e6a..f0fe14c 100644 --- a/src/cursor/treeScanPolicy.ts +++ b/src/cursor/treeScanPolicy.ts @@ -11,7 +11,7 @@ const EXPLORE_TASK = /\b(explor(e|ing)|search the (code|repo|codebase)|find where|how does)\b/i; export const TREE_SCAN_DENY_MESSAGE = - 'Use MCP user-local-code-intelligence first (search_codebase / search_symbol / get_file_context). Corpus embeddings already live in local LanceDB; Ollama only embeds the query. Grep/Glob/explore is allowed only after those tools miss, and only against a specific file or subdirectory — not the whole repo.'; + 'Use MCP user-local-code-intelligence first (get_task_context / search_codebase / search_symbol / get_file_context). Corpus embeddings already live in local LanceDB; Ollama only embeds the query. Grep/Glob/explore is allowed after those tools miss, after low-confidence retrieval, or against a specific file or subdirectory — not the whole repo.'; function asString(value: unknown): string { return typeof value === 'string' ? value : ''; @@ -33,7 +33,12 @@ export function isWorkspaceRoot(target: string | undefined, roots: string[]): bo return roots.some((root) => normalized === root.replace(/\/+$/, '') || normalized === '.'); } -export function shouldDenyTreeScan(input: TreeScanInput): boolean { +export interface TreeScanPolicyOptions { + /** When true, workspace-wide Grep/Glob is allowed (low-confidence / stale / unindexed retrieval). Explore tasks stay denied. */ + allowFallback?: boolean; +} + +export function shouldDenyTreeScan(input: TreeScanInput, options: TreeScanPolicyOptions = {}): boolean { const event = input.hook_event_name ?? ''; const tool = input.tool_name ?? ''; const args = input.tool_input ?? {}; @@ -49,6 +54,10 @@ export function shouldDenyTreeScan(input: TreeScanInput): boolean { return sub === 'explore' || EXPLORE_TASK.test(blob); } + if (options.allowFallback && (tool === 'Grep' || tool === 'Glob')) { + return false; + } + if (tool === 'Grep') { const target = asString(args.path) || asString(args.target_directory); return !looksLikeFile(target) && isWorkspaceRoot(target || undefined, roots); diff --git a/src/cursor/userRule.ts b/src/cursor/userRule.ts index cc597c1..81350e1 100644 --- a/src/cursor/userRule.ts +++ b/src/cursor/userRule.ts @@ -1,26 +1,28 @@ export const LOCAL_CODE_INTEL_RULE_FILENAME = 'use-local-code-intel.mdc'; -export const LOCAL_CODE_INTEL_USER_RULE = `--- -description: Mandatory local LanceDB/Ollama code search via MCP — never tree-scan or re-embed first -alwaysApply: true ---- - -# Local code index is mandatory - -MCP namespace: \`user-local-code-intelligence\`. The corpus is already embedded in local LanceDB with Ollama (\`nomic-embed-text\`). The chat model must not re-embed or re-index source. - -## Always do this first - -1. \`search_codebase\` (pass \`max_tokens\` 800–1500). Only the query is embedded locally; stored vectors retrieve snippets. -2. \`search_symbol\` for a known name; \`find_references\` for occurrences. -3. \`get_file_context\` with \`start_line\`/\`end_line\` for the range you will edit. -4. \`get_repo_context\` / \`list_indexed_repos\` / \`index_status\` for orientation. - -## Never do this first - -- Grep, Glob, or Task \`explore\` across the repo -- Read whole files to "see how it works" -- Ask the cloud model to embed or index code - -If the workspace is a parent (e.g. Savor), \`list_indexed_repos\` and pass \`repo\`. If unindexed, run \`code-intel setup --repo \` via Shell, then search again. -`; +export const LOCAL_CODE_INTEL_USER_RULE = [ + '---', + 'description: Mandatory local LanceDB/Ollama code search via MCP — never tree-scan or re-embed first', + 'alwaysApply: true', + '---', + '', + '# Local code index is mandatory', + '', + 'MCP namespace: `user-local-code-intelligence`. The corpus is already embedded in local LanceDB with Ollama (`nomic-embed-text`). The chat model must not re-embed or re-index source.', + '', + '## Always do this first', + '', + '1. For a new or broad coding task, `get_task_context`. Only the query is embedded locally; stored vectors retrieve snippets.', + '2. `search_symbol` for a known name; `search_codebase` for conceptual exploration; `find_references` for occurrences.', + '3. `get_file_context` with `start_line`/`end_line` for the range you will edit.', + '4. `get_repo_context` / `list_indexed_repos` / `index_status` for orientation.', + '', + '## Never do this first', + '', + '- Grep, Glob, or Task `explore` across the repo', + '- Read whole files to "see how it works"', + '- Ask the cloud model to embed or index code', + '', + 'If retrieval reports low confidence, the index is stale, or the path is unindexed, targeted filesystem search is allowed. If the workspace is a parent (e.g. Savor), `list_indexed_repos` and pass `repo`. If unindexed, run `code-intel setup --repo ` via Shell, then search again.', + '' +].join('\n'); diff --git a/src/discovery/discover.ts b/src/discovery/discover.ts index a748f0f..302c4fd 100644 --- a/src/discovery/discover.ts +++ b/src/discovery/discover.ts @@ -26,6 +26,18 @@ function loadGitignore(repoRoot: string): string[] { return readFileSync(gitignorePath, 'utf-8').split('\n'); } +export function isIndexableRelativePath(repoRoot: string, relativePath: string, options: DiscoveryOptions): boolean { + const posixPath = toPosix(relativePath); + if (!posixPath || posixPath.startsWith('../') || posixPath === '..') return false; + const ig = ignoreFactory() + .add(DEFAULT_IGNORE_PATTERNS) + .add(loadGitignore(repoRoot)) + .add(options.extraIgnorePatterns); + if (ig.ignores(posixPath)) return false; + if (!options.allowSensitiveFiles && ignoreFactory().add(SECRET_FILE_PATTERNS).ignores(posixPath)) return false; + return true; +} + /** * Recursively discovers candidate files under `repoRoot`, respecting * `.gitignore`, built-in default exclusions, user-configured extra patterns, diff --git a/src/indexer/Indexer.ts b/src/indexer/Indexer.ts index f5e2bc4..a233b8a 100644 --- a/src/indexer/Indexer.ts +++ b/src/indexer/Indexer.ts @@ -1,12 +1,13 @@ import { readFile } from 'node:fs/promises'; import { statSync } from 'node:fs'; -import { basename } from 'node:path'; +import { basename, join } from 'node:path'; import type { CodeIntelConfig } from '../config/types.js'; import type { RepoPaths } from '../config/paths.js'; import { discoverFiles, type DiscoveredFile } from '../discovery/discover.js'; import { looksBinary } from '../discovery/binary-check.js'; import { containsLikelySecret } from '../discovery/secret-scan.js'; import { chunkFile } from '../chunker/chunker.js'; +import { buildChunkExtraMetadata, serializeChunkExtraMetadata } from '../chunker/chunkMetadata.js'; import { computeChunkId, hashChunkContent, hashFileContent } from '../hashing/hash.js'; import { normalizeVector } from '../embeddings/vectorMath.js'; import type { EmbeddingProvider } from '../embeddings/EmbeddingProvider.js'; @@ -17,7 +18,8 @@ import { createLogger } from '../utils/logger.js'; import { mapPool, Mutex } from '../utils/pool.js'; import { acquireLock } from './lock.js'; import { registryEntryFrom, upsertRegistryEntry } from './registry.js'; -import { removeProgress, writeProgress, writeState, type IndexProgress } from './state.js'; +import { computeSampleFingerprint, pickSamplePaths } from './freshness.js'; +import { readState, removeProgress, writeProgress, writeState, type IndexProgress, type IndexState } from './state.js'; const logger = createLogger('indexer'); /** Safety cap so one abnormally large file (e.g. a generated bundle that slipped past ignore rules) can't stall a run. */ @@ -134,36 +136,120 @@ export class Indexer { summary.durationMs = Date.now() - started; const filesInIndex = summary.filesIndexed + summary.filesUnchanged + summary.filesRenamed; + await this.persistState(summary, { + filesIndexed: filesInIndex, + filesDiscovered: summary.filesDiscovered, + samplePaths: pickSamplePaths(discovered.map((file) => file.relativePath)) + }); + removeProgress(this.deps.paths.progressFile); + + logger.info('[DONE]', { ...summary }); + return summary; + } + + /** + * Index or delete specific relative paths without walking the whole tree. + * Used by the file watcher. Callers should fall back to `runFullIndex` for large bursts. + */ + async runChangedPaths(relativePaths: string[], deletedPaths: string[] = []): Promise { + const started = Date.now(); + const release = acquireLock(this.deps.paths.lockFile); + try { + const { repoRoot, vectorStore } = this.deps; + const previousHashes = await vectorStore.getAllFileHashes(); + const previous = readState(this.deps.paths.stateFile); + const removedHashToPath = new Map(); + const handledRemoved = new Set(); + const summary: IndexSummary = { + filesDiscovered: previous?.filesDiscovered ?? previousHashes.size, + filesIndexed: 0, + filesUnchanged: 0, + filesRenamed: 0, + filesDeleted: 0, + filesSkipped: 0, + chunksEmbedded: 0, + chunksReused: 0, + chunksDeleted: 0, + durationMs: 0 + }; + + const uniqueDeletes = [...new Set(deletedPaths.filter(Boolean))]; + for (const relativePath of uniqueDeletes) { + await vectorStore.deleteByFile(relativePath); + summary.filesDeleted++; + logger.info(`[DELETE] ${relativePath}`); + } + + const uniqueUpserts = [...new Set(relativePaths.filter((path) => path && !uniqueDeletes.includes(path)))]; + for (const relativePath of uniqueUpserts) { + const file: DiscoveredFile = { + relativePath, + absolutePath: join(repoRoot, ...relativePath.split('/')) + }; + try { + await this.processDiscoveredFile(file, previousHashes, removedHashToPath, handledRemoved, summary); + } catch (error) { + if (!isEmbeddingInputTooLargeError(error)) throw error; + summary.filesSkipped++; + logger.warn(`[SKIP] ${relativePath} exceeds the embedding model context window after chunking`); + } + } + + summary.durationMs = Date.now() - started; + const newFileCount = uniqueUpserts.filter((path) => !previousHashes.has(path)).length; + const deletedIndexedCount = uniqueDeletes.filter((path) => previousHashes.has(path)).length; + const netFiles = (previous?.filesIndexed ?? previousHashes.size) + newFileCount - deletedIndexedCount; + const filesDiscovered = Math.max( + 0, + (previous?.filesDiscovered ?? previousHashes.size) + newFileCount - deletedIndexedCount + ); + const remaining = [...previousHashes.keys(), ...uniqueUpserts].filter((path) => !uniqueDeletes.includes(path)); + const keptSample = previous?.samplePaths?.filter((path) => !uniqueDeletes.includes(path)) ?? []; + const samplePaths = keptSample.length > 0 ? keptSample : pickSamplePaths(remaining); + await this.persistState(summary, { + filesIndexed: Math.max(0, netFiles), + filesDiscovered, + samplePaths + }); + return summary; + } finally { + release(); + } + } + + private async persistState( + summary: IndexSummary, + counts: { filesIndexed: number; filesDiscovered: number; samplePaths: string[] } + ): Promise { + const { repoRoot, vectorStore } = this.deps; const lastIndexedAt = new Date().toISOString(); const embeddingModel = this.deps.embeddingProvider.modelName(); const embeddingDimensions = await this.deps.embeddingProvider.dimensions(); const chunksIndexed = await vectorStore.countRows(); - - writeState(this.deps.paths.stateFile, { + const sampleFingerprint = computeSampleFingerprint(repoRoot, counts.samplePaths) ?? undefined; + const state: IndexState = { lastIndexedAt, lastDurationMs: summary.durationMs, - filesIndexed: filesInIndex, + filesIndexed: counts.filesIndexed, chunksIndexed, embeddingModel, embeddingDimensions, repoRoot, repoName: basename(repoRoot), - filesDiscovered: summary.filesDiscovered - }); - + filesDiscovered: counts.filesDiscovered, + samplePaths: counts.samplePaths, + sampleFingerprint + }; + writeState(this.deps.paths.stateFile, state); upsertRegistryEntry( this.deps.config.database.path, registryEntryFrom(repoRoot, this.deps.repoId, { lastIndexedAt, - filesIndexed: filesInIndex, + filesIndexed: counts.filesIndexed, chunksIndexed, embeddingModel }) ); - removeProgress(this.deps.paths.progressFile); - - logger.info('[DONE]', { ...summary }); - return summary; } private async processDiscoveredFile( @@ -222,8 +308,12 @@ export class Indexer { if (previousHash !== undefined) { if (previousHash === fileHash) { - summary.filesUnchanged++; - return 'unchanged'; + const existing = await this.deps.vectorStore.getChunksForFile(file.relativePath); + const missingMetadata = existing.some((chunk) => !chunk.extraMetadata); + if (!missingMetadata) { + summary.filesUnchanged++; + return 'unchanged'; + } } return 'index'; } @@ -264,14 +354,23 @@ export class Indexer { const records: ChunkRecord[] = []; const pendingEmbedIndexes: number[] = []; const pendingEmbedTexts: string[] = []; + const occurrences = new Map(); chunks.forEach((chunk, index) => { - const id = computeChunkId( - repoId, - file.relativePath, + const identity: string[] = [ chunk.parentSymbol ?? '', chunk.symbolType ?? 'text', chunk.symbolName ?? `#${index}` + ]; + // Same-named siblings (overloads, repeated headings) share a symbol identity. LanceDB rejects a + // merge batch holding two rows with one key, so repeats are suffixed; the first keeps the bare + // id to stay reusable against indexes built before this. + const repeat = occurrences.get(identity.join(':')) ?? 0; + occurrences.set(identity.join(':'), repeat + 1); + const id = computeChunkId( + repoId, + file.relativePath, + ...(repeat === 0 ? identity : [...identity, `@${repeat}`]) ); newIds.add(id); const contentHash = hashChunkContent(chunk.content); @@ -295,7 +394,9 @@ export class Indexer { embedding: reused && prior ? prior.embedding : [], last_indexed_at: now, git_commit: null, - extra_metadata: null + extra_metadata: serializeChunkExtraMetadata( + buildChunkExtraMetadata(file.relativePath, content, chunk.content, language) + ) }); if (!reused) { diff --git a/src/indexer/freshness.ts b/src/indexer/freshness.ts new file mode 100644 index 0000000..68eb9f1 --- /dev/null +++ b/src/indexer/freshness.ts @@ -0,0 +1,40 @@ +import { existsSync, readFileSync } from 'node:fs'; +import { join } from 'node:path'; +import { hashFileContent, sha256Hex } from '../hashing/hash.js'; +import type { IndexState } from './state.js'; + +export const FRESHNESS_SAMPLE_SIZE = 32; + +/** Evenly spaced, sorted paths so status can re-hash a stable subset without reading the whole tree. */ +export function pickSamplePaths(relativePaths: string[], size = FRESHNESS_SAMPLE_SIZE): string[] { + const unique = [...new Set(relativePaths)].sort(); + if (unique.length <= size) return unique; + const sampled: string[] = []; + for (let i = 0; i < size; i++) { + const index = Math.floor((i * unique.length) / size); + const path = unique[index]; + if (path) sampled.push(path); + } + return [...new Set(sampled)]; +} + +export function computeSampleFingerprint(repoRoot: string, samplePaths: string[]): string | null { + const lines: string[] = []; + for (const relativePath of samplePaths) { + const absolutePath = join(repoRoot, relativePath); + if (!existsSync(absolutePath)) return null; + try { + lines.push(`${relativePath}=${hashFileContent(readFileSync(absolutePath))}`); + } catch { + return null; + } + } + return sha256Hex(lines.join('\n')); +} + +/** True when the stored sample no longer matches disk. Missing sample data is not treated as stale. */ +export function isContentSampleStale(repoRoot: string, state: Pick): boolean { + if (!state.samplePaths?.length || !state.sampleFingerprint) return false; + const next = computeSampleFingerprint(repoRoot, state.samplePaths); + return next !== state.sampleFingerprint; +} diff --git a/src/indexer/state.ts b/src/indexer/state.ts index b51b136..333017e 100644 --- a/src/indexer/state.ts +++ b/src/indexer/state.ts @@ -14,6 +14,9 @@ export interface IndexState { repoName?: string; /** Discoverable file count at last index (used for stale detection, including skipped files). */ filesDiscovered?: number; + /** Stable subset of relative paths hashed to detect edits that do not change file count. */ + samplePaths?: string[]; + sampleFingerprint?: string; } export interface IndexProgress { diff --git a/src/indexer/status.ts b/src/indexer/status.ts index 04a2eba..de633c4 100644 --- a/src/indexer/status.ts +++ b/src/indexer/status.ts @@ -5,8 +5,10 @@ import { resolveRepoPaths } from '../config/paths.js'; import { discoverFiles } from '../discovery/discover.js'; import { getDirectorySizeBytes } from '../utils/dirSize.js'; import { computeRepoId } from '../utils/repo-id.js'; +import { isContentSampleStale } from './freshness.js'; import { indexedChildrenOf, listIndexedRepos, type RegistryEntry } from './registry.js'; import { readProgress, readState, type IndexProgress } from './state.js'; +import { readWatchStatus, type WatchStatus } from './watchStatus.js'; export const INDEX_HINT = 'Not indexed yet — run `code-intel setup --repo ` (or `code-intel index --repo `).'; @@ -24,6 +26,7 @@ export interface IndexStatus { indexLocation: string | null; databaseBytes: number | null; progress?: IndexProgress & { active: boolean }; + watch?: WatchStatus & { skipped?: boolean }; message?: string; indexedChildren?: RegistryEntry[]; } @@ -93,9 +96,11 @@ export async function getIndexStatus(repoRoot: string): Promise { extraIgnorePatterns: config.ignore }); filesDiscoverable = discovered.length; - stale = discovered.length !== (state.filesDiscovered ?? state.filesIndexed); + const countStale = discovered.length !== (state.filesDiscovered ?? state.filesIndexed); + stale = countStale || isContentSampleStale(repoRoot, state); } + const watch = readWatchStatus(paths.watchStatusFile); return { repoRoot, repoId, @@ -109,7 +114,8 @@ export async function getIndexStatus(repoRoot: string): Promise { embeddingModel: state.embeddingModel, indexLocation: paths.indexDir, databaseBytes: getDirectorySizeBytes(paths.dbDir), - progress + progress, + watch: watch ? { ...watch, skipped: Boolean(watch.lastError) } : undefined }; } diff --git a/src/indexer/watch.ts b/src/indexer/watch.ts index afe39e7..a457e06 100644 --- a/src/indexer/watch.ts +++ b/src/indexer/watch.ts @@ -1,12 +1,18 @@ +import { relative, sep } from 'node:path'; import watcher from '@parcel/watcher'; import { createLogger } from '../utils/logger.js'; import { DEFAULT_IGNORE_PATTERNS } from '../discovery/default-ignore.js'; +import { isIndexableRelativePath } from '../discovery/discover.js'; import type { AppContext } from '../context.js'; import { Indexer, type IndexSummary } from './Indexer.js'; import { IndexerLockedError } from './lock.js'; +import { recordWatchError, recordWatchSuccess } from './watchStatus.js'; const logger = createLogger('watch'); +/** More changed files than this in one burst (e.g. git checkout) fall back to a full incremental scan. */ +export const WATCH_FULL_INDEX_THRESHOLD = 40; + export function watchIgnorePatterns(): string[] { return DEFAULT_IGNORE_PATTERNS.map((pattern) => { if (pattern.endsWith('/')) { @@ -18,6 +24,41 @@ export function watchIgnorePatterns(): string[] { }); } +export interface WatchFsEvent { + type: string; + path: string; +} + +export function relativeWatchPath(repoRoot: string, absolutePath: string): string { + return relative(repoRoot, absolutePath).split(sep).join('/'); +} + +export function planWatchIndex( + repoRoot: string, + events: WatchFsEvent[], + options: { allowSensitiveFiles: boolean; extraIgnorePatterns: string[] } +): { mode: 'full' | 'partial'; upserts: string[]; deletes: string[] } { + const upserts = new Set(); + const deletes = new Set(); + for (const event of events) { + const relativePath = relativeWatchPath(repoRoot, event.path); + if (!relativePath || relativePath.startsWith('..')) continue; + if (event.type === 'delete') { + deletes.add(relativePath); + upserts.delete(relativePath); + continue; + } + if (!isIndexableRelativePath(repoRoot, relativePath, options)) continue; + deletes.delete(relativePath); + upserts.add(relativePath); + } + const total = upserts.size + deletes.size; + if (total > WATCH_FULL_INDEX_THRESHOLD) { + return { mode: 'full', upserts: [...upserts], deletes: [...deletes] }; + } + return { mode: 'partial', upserts: [...upserts], deletes: [...deletes] }; +} + /** * Watch the working tree and re-run an incremental index after a quiet period. * The PID lock is held only during a reindex so MCP readers stay unblocked. @@ -37,37 +78,51 @@ export async function watchRepo( let timer: ReturnType | undefined; let running = false; - let pending = false; + let pendingFull = Boolean(options.immediate); + const queued: WatchFsEvent[] = []; const debounceMs = context.config.indexing.debounceMs; + const discoveryOptions = { + allowSensitiveFiles: context.config.security.allowSensitiveFiles, + extraIgnorePatterns: context.config.ignore + }; const run = async (): Promise => { - if (running) { - pending = true; - return; - } + if (running) return; running = true; try { - const summary = await indexer.runFullIndex(); + const batch = queued.splice(0, queued.length); + const forceFull = pendingFull; + pendingFull = false; + const plan = forceFull + ? { mode: 'full' as const, upserts: [] as string[], deletes: [] as string[] } + : planWatchIndex(context.repoRoot, batch, discoveryOptions); + if (!forceFull && plan.mode === 'partial' && plan.upserts.length === 0 && plan.deletes.length === 0) { + return; + } + const summary = + plan.mode === 'full' || forceFull + ? await indexer.runFullIndex() + : await indexer.runChangedPaths(plan.upserts, plan.deletes); + recordWatchSuccess(context.paths.watchStatusFile); options.onIndex?.(summary); logger.info('[WATCH] incremental index complete', { + mode: forceFull || plan.mode === 'full' ? 'full' : 'partial', filesIndexed: summary.filesIndexed, chunksEmbedded: summary.chunksEmbedded, durationMs: summary.durationMs }); } catch (error) { + const message = error instanceof Error ? error.message : String(error); if (error instanceof IndexerLockedError) { logger.warn('[WATCH] skipped — another indexer holds the lock'); + recordWatchError(context.paths.watchStatusFile, `skipped — ${message}`); } else { - logger.warn('[WATCH] incremental index failed', { - error: error instanceof Error ? error.message : String(error) - }); + logger.warn('[WATCH] incremental index failed', { error: message }); + recordWatchError(context.paths.watchStatusFile, message); } } finally { running = false; - if (pending) { - pending = false; - await run(); - } + if (pendingFull || queued.length > 0) await run(); } }; @@ -83,9 +138,12 @@ export async function watchRepo( (error, events) => { if (error) { logger.warn('[WATCH] filesystem error', { error: error.message }); + recordWatchError(context.paths.watchStatusFile, error.message); return; } if (events.length === 0) return; + queued.push(...events); + if (events.length > WATCH_FULL_INDEX_THRESHOLD) pendingFull = true; logger.info(`[WATCH] ${events.length} change(s) — indexing in ${debounceMs}ms`); schedule(); }, diff --git a/src/indexer/watchStatus.ts b/src/indexer/watchStatus.ts new file mode 100644 index 0000000..0b81171 --- /dev/null +++ b/src/indexer/watchStatus.ts @@ -0,0 +1,37 @@ +import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs'; +import { dirname } from 'node:path'; + +export interface WatchStatus { + lastError?: string; + lastErrorAt?: string; + lastSuccessAt?: string; +} + +export function readWatchStatus(path: string): WatchStatus | null { + if (!existsSync(path)) return null; + try { + return JSON.parse(readFileSync(path, 'utf-8')) as WatchStatus; + } catch { + return null; + } +} + +export function recordWatchError(path: string, message: string, now = new Date()): void { + const previous = readWatchStatus(path) ?? {}; + writeWatchStatus(path, { + ...previous, + lastError: message, + lastErrorAt: now.toISOString() + }); +} + +export function recordWatchSuccess(path: string, now = new Date()): void { + writeWatchStatus(path, { + lastSuccessAt: now.toISOString() + }); +} + +function writeWatchStatus(path: string, status: WatchStatus): void { + mkdirSync(dirname(path), { recursive: true }); + writeFileSync(path, JSON.stringify(status, null, 2)); +} diff --git a/src/mcp/server.ts b/src/mcp/server.ts index f0a92e8..bee8603 100644 --- a/src/mcp/server.ts +++ b/src/mcp/server.ts @@ -6,10 +6,14 @@ import { searchSymbol } from '../search/searchSymbol.js'; import { findReferences } from '../search/findReferences.js'; import { getFileContext } from '../search/getFileContext.js'; import { getRepoContext } from '../search/getRepoContext.js'; +import { getTaskContext } from '../retrieval/taskContext.js'; +import { grantFilesystemFallback } from '../retrieval/fallback.js'; import { MCP_SERVER_INSTRUCTIONS } from '../cursor/mcpInstructions.js'; import { createMcpRuntime, type McpRuntime, type ResolveOk } from './runtime.js'; import { startWorkspaceWatchers } from './watchOnStart.js'; import { recordMcpRetrieval } from '../usage/record.js'; +import { getIndexStatus } from '../indexer/status.js'; +import { recoverySteps } from '../cli/formatCliFailure.js'; function textResult(value: unknown) { return { content: [{ type: 'text' as const, text: JSON.stringify(value, null, 2) }] }; @@ -28,17 +32,34 @@ async function withRepo( ) { const started = Date.now(); const resolved = await runtime.resolve(repo); - if (!resolved.ok) return textResult(resolved); - const value = await fn(resolved.context); - if (usageTool) { - recordMcpRetrieval({ - tool: usageTool, - repo: resolved.context.repoRoot, - payloadText: JSON.stringify(value), - latencyMs: Date.now() - started + if (!resolved.ok) { + if (runtime.config.retrieval.allowFallbackAfterFailedRetrieval) { + grantFilesystemFallback(runtime.config.database.path, resolved.message, resolved.repo); + } + return textResult({ + ...resolved, + recovery: recoverySteps(new Error(resolved.message)) + }); + } + try { + const value = await fn(resolved.context); + if (usageTool) { + recordMcpRetrieval({ + tool: usageTool, + repo: resolved.context.repoRoot, + payloadText: JSON.stringify(value), + latencyMs: Date.now() - started + }); + } + return textResult(value); + } catch (error) { + const message = error instanceof Error ? error.message : String(error); + return textResult({ + ok: false, + message, + recovery: recoverySteps(error) }); } - return textResult(value); } /** Exposes the local index to any MCP client (Cursor, VS Code, Claude Code, Codex, ...) — read-only (spec section 20/22). */ @@ -72,7 +93,7 @@ export function buildServer(runtime: McpRuntime): McpServer { 'search_codebase', { description: - 'REQUIRED first tool for code discovery. Uses precomputed local Ollama embeddings in LanceDB; only the query is embedded (corpus is not re-embedded). Returns retrieved snippets — never send vectors to the chat model. Prefer this over Grep/Glob/explore. Pass max_tokens to cap payload size.', + 'Conceptual hybrid search (semantic + keyword + symbol). For a new or broad coding task prefer get_task_context. Uses precomputed local embeddings in LanceDB; only the query is embedded. Pass max_tokens to cap payload size.', inputSchema: z.object({ query: z.string().describe('Natural-language or keyword query'), limit: z.number().int().positive().max(50).optional(), @@ -160,6 +181,36 @@ export function buildServer(runtime: McpRuntime): McpServer { })) ); + server.registerTool( + 'get_task_context', + { + description: + 'Given a coding task, retrieve and assemble the minimum useful repository context (ranked chunks, relationships, token budget, confidence). Prefer this for new or broad tasks.', + inputSchema: z.object({ + task: z.string().describe('Natural-language coding task or question'), + max_tokens: z.number().int().positive().optional(), + mode: z.enum(['minimal', 'normal', 'deep']).optional(), + repo: repoField + }) + }, + async ({ task, max_tokens, mode, repo }) => + withRepo( + runtime, + repo, + async (context) => { + const status = await getIndexStatus(context.repoRoot); + const pkg = await getTaskContext(task, context, { + maxTokens: max_tokens, + mode, + stale: status.stale + }); + const { trace: _trace, ...rest } = pkg; + return { repo: context.repoRoot, ...rest }; + }, + 'get_task_context' + ) + ); + server.registerTool( 'find_references', { diff --git a/src/retrieval/confidence.ts b/src/retrieval/confidence.ts new file mode 100644 index 0000000..902414f --- /dev/null +++ b/src/retrieval/confidence.ts @@ -0,0 +1,54 @@ +import type { RetrievalCandidate, RetrievalConfidence } from './types.js'; + +const DOC_NAME = /(?:^|\/)(?:readme|changelog|contributing|instructions|license)(?:\.[^/]+)?$/i; +const DOC_EXT = /\.(?:md|mdc|markdown|rst|txt)$/i; +const CODE_CHANGE = + /\b(add|fix|implement|update|refactor|change|bug|endpoint|api|function|class|test)\b/i; + +export function isDocHeavyPath(filePath: string): boolean { + const normalized = filePath.replaceAll('\\', '/'); + if (DOC_EXT.test(normalized) || DOC_NAME.test(normalized)) return true; + const lower = normalized.toLowerCase(); + return ( + lower.includes('code_intel_next_phase') || + lower.includes('/examples/cursor/') || + lower.endsWith('.skill.md') || + lower.includes('skill.md') + ); +} + +export function queryLooksLikeCodeChange(query: string): boolean { + return CODE_CHANGE.test(query); +} + +export function confidenceFor( + selected: RetrievalCandidate[], + threshold: number, + options: { stale?: boolean | null; query?: string } = {} +): RetrievalConfidence { + if (options.stale) { + return { score: 0.2, reason: 'Index is stale relative to the working tree.' }; + } + if (selected.length === 0) { + return { score: 0, reason: 'Low-confidence retrieval. Recommended fallback: targeted repository search.' }; + } + const top = selected[0]!; + if ( + options.query && + queryLooksLikeCodeChange(options.query) && + isDocHeavyPath(top.file) + ) { + return { + score: Math.min(top.score.total, Math.max(0, threshold - 0.01)), + reason: + 'Top hit is documentation while the query looks like a code change. Recommended fallback: targeted repository search.' + }; + } + if (top.score.total < threshold) { + return { + score: top.score.total, + reason: 'Low-confidence retrieval. Recommended fallback: targeted repository search.' + }; + } + return { score: top.score.total, reason: 'Top results exceeded the confidence threshold.' }; +} diff --git a/src/retrieval/explain.ts b/src/retrieval/explain.ts new file mode 100644 index 0000000..fb8bd1f --- /dev/null +++ b/src/retrieval/explain.ts @@ -0,0 +1,44 @@ +import type { RetrievalCandidate, RetrievalTrace } from './types.js'; + +export function formatSearchExplain(trace: RetrievalTrace, selected: RetrievalCandidate[]): string { + const lines: string[] = []; + lines.push(`Query: ${trace.query}`); + lines.push(''); + lines.push(`Candidates: ${trace.candidateCount}`); + lines.push(''); + + const top = selected[0]; + if (top) { + lines.push('Top result:'); + lines.push(`${top.file}:${top.startLine}-${top.endLine}`); + lines.push(`score: ${top.score.total}`); + lines.push(` semantic: ${top.score.semantic.toFixed(2)}`); + lines.push(` keyword: ${top.score.keyword.toFixed(2)}`); + lines.push(` symbol: ${top.score.symbol.toFixed(2)}`); + lines.push(` path: ${top.score.path.toFixed(2)}`); + lines.push(` structural: ${top.score.structural.toFixed(2)}`); + lines.push(` test: ${top.score.test.toFixed(2)}`); + lines.push(` recency: ${top.score.recency.toFixed(2)}`); + lines.push(''); + } + + if (trace.discarded.length > 0) { + const byReason = new Map(); + for (const item of trace.discarded) { + byReason.set(item.reason, (byReason.get(item.reason) ?? 0) + 1); + } + lines.push('Discarded:'); + for (const [reason, count] of byReason) { + lines.push(` ${reason}: ${count}`); + } + lines.push(''); + } + + const files = new Set(selected.map((c) => c.file)); + lines.push('Final context:'); + lines.push(`${selected.length} chunks`); + lines.push(`${files.size} files`); + lines.push(`${trace.estimatedTokens.toLocaleString()} estimated tokens`); + lines.push(`${trace.latencyMs}ms`); + return lines.join('\n'); +} diff --git a/src/retrieval/fallback.ts b/src/retrieval/fallback.ts new file mode 100644 index 0000000..abeaf53 --- /dev/null +++ b/src/retrieval/fallback.ts @@ -0,0 +1,42 @@ +import { existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs'; +import { join } from 'node:path'; + +const FALLBACK_FILE = 'fallback-ok.json'; +const WINDOW_MS = 10 * 60 * 1000; + +interface FallbackRecord { + allowedUntil: string; + repo?: string; + reason: string; +} + +export function fallbackStatePath(databasePath: string): string { + return join(databasePath, FALLBACK_FILE); +} + +export function grantFilesystemFallback( + databasePath: string, + reason: string, + repo?: string, + now = Date.now() +): void { + mkdirSync(databasePath, { recursive: true }); + const record: FallbackRecord = { + allowedUntil: new Date(now + WINDOW_MS).toISOString(), + repo, + reason + }; + writeFileSync(fallbackStatePath(databasePath), JSON.stringify(record), 'utf8'); +} + +export function isFilesystemFallbackOpen(databasePath: string, now = Date.now()): boolean { + const path = fallbackStatePath(databasePath); + if (!existsSync(path)) return false; + try { + const record = JSON.parse(readFileSync(path, 'utf8')) as FallbackRecord; + const until = Date.parse(record.allowedUntil); + return Number.isFinite(until) && until > now; + } catch { + return false; + } +} diff --git a/src/retrieval/hybrid.ts b/src/retrieval/hybrid.ts new file mode 100644 index 0000000..c90e720 --- /dev/null +++ b/src/retrieval/hybrid.ts @@ -0,0 +1,117 @@ +import type { EmbeddingProvider } from '../embeddings/EmbeddingProvider.js'; +import { normalizeVector } from '../embeddings/vectorMath.js'; +import type { SearchConfig } from '../config/types.js'; +import type { LanceVectorStore } from '../vector-store/LanceVectorStore.js'; +import type { ChunkSearchResult } from '../vector-store/schema.js'; +import { candidateFromRecord } from './score.js'; +import { mergeCandidates, selectCandidates, type SelectResult } from './select.js'; +import type { RetrievalCandidate, RetrievalTrace } from './types.js'; + +const embedCache = new Map(); +const EMBED_CACHE_LIMIT = 64; + +export interface HybridSearchOptions { + limit?: number; + minScore?: number; + maxTokens?: number; + maxChunksPerFile?: number; + maxChunksPerSymbol?: number; +} + +export interface HybridSearchResult { + selected: RetrievalCandidate[]; + allCandidates: RetrievalCandidate[]; + discarded: SelectResult['discarded']; + estimatedTokens: number; + latencyMs: number; +} + +async function embedQuery(query: string, embeddingProvider: EmbeddingProvider): Promise { + const key = `${embeddingProvider.modelName()}:${query}`; + const cached = embedCache.get(key); + if (cached) return cached; + const vector = normalizeVector(await embeddingProvider.embed(query)); + embedCache.set(key, vector); + if (embedCache.size > EMBED_CACHE_LIMIT) { + const first = embedCache.keys().next().value; + if (first !== undefined) embedCache.delete(first); + } + return vector; +} + +export async function hybridSearch( + query: string, + vectorStore: LanceVectorStore, + embeddingProvider: EmbeddingProvider, + weights: SearchConfig, + options: HybridSearchOptions = {} +): Promise { + const started = Date.now(); + const limit = options.limit ?? weights.defaultLimit; + const fetchLimit = Math.max(limit * 4, 20); + const queryLower = query.toLowerCase(); + + const queryVector = await embedQuery(query, embeddingProvider); + const [vectorResults, keywordResults] = await Promise.all([ + vectorStore.vectorSearch(queryVector, fetchLimit), + vectorStore.fullTextSearch(query, fetchLimit).catch(() => [] as ChunkSearchResult[]) + ]); + + const combined = new Map(); + + for (const record of vectorResults) { + const distance = record._distance ?? 2; + const similarity = Math.min(1, Math.max(0, 1 - distance / 2)); + combined.set( + record.id, + candidateFromRecord( + record, + { semantic: similarity, keyword: 0, sources: ['semantic'], reason: 'semantic similarity' }, + weights, + queryLower + ) + ); + } + + const maxFtsScore = Math.max(...keywordResults.map((r) => r._score ?? 0), 1e-9); + for (const record of keywordResults) { + const normalized = Math.min(1, Math.max(0, (record._score ?? 0) / maxFtsScore)); + const incoming = candidateFromRecord( + record, + { semantic: 0, keyword: normalized, sources: ['keyword'], reason: 'keyword match' }, + weights, + queryLower + ); + const existing = combined.get(record.id); + combined.set(record.id, existing ? mergeCandidates(existing, incoming) : incoming); + } + + const allCandidates = [...combined.values()].sort((a, b) => b.score.total - a.score.total); + const packed = selectCandidates(allCandidates, { + limit, + maxTokens: options.maxTokens, + minScore: options.minScore, + maxChunksPerFile: options.maxChunksPerFile ?? weights.maxChunksPerFile, + maxChunksPerSymbol: options.maxChunksPerSymbol ?? weights.maxChunksPerSymbol + }); + + return { + selected: packed.selected, + allCandidates, + discarded: packed.discarded, + estimatedTokens: packed.estimatedTokens, + latencyMs: Date.now() - started + }; +} + +export function hybridTrace(query: string, result: HybridSearchResult): RetrievalTrace { + return { + query, + candidateCount: result.allCandidates.length, + selectedCount: result.selected.length, + discarded: result.discarded, + estimatedTokens: result.estimatedTokens, + latencyMs: result.latencyMs, + expansionHops: 0 + }; +} diff --git a/src/retrieval/intent.ts b/src/retrieval/intent.ts new file mode 100644 index 0000000..756343f --- /dev/null +++ b/src/retrieval/intent.ts @@ -0,0 +1,118 @@ +import type { ContextMode } from './types.js'; + +export type RetrievalOperation = + | 'find_implementation' + | 'find_tests' + | 'find_config' + | 'find_references' + | 'find_dependencies'; + +export interface RetrievalIntent { + rawQuery: string; + concepts: string[]; + symbols: string[]; + files: string[]; + likelyLanguages: string[]; + operations: RetrievalOperation[]; + requestedContext: ContextMode; +} + +const FILE_RE = /(?:[\w./-]+\.(?:ts|tsx|js|jsx|py|go|rs|java|kt|rb|php|cs|c|h|cpp|swift|md|json|ya?ml|toml))/gi; +const QUOTED = /['"`]([A-Za-z_][\w./-]*)['"`]/g; +const IDENT = /\b([A-Z][A-Za-z0-9]+|[a-z][a-zA-Z0-9]{2,}|[a-z]+(?:_[a-z0-9]+)+)\b/g; +const EXT_TO_LANG: Record = { + ts: 'typescript', + tsx: 'tsx', + js: 'javascript', + jsx: 'javascript', + py: 'python', + go: 'go', + rs: 'rust', + java: 'java', + md: 'markdown', + json: 'json', + yml: 'yaml', + yaml: 'yaml' +}; + +const STOP = new Set([ + 'the', + 'and', + 'for', + 'that', + 'with', + 'this', + 'from', + 'into', + 'when', + 'where', + 'what', + 'how', + 'does', + 'add', + 'fix', + 'update', + 'remove', + 'refactor', + 'find', + 'tests', + 'test', + 'config', + 'please', + 'should', + 'could', + 'would' +]); + +function unique(values: string[]): string[] { + return [...new Set(values.filter(Boolean))]; +} + +export function analyzeQuery(rawQuery: string, mode?: ContextMode): RetrievalIntent { + const files = unique([...rawQuery.matchAll(FILE_RE)].map((m) => m[0])); + const quoted = unique([...rawQuery.matchAll(QUOTED)].map((m) => m[1] ?? '')); + const identifiers = unique( + [...rawQuery.matchAll(IDENT)] + .map((m) => m[1] ?? '') + .filter((token) => !STOP.has(token.toLowerCase())) + ); + + const symbols = unique([ + ...quoted.filter((t) => /^[A-Za-z_][\w]*$/.test(t)), + ...identifiers.filter((t) => t.includes('_') || /[a-z0-9][A-Z]/.test(t)) + ]); + + const lower = rawQuery.toLowerCase(); + const operations: RetrievalOperation[] = ['find_implementation']; + if (/\b(test|tests|spec|coverage)\b/.test(lower)) operations.push('find_tests'); + if (/\b(config|configured|configuration|env|dockerfile|yaml|toml)\b/.test(lower)) { + operations.push('find_config'); + } + if (/\b(reference|references|callers?|usages?|used by)\b/.test(lower)) operations.push('find_references'); + if (/\b(dependenc|import|caller|callee)\b/.test(lower)) operations.push('find_dependencies'); + + let requestedContext: ContextMode = mode ?? 'normal'; + if (!mode) { + if (/\b(architect|cross-cutting|unfamiliar|deep)\b/.test(lower)) requestedContext = 'deep'; + } + + const likelyLanguages = unique( + files.map((file) => EXT_TO_LANG[file.split('.').pop()?.toLowerCase() ?? ''] ?? '').filter(Boolean) + ); + + const concepts = unique( + identifiers + .map((t) => t.toLowerCase()) + .filter((t) => t.length >= 4 && !symbols.map((s) => s.toLowerCase()).includes(t)) + ); + + return { + rawQuery, + concepts, + symbols, + files, + likelyLanguages, + operations, + requestedContext + }; +} diff --git a/src/retrieval/resolveImport.ts b/src/retrieval/resolveImport.ts new file mode 100644 index 0000000..a2812b6 --- /dev/null +++ b/src/retrieval/resolveImport.ts @@ -0,0 +1,111 @@ +import { existsSync, readFileSync } from 'node:fs'; +import { join, posix } from 'node:path'; + +const SOURCE_EXTS = ['.ts', '.tsx', '.js', '.jsx', '.mts', '.cts', '.mjs', '.cjs']; + +export interface PathAlias { + prefix: string; + targets: string[]; +} + +function withSourceExtensions(base: string): string[] { + const stripped = base.replace(/\.(js|jsx|ts|tsx|mjs|cjs|mts|cts)$/i, ''); + const paths = new Set([base, stripped]); + for (const ext of SOURCE_EXTS) paths.add(stripped + ext); + paths.add(posix.join(stripped, 'index.ts')); + paths.add(posix.join(stripped, 'index.tsx')); + paths.add(posix.join(stripped, 'index.js')); + return [...paths].filter((path) => !path.startsWith('../') && path !== '..' && !path.startsWith('/')); +} + +function pythonModuleFiles(modulePath: string): string[] { + const normalized = modulePath.replace(/^\//, ''); + if (!normalized || normalized.startsWith('..')) return []; + return [`${normalized}.py`, posix.join(normalized, '__init__.py')]; +} + +/** Load `compilerOptions.paths` from tsconfig/jsconfig at the repo root. */ +export function loadTsPathAliases(repoRoot: string): PathAlias[] { + for (const name of ['tsconfig.json', 'jsconfig.json']) { + const file = join(repoRoot, name); + if (!existsSync(file)) continue; + try { + const raw = readFileSync(file, 'utf-8').replace(/\/\*[\s\S]*?\*\//g, '').replace(/^\s*\/\/.*$/gm, ''); + const parsed = JSON.parse(raw) as { + compilerOptions?: { baseUrl?: string; paths?: Record }; + }; + const paths = parsed.compilerOptions?.paths; + if (!paths) continue; + const baseUrl = (parsed.compilerOptions?.baseUrl ?? '.').replace(/\\/g, '/').replace(/\/$/, ''); + const aliases: PathAlias[] = []; + for (const [pattern, targets] of Object.entries(paths)) { + const prefix = pattern.replace(/\*$/, '').replace(/\\/g, '/'); + aliases.push({ + prefix, + targets: targets.map((target) => { + const mapped = target.replace(/\*$/, '').replace(/\\/g, '/'); + return posix.normalize(baseUrl === '.' ? mapped : posix.join(baseUrl, mapped)); + }) + }); + } + return aliases.sort((a, b) => b.prefix.length - a.prefix.length); + } catch { + continue; + } + } + return []; +} + +/** + * Turn an import specifier into candidate repo-relative file paths. + * Handles relative JS/TS, tsconfig path aliases, and Python dotted modules. + */ +export function candidateImportPaths( + fromFile: string, + spec: string, + aliases: PathAlias[] = [] +): string[] { + const trimmed = spec.trim().replace(/['"]/g, ''); + if (!trimmed) return []; + const from = fromFile.replaceAll('\\', '/'); + const python = from.endsWith('.py'); + + if (trimmed.startsWith('.')) { + const dir = posix.dirname(from); + if (python) { + const dots = trimmed.match(/^(\.+)/)?.[1]?.length ?? 1; + const rest = trimmed.slice(dots).replaceAll('.', '/'); + let base = dir; + for (let i = 0; i < dots - 1; i++) base = posix.dirname(base); + const joined = rest ? posix.normalize(posix.join(base, rest)) : base; + if (joined.startsWith('../') || joined === '..' || joined.startsWith('/')) return []; + return pythonModuleFiles(joined); + } + const joined = posix.normalize(posix.join(dir, trimmed)); + if (joined.startsWith('../') || joined === '..' || joined.startsWith('/')) return []; + return withSourceExtensions(joined); + } + + for (const alias of aliases) { + if (trimmed === alias.prefix || trimmed.startsWith(alias.prefix)) { + const rest = trimmed.slice(alias.prefix.length); + const mapped: string[] = []; + for (const target of alias.targets) { + mapped.push(...withSourceExtensions(posix.normalize(posix.join(target, rest)))); + } + if (mapped.length > 0) return mapped; + } + } + + if (python && trimmed.includes('.')) { + return pythonModuleFiles(trimmed.replaceAll('.', '/')); + } + + return []; +} + +export function importPathSqlList(fromFile: string, spec: string, aliases: PathAlias[] = []): string { + return candidateImportPaths(fromFile, spec, aliases) + .map((path) => `'${path.replace(/'/g, "''")}'`) + .join(', '); +} diff --git a/src/retrieval/score.ts b/src/retrieval/score.ts new file mode 100644 index 0000000..5f41a17 --- /dev/null +++ b/src/retrieval/score.ts @@ -0,0 +1,178 @@ +import type { SearchConfig } from '../config/types.js'; +import { isConfigPath, isTestPath, parseChunkExtraMetadata } from '../chunker/chunkMetadata.js'; +import { estimateTokensFromChars } from '../utils/tokens.js'; +import type { ChunkSearchResult } from '../vector-store/schema.js'; +import type { CandidateSource, RetrievalCandidate, RetrievalScore } from './types.js'; + +export function clamp01(value: number): number { + return Math.min(1, Math.max(0, value)); +} + +export function qualifiedSymbolName( + symbolName: string | null | undefined, + parentSymbol: string | null | undefined +): string | null { + if (!symbolName) return null; + return parentSymbol ? `${parentSymbol}.${symbolName}` : symbolName; +} + +const PATH_QUERY_STOP = new Set([ + 'add', + 'and', + 'does', + 'find', + 'fix', + 'how', + 'implemented', + 'implementation', + 'into', + 'the', + 'update', + 'where', + 'with' +]); + +function identifierParts(value: string): string[] { + return value + .replace(/([a-z0-9])([A-Z])/g, '$1 $2') + .toLowerCase() + .split(/[^a-z0-9]+/) + .filter((token) => token.length >= 3 && !PATH_QUERY_STOP.has(token)); +} + +export function basenameMatchScore(filePath: string, queryLower: string): number { + const base = filePath.replaceAll('\\', '/').split('/').pop()?.replace(/\.[^.]+$/, '') ?? ''; + const normalizedBase = base.toLowerCase().replace(/[^a-z0-9]/g, ''); + if (normalizedBase.length < 4) return 0; + const compactQuery = queryLower.toLowerCase().replace(/[^a-z0-9]/g, ''); + return compactQuery.includes(normalizedBase) ? 1 : 0; +} + +export function pathScore(filePath: string, queryLower: string): number { + const path = filePath.toLowerCase().replaceAll('\\', '/'); + if (basenameMatchScore(filePath, queryLower) === 1) return 1; + const tokens = identifierParts(queryLower); + if (tokens.length === 0) return 0; + let hits = 0; + for (const token of tokens) { + if (path.includes(token)) hits += 1; + } + return clamp01(hits / tokens.length); +} + +export function structuralScore(record: { + symbol_name: string | null; + parent_symbol: string | null; +}): number { + if (record.symbol_name && record.parent_symbol) return 1; + if (record.symbol_name) return 0.7; + if (record.parent_symbol) return 0.4; + return 0; +} + +export function recencyScore(lastIndexedAt: string | null, now = Date.now()): number { + if (!lastIndexedAt) return 0; + const then = Date.parse(lastIndexedAt); + if (!Number.isFinite(then)) return 0; + const ageMs = Math.max(0, now - then); + const week = 7 * 24 * 60 * 60 * 1000; + return clamp01(1 - ageMs / week); +} + +export function symbolMatchScore( + record: { symbol_name: string | null; parent_symbol: string | null }, + queryLower: string +): number { + const name = record.symbol_name?.toLowerCase(); + if (name && queryLower.includes(name)) return 1; + const parent = record.parent_symbol?.toLowerCase(); + if (parent && queryLower.includes(parent)) return 0.5; + return 0; +} + +export function combineScore(parts: Omit, weights: SearchConfig): number { + return clamp01( + weights.vectorWeight * parts.semantic + + weights.keywordWeight * parts.keyword + + weights.symbolWeight * parts.symbol + + weights.pathWeight * parts.path + + weights.structuralWeight * parts.structural + + weights.dependencyWeight * parts.dependency + + weights.referenceWeight * parts.reference + + weights.testWeight * parts.test + + weights.recencyWeight * parts.recency + ); +} + +export function buildRetrievalScore( + parts: Omit, + weights: SearchConfig +): RetrievalScore { + const total = Math.round(combineScore(parts, weights) * 1000) / 1000; + return { total, ...parts }; +} + +export function candidateFromRecord( + record: ChunkSearchResult, + parts: { + semantic: number; + keyword: number; + symbol?: number; + path?: number; + dependency?: number; + reference?: number; + exactIdentifier?: boolean; + sources: CandidateSource[]; + reason: string; + }, + weights: SearchConfig, + queryLower: string +): RetrievalCandidate { + const extra = parseChunkExtraMetadata(record.extra_metadata); + const isTest = extra.isTest || isTestPath(record.file_path); + const isConfig = extra.isConfig || isConfigPath(record.file_path); + extra.isTest = isTest; + extra.isConfig = isConfig; + const symbol = parts.symbol ?? symbolMatchScore(record, queryLower); + const path = clamp01(parts.path ?? pathScore(record.file_path, queryLower)); + const wantsTests = /\b(test|tests|spec|coverage)\b/i.test(queryLower); + const scoreParts = { + semantic: clamp01(parts.semantic), + keyword: clamp01(parts.keyword), + symbol, + path, + structural: structuralScore(record), + dependency: clamp01(parts.dependency ?? 0), + reference: clamp01(parts.reference ?? 0), + test: isTest && wantsTests ? 1 : 0, + recency: recencyScore(record.last_indexed_at) + }; + const score = buildRetrievalScore(scoreParts, weights); + const exactSymbol = symbol === 1; + const exactBasename = parts.path === 1 || basenameMatchScore(record.file_path, queryLower) === 1; + if (exactSymbol) score.total = Math.max(score.total, 0.95); + else if (parts.exactIdentifier) score.total = Math.max(score.total, 0.82 + 0.13 * scoreParts.keyword); + else if (exactBasename) score.total = Math.max(score.total, 0.9); + return { + id: record.id, + file: record.file_path, + symbol: qualifiedSymbolName(record.symbol_name, record.parent_symbol), + symbolType: record.symbol_type, + parentSymbol: record.parent_symbol, + startLine: record.start_line, + endLine: record.end_line, + content: record.content, + lastIndexedAt: record.last_indexed_at, + extra, + sources: parts.sources, + score, + estimatedTokens: estimateTokensFromChars(record.content.length), + reason: exactSymbol + ? `exact symbol match: ${record.symbol_name}` + : parts.exactIdentifier + ? parts.reason + : exactBasename + ? `exact file match: ${record.file_path}` + : parts.reason + }; +} diff --git a/src/retrieval/select.ts b/src/retrieval/select.ts new file mode 100644 index 0000000..2c74dbb --- /dev/null +++ b/src/retrieval/select.ts @@ -0,0 +1,112 @@ +import type { RetrievalCandidate } from './types.js'; + +export interface SelectOptions { + limit: number; + maxTokens?: number; + minScore?: number; + maxChunksPerFile: number; + maxChunksPerSymbol: number; + /** Prevent relationship expansion from crowding direct hits out of the first file slots. */ + maxExpansionOnlyFilesInTopK?: number; + expansionTopK?: number; +} + +export interface SelectResult { + selected: RetrievalCandidate[]; + discarded: Array<{ id: string; file: string; reason: string }>; + estimatedTokens: number; +} + +function symbolKey(candidate: RetrievalCandidate): string { + return (candidate.symbol ?? `${candidate.file}:${candidate.startLine}`).toLowerCase(); +} + +/** + * Ranked greedy pack: keep the highest-value remaining candidate that fits + * diversity caps and the token budget. Always keeps the first accepted chunk + * even if it exceeds maxTokens so the caller is never empty when candidates exist. + */ +export function selectCandidates( + ranked: RetrievalCandidate[], + options: SelectOptions +): SelectResult { + const perFile = new Map(); + const perSymbol = new Map(); + const selected: RetrievalCandidate[] = []; + const discarded: SelectResult['discarded'] = []; + const expansionOnlyFiles = new Set(); + let estimatedTokens = 0; + const minScore = options.minScore ?? 0; + + for (const candidate of ranked) { + if (selected.length >= options.limit) { + discarded.push({ id: candidate.id, file: candidate.file, reason: 'limit' }); + continue; + } + if (candidate.score.total < minScore) { + discarded.push({ id: candidate.id, file: candidate.file, reason: 'min_score' }); + continue; + } + const fileCount = perFile.get(candidate.file) ?? 0; + if (fileCount >= options.maxChunksPerFile) { + discarded.push({ id: candidate.id, file: candidate.file, reason: 'per_file_cap' }); + continue; + } + const key = symbolKey(candidate); + const symbolCount = perSymbol.get(key) ?? 0; + if (symbolCount >= options.maxChunksPerSymbol) { + discarded.push({ id: candidate.id, file: candidate.file, reason: 'per_symbol_cap' }); + continue; + } + const expansionOnly = + candidate.sources.length > 0 && candidate.sources.every((source) => source === 'expansion'); + const expansionTopK = options.expansionTopK ?? 5; + const maxExpansionFiles = options.maxExpansionOnlyFilesInTopK; + if ( + expansionOnly && + selected.length < expansionTopK && + maxExpansionFiles !== undefined && + !expansionOnlyFiles.has(candidate.file) && + expansionOnlyFiles.size >= maxExpansionFiles + ) { + discarded.push({ id: candidate.id, file: candidate.file, reason: 'expansion_top_k_cap' }); + continue; + } + if ( + options.maxTokens && + selected.length > 0 && + estimatedTokens + candidate.estimatedTokens > options.maxTokens + ) { + discarded.push({ id: candidate.id, file: candidate.file, reason: 'token_budget' }); + continue; + } + + selected.push(candidate); + estimatedTokens += candidate.estimatedTokens; + perFile.set(candidate.file, fileCount + 1); + perSymbol.set(key, symbolCount + 1); + if (expansionOnly) expansionOnlyFiles.add(candidate.file); + } + + return { selected, discarded, estimatedTokens }; +} + +export function mergeCandidates(existing: RetrievalCandidate, incoming: RetrievalCandidate): RetrievalCandidate { + const sources = [...new Set([...existing.sources, ...incoming.sources])]; + const score = existing.score.total >= incoming.score.total ? existing.score : incoming.score; + return { + ...existing, + sources, + score, + extra: { + imports: [...new Set([...existing.extra.imports, ...incoming.extra.imports])], + exports: [...new Set([...existing.extra.exports, ...incoming.extra.exports])], + referencedSymbols: [ + ...new Set([...existing.extra.referencedSymbols, ...incoming.extra.referencedSymbols]) + ], + isTest: existing.extra.isTest || incoming.extra.isTest, + isConfig: existing.extra.isConfig || incoming.extra.isConfig + }, + reason: existing.score.total >= incoming.score.total ? existing.reason : incoming.reason + }; +} diff --git a/src/retrieval/taskContext.ts b/src/retrieval/taskContext.ts new file mode 100644 index 0000000..dcdede6 --- /dev/null +++ b/src/retrieval/taskContext.ts @@ -0,0 +1,459 @@ +import { basename } from 'node:path'; +import type { AppContext } from '../context.js'; +import { findReferences } from '../search/findReferences.js'; +import { searchSymbol } from '../search/searchSymbol.js'; +import { estimateTokensFromText } from '../utils/tokens.js'; +import type { ChunkSearchResult } from '../vector-store/schema.js'; +import type { LanceVectorStore } from '../vector-store/LanceVectorStore.js'; +import { grantFilesystemFallback } from './fallback.js'; +import { hybridSearch } from './hybrid.js'; +import { analyzeQuery } from './intent.js'; +import { confidenceFor } from './confidence.js'; +import { candidateImportPaths, loadTsPathAliases } from './resolveImport.js'; +import { candidateFromRecord, pathScore } from './score.js'; +import { mergeCandidates, selectCandidates } from './select.js'; +import type { + ContextMode, + ContextPackage, + RetrievalCandidate, + RetrievalConfidence, + RetrievalTrace +} from './types.js'; + +const MODE_LIMITS: Record = { + minimal: { chunks: 5, tokens: 5000, perFile: 2, perSymbol: 1 }, + normal: { chunks: 8, tokens: 8000, perFile: 2, perSymbol: 2 }, + deep: { chunks: 24, tokens: 25_000, perFile: 6, perSymbol: 4 } +}; + +function escapeSqlString(value: string): string { + return value.replace(/'/g, "''"); +} + +function fileStem(filePath: string): string { + return basename(filePath).replace(/\.[^.]+$/, ''); +} + +const CHUNK_COLUMNS = [ + 'id', + 'file_path', + 'symbol_name', + 'symbol_type', + 'parent_symbol', + 'start_line', + 'end_line', + 'content', + 'last_indexed_at', + 'extra_metadata' +]; + +async function queryFilesLike( + vectorStore: LanceVectorStore, + pattern: string, + limit: number +): Promise { + return vectorStore.queryAll( + CHUNK_COLUMNS, + `file_path LIKE '%${escapeSqlString(pattern)}%'`, + limit + ); +} + +async function queryFilesExact( + vectorStore: LanceVectorStore, + paths: string[], + limit: number +): Promise { + const unique = [...new Set(paths.filter(Boolean))]; + if (unique.length === 0) return []; + const list = unique.map((path) => `'${escapeSqlString(path)}'`).join(', '); + return vectorStore.queryAll(CHUNK_COLUMNS, `file_path IN (${list})`, limit); +} + +async function queryFilesLikeTestStem( + vectorStore: LanceVectorStore, + stem: string, + limit: number +): Promise { + const s = escapeSqlString(stem); + return vectorStore.queryAll( + CHUNK_COLUMNS, + `(file_path LIKE '%/${s}.test.%' OR file_path LIKE '${s}.test.%' OR file_path LIKE '%/${s}.spec.%' OR file_path LIKE '${s}.spec.%')`, + limit + ); +} + +function addCandidate( + map: Map, + incoming: RetrievalCandidate +): void { + const existing = map.get(incoming.id); + map.set(incoming.id, existing ? mergeCandidates(existing, incoming) : incoming); +} + +function packageFromSelected( + query: string, + selected: RetrievalCandidate[], + considered: number, + hops: number, + confidence: RetrievalConfidence, + relationships: ContextPackage['relationships'], + trace?: RetrievalTrace +): ContextPackage { + const byFile = new Map(); + for (const candidate of selected) { + const list = byFile.get(candidate.file) ?? []; + list.push(candidate); + byFile.set(candidate.file, list); + } + + const files = [...byFile.entries()] + .map(([path, chunks]) => { + const rankedChunks = [...chunks].sort((a, b) => b.score.total - a.score.total); + const best = rankedChunks[0]!; + return { + path, + reason: best.reason, + score: best.score, + chunks: rankedChunks.map((chunk) => ({ + symbol: chunk.symbol ?? undefined, + startLine: chunk.startLine, + endLine: chunk.endLine, + content: chunk.content + })) + }; + }) + .sort((a, b) => b.score.total - a.score.total || a.path.localeCompare(b.path)); + + const estimatedTokens = estimateTokensFromText(JSON.stringify({ files: files.map((f) => ({ path: f.path, chunks: f.chunks })) })); + + return { + query, + summary: `Retrieved ${selected.length} chunk(s) across ${files.length} file(s) for: ${query.slice(0, 120)}`, + files, + relationships, + estimatedTokens, + retrievalStats: { + candidatesConsidered: considered, + candidatesSelected: selected.length, + expansionHops: hops + }, + confidence, + trace + }; +} + +export interface TaskContextOptions { + maxTokens?: number; + mode?: ContextMode; + stale?: boolean | null; +} + +export async function getTaskContext( + task: string, + app: AppContext, + options: TaskContextOptions = {} +): Promise { + const intent = analyzeQuery(task, options.mode); + const mode = intent.requestedContext; + const caps = MODE_LIMITS[mode]; + const retrieval = app.config.retrieval; + const maxTokens = options.maxTokens ?? Math.min(retrieval.maxContextTokens, caps.tokens); + const seedLimit = Math.min(retrieval.seedResults, caps.chunks); + const queryLower = task.toLowerCase(); + const started = Date.now(); + const aliases = loadTsPathAliases(app.repoRoot); + + const seed = await hybridSearch(task, app.vectorStore, app.embeddingProvider, app.config.search, { + limit: seedLimit, + maxTokens, + maxChunksPerFile: caps.perFile, + maxChunksPerSymbol: caps.perSymbol + }); + + const merged = new Map(); + for (const candidate of seed.allCandidates) addCandidate(merged, candidate); + + for (const symbol of intent.symbols.slice(0, 8)) { + const matches = await searchSymbol(symbol, app.vectorStore, 8); + let foundExactSymbol = false; + if (matches.length > 0) { + const rows = await app.vectorStore.queryAll( + CHUNK_COLUMNS, + `symbol_name = '${escapeSqlString(symbol)}'`, + 8 + ); + foundExactSymbol = rows.length > 0; + for (const record of rows) { + addCandidate( + merged, + candidateFromRecord( + record, + { semantic: 0, keyword: 0.8, symbol: 1, sources: ['symbol'], reason: `symbol ${symbol}` }, + app.config.search, + queryLower + ) + ); + } + } + + if (!foundExactSymbol) { + const lexical = await app.vectorStore.fullTextSearch(symbol, 12).catch(() => []); + const maxScore = Math.max(...lexical.map((row) => row._score ?? 0), 1e-9); + for (const record of lexical) { + addCandidate( + merged, + candidateFromRecord( + record, + { + semantic: 0, + keyword: Math.min(1, Math.max(0, (record._score ?? 0) / maxScore)), + exactIdentifier: true, + sources: ['keyword'], + reason: `exact identifier occurrence: ${symbol}` + }, + app.config.search, + queryLower + ) + ); + } + } + } + + const pathTerms = intent.files + .map((file) => file.replaceAll('\\', '/').split('/').pop() ?? file) + .slice(0, 8); + if (intent.operations.includes('find_config')) { + pathTerms.push( + ...intent.concepts.filter((concept) => /^(defaults?|configs?|settings?)$/.test(concept)) + ); + } + for (const term of [...new Set(pathTerms)]) { + const rows = await queryFilesLike(app.vectorStore, term, 4); + for (const record of rows) { + addCandidate( + merged, + candidateFromRecord( + record, + { + semantic: 0, + keyword: 0.8, + path: 1, + sources: ['keyword'], + reason: `file path match: ${term}` + }, + app.config.search, + queryLower + ) + ); + } + } + + for (const term of [...new Set(intent.concepts.filter((concept) => concept.length >= 5))].slice(0, 8)) { + const rows = await queryFilesLike(app.vectorStore, term, 6); + for (const record of rows) { + const measuredPathScore = pathScore(record.file_path, queryLower); + if (measuredPathScore < 0.25) continue; + addCandidate( + merged, + candidateFromRecord( + record, + { + semantic: 0, + keyword: 0.6, + path: measuredPathScore, + sources: ['keyword'], + reason: `multi-token file path match: ${record.file_path}` + }, + app.config.search, + queryLower + ) + ); + } + } + + const relationships: NonNullable = []; + let hops = 0; + const maxHops = retrieval.maxExpansionHops; + const rankedSeeds = [...merged.values()].sort((a, b) => b.score.total - a.score.total); + let frontier = selectCandidates(rankedSeeds, { + limit: seedLimit, + maxTokens, + maxChunksPerFile: caps.perFile, + maxChunksPerSymbol: caps.perSymbol + }).selected; + + while (hops < maxHops && frontier.length > 0) { + hops += 1; + const next: RetrievalCandidate[] = []; + for (const seedChunk of frontier) { + const importSpecs = seedChunk.extra.imports.filter((imp) => imp.startsWith('.') || !imp.startsWith('http')); + for (const imp of importSpecs.slice(0, 8)) { + const paths = candidateImportPaths(seedChunk.file, imp, aliases); + if (paths.length === 0) continue; + const rows = await queryFilesExact(app.vectorStore, paths, 8); + for (const record of rows) { + const candidate = candidateFromRecord( + record, + { + semantic: 0, + keyword: 0.2, + dependency: 1, + sources: ['expansion'], + reason: `import of ${imp}` + }, + app.config.search, + queryLower + ); + addCandidate(merged, candidate); + next.push(candidate); + relationships.push({ + from: seedChunk.file, + to: record.file_path, + type: 'imports' + }); + } + } + + if (intent.operations.includes('find_references') && seedChunk.symbol) { + const refs = await findReferences(seedChunk.symbol.split('.').pop() ?? seedChunk.symbol, app.vectorStore, 12); + for (const ref of refs.slice(0, 8)) { + const rows = await queryFilesExact(app.vectorStore, [ref.file], 4); + for (const record of rows) { + const candidate = candidateFromRecord( + record, + { + semantic: 0, + keyword: 0.2, + reference: 1, + sources: ['expansion'], + reason: `reference to ${seedChunk.symbol}` + }, + app.config.search, + queryLower + ); + addCandidate(merged, candidate); + next.push(candidate); + relationships.push({ + from: seedChunk.file, + to: record.file_path, + type: 'references' + }); + } + } + } + + if (intent.operations.includes('find_tests')) { + const stem = fileStem(seedChunk.file); + const rows = await queryFilesLikeTestStem(app.vectorStore, stem, 4); + for (const record of rows) { + const candidate = candidateFromRecord( + record, + { + semantic: 0, + keyword: 0.3, + sources: ['expansion'], + reason: `tests for ${stem}` + }, + app.config.search, + queryLower + ); + addCandidate(merged, candidate); + next.push(candidate); + relationships.push({ from: seedChunk.file, to: record.file_path, type: 'tests' }); + } + } + } + + if (intent.operations.includes('find_config') && hops === 1) { + const configs = await app.vectorStore.queryAll( + CHUNK_COLUMNS, + `extra_metadata LIKE '%"isConfig":true%'`, + 8 + ); + for (const record of configs) { + addCandidate( + merged, + candidateFromRecord( + record, + { semantic: 0, keyword: 0.4, sources: ['expansion'], reason: 'configuration file' }, + app.config.search, + queryLower + ) + ); + } + } + + frontier = next.filter((c) => c.score.total >= retrieval.confidenceThreshold).slice(0, seedLimit); + } + + const all = [...merged.values()].sort((a, b) => b.score.total - a.score.total); + const packed = selectCandidates(all, { + limit: Math.min(retrieval.maxContextChunks, caps.chunks), + maxTokens, + maxChunksPerFile: caps.perFile, + maxChunksPerSymbol: caps.perSymbol, + maxExpansionOnlyFilesInTopK: 3, + expansionTopK: 5 + }); + + const confidence = confidenceFor(packed.selected, retrieval.confidenceThreshold, { + stale: options.stale, + query: task + }); + if ( + app.config.retrieval.allowFallbackAfterFailedRetrieval && + confidence.score < retrieval.confidenceThreshold + ) { + grantFilesystemFallback(app.config.database.path, confidence.reason, app.repoRoot); + } + + const uniqueRels = relationships.filter( + (rel, index, arr) => arr.findIndex((r) => r.from === rel.from && r.to === rel.to && r.type === rel.type) === index + ); + + const trace: RetrievalTrace = { + query: task, + candidateCount: all.length, + selectedCount: packed.selected.length, + discarded: packed.discarded, + estimatedTokens: packed.estimatedTokens, + latencyMs: Date.now() - started, + expansionHops: hops + }; + + return packageFromSelected(task, packed.selected, all.length, hops, confidence, uniqueRels, trace); +} + +export function formatContextPackage(pkg: ContextPackage, explain = false): string { + const lines: string[] = []; + lines.push(pkg.summary); + lines.push(`Estimated tokens: ${pkg.estimatedTokens}`); + lines.push(`Confidence: ${pkg.confidence.score} (${pkg.confidence.reason})`); + lines.push(''); + for (const file of pkg.files) { + lines.push(`${file.path} — ${file.reason} (score ${file.score.total})`); + for (const chunk of file.chunks) { + const symbol = chunk.symbol ? ` ${chunk.symbol}` : ''; + lines.push(` L${chunk.startLine}-${chunk.endLine}${symbol}`); + } + } + if (explain && pkg.trace) { + lines.push(''); + lines.push(`Candidates considered: ${pkg.trace.candidateCount}`); + lines.push(`Expansion hops: ${pkg.trace.expansionHops}`); + lines.push(`Latency: ${pkg.trace.latencyMs}ms`); + } + return lines.join('\n'); +} + +/** Used by tests that only need packaging, not a live index. */ +export function buildContextPackageForTest( + query: string, + selected: RetrievalCandidate[], + considered: number, + hops: number, + confidence: RetrievalConfidence +): ContextPackage { + return packageFromSelected(query, selected, considered, hops, confidence, []); +} diff --git a/src/retrieval/types.ts b/src/retrieval/types.ts new file mode 100644 index 0000000..4351fe4 --- /dev/null +++ b/src/retrieval/types.ts @@ -0,0 +1,79 @@ +import type { ChunkExtraMetadata } from '../chunker/chunkMetadata.js'; + +export interface RetrievalScore { + total: number; + semantic: number; + keyword: number; + symbol: number; + path: number; + structural: number; + dependency: number; + reference: number; + test: number; + recency: number; +} + +export type CandidateSource = 'semantic' | 'keyword' | 'symbol' | 'expansion'; + +export interface RetrievalCandidate { + id: string; + file: string; + symbol: string | null; + symbolType: string | null; + parentSymbol: string | null; + startLine: number; + endLine: number; + content: string; + lastIndexedAt: string | null; + extra: ChunkExtraMetadata; + sources: CandidateSource[]; + score: RetrievalScore; + estimatedTokens: number; + reason: string; +} + +export interface RetrievalTrace { + query: string; + candidateCount: number; + selectedCount: number; + discarded: Array<{ id: string; file: string; reason: string }>; + estimatedTokens: number; + latencyMs: number; + expansionHops: number; +} + +export interface RetrievalConfidence { + score: number; + reason: string; +} + +export interface ContextPackage { + query: string; + summary: string; + files: Array<{ + path: string; + reason: string; + score: RetrievalScore; + chunks: Array<{ + symbol?: string; + startLine: number; + endLine: number; + content: string; + }>; + }>; + relationships?: Array<{ + from: string; + to: string; + type: 'imports' | 'calls' | 'references' | 'tests'; + }>; + estimatedTokens: number; + retrievalStats: { + candidatesConsidered: number; + candidatesSelected: number; + expansionHops: number; + }; + confidence: RetrievalConfidence; + trace?: RetrievalTrace; +} + +export type ContextMode = 'minimal' | 'normal' | 'deep'; diff --git a/src/search/searchCodebase.ts b/src/search/searchCodebase.ts index 8aa2af6..ce53696 100644 --- a/src/search/searchCodebase.ts +++ b/src/search/searchCodebase.ts @@ -1,9 +1,8 @@ import type { EmbeddingProvider } from '../embeddings/EmbeddingProvider.js'; -import { normalizeVector } from '../embeddings/vectorMath.js'; import type { SearchConfig } from '../config/types.js'; import type { LanceVectorStore } from '../vector-store/LanceVectorStore.js'; -import type { ChunkSearchResult } from '../vector-store/schema.js'; -import { estimateTokensFromChars } from '../utils/tokens.js'; +import { hybridSearch, hybridTrace, type HybridSearchOptions } from '../retrieval/hybrid.js'; +import type { RetrievalCandidate, RetrievalTrace } from '../retrieval/types.js'; export interface SearchOptions { limit?: number; @@ -20,81 +19,51 @@ export interface SearchResultItem { content: string; } +export interface SearchDetailedResult { + results: SearchResultItem[]; + selected: RetrievalCandidate[]; + trace: RetrievalTrace; +} -/** Hybrid ranking: LanceDB vector search + LanceDB full-text search, blended with the configured weights (spec section 16). */ -export async function searchCodebase( +function toItem(candidate: RetrievalCandidate): SearchResultItem { + return { + file: candidate.file, + symbol: candidate.symbol, + startLine: candidate.startLine, + endLine: candidate.endLine, + score: candidate.score.total, + content: candidate.content + }; +} + +/** Hybrid ranking: LanceDB vector search + FTS + inspectable scores, diversity, and token budget. */ +export async function searchCodebaseDetailed( query: string, vectorStore: LanceVectorStore, embeddingProvider: EmbeddingProvider, weights: SearchConfig, options: SearchOptions = {} -): Promise { - const limit = options.limit ?? weights.defaultLimit; - const fetchLimit = Math.max(limit * 4, 20); - - const queryVector = normalizeVector(await embeddingProvider.embed(query)); - const [vectorResults, keywordResults] = await Promise.all([ - vectorStore.vectorSearch(queryVector, fetchLimit), - vectorStore.fullTextSearch(query, fetchLimit).catch(() => [] as ChunkSearchResult[]) - ]); - - const combined = new Map(); - - for (const record of vectorResults) { - const distance = record._distance ?? 2; - const similarity = clamp01(1 - distance / 2); // unit vectors -> squared L2 distance in [0, 4]; keep clamp defensive - combined.set(record.id, { record, vectorScore: similarity, keywordScore: 0 }); - } - - const maxFtsScore = Math.max(...keywordResults.map((r) => r._score ?? 0), 1e-9); - for (const record of keywordResults) { - const normalized = clamp01((record._score ?? 0) / maxFtsScore); - const existing = combined.get(record.id); - if (existing) existing.keywordScore = normalized; - else combined.set(record.id, { record, vectorScore: 0, keywordScore: normalized }); - } - - const queryLower = query.toLowerCase(); - const scored = [...combined.values()].map(({ record, vectorScore, keywordScore }) => { - const symbolScore = symbolMatchScore(record, queryLower); - const score = weights.vectorWeight * vectorScore + weights.keywordWeight * keywordScore + weights.symbolWeight * symbolScore; - return { record, score }; - }); - scored.sort((a, b) => b.score - a.score); - - const minScore = options.minScore ?? 0; - const results: SearchResultItem[] = []; - let tokenBudget = 0; - for (const { record, score } of scored) { - if (results.length >= limit) break; - if (score < minScore) continue; - const estimatedTokens = estimateTokensFromChars(record.content.length); - if (options.maxTokens && results.length > 0 && tokenBudget + estimatedTokens > options.maxTokens) break; - tokenBudget += estimatedTokens; - results.push({ - file: record.file_path, - symbol: record.symbol_name ? qualifiedSymbolName(record) : null, - startLine: record.start_line, - endLine: record.end_line, - score: Math.round(score * 1000) / 1000, - content: record.content - }); - } - return results; +): Promise { + const hybridOptions: HybridSearchOptions = { + limit: options.limit, + minScore: options.minScore, + maxTokens: options.maxTokens + }; + const result = await hybridSearch(query, vectorStore, embeddingProvider, weights, hybridOptions); + return { + results: result.selected.map(toItem), + selected: result.selected, + trace: hybridTrace(query, result) + }; } -function symbolMatchScore(record: ChunkSearchResult, queryLower: string): number { - const name = record.symbol_name?.toLowerCase(); - if (name && queryLower.includes(name)) return 1; - const parent = record.parent_symbol?.toLowerCase(); - if (parent && queryLower.includes(parent)) return 0.5; - return 0; -} - -function qualifiedSymbolName(record: ChunkSearchResult): string { - return record.parent_symbol ? `${record.parent_symbol}.${record.symbol_name}` : (record.symbol_name ?? ''); -} - -function clamp01(value: number): number { - return Math.min(1, Math.max(0, value)); +export async function searchCodebase( + query: string, + vectorStore: LanceVectorStore, + embeddingProvider: EmbeddingProvider, + weights: SearchConfig, + options: SearchOptions = {} +): Promise { + const detailed = await searchCodebaseDetailed(query, vectorStore, embeddingProvider, weights, options); + return detailed.results; } diff --git a/src/vector-store/LanceVectorStore.ts b/src/vector-store/LanceVectorStore.ts index 8eb44b1..3ede3f0 100644 --- a/src/vector-store/LanceVectorStore.ts +++ b/src/vector-store/LanceVectorStore.ts @@ -10,6 +10,7 @@ export interface FileChunkSummary { startLine: number; endLine: number; embedding: number[]; + extraMetadata: string | null; } /** @@ -71,13 +72,14 @@ export class LanceVectorStore { /** Upsert by `id` — matched rows are fully replaced (new embedding/content/lines), unmatched rows are inserted. */ async upsertChunks(records: ChunkRecord[]): Promise { if (records.length === 0) return; + const batch = dedupeById(records); // `ChunkRecord` is an exact interface (no index signature) but LanceDB's `Data` param requires one; // the object shapes are otherwise identical, so this cast is safe. await this.table .mergeInsert('id') .whenMatchedUpdateAll() .whenNotMatchedInsertAll() - .execute(records as unknown as Record[]); + .execute(batch as unknown as Record[]); } async deleteByFile(filePath: string): Promise { @@ -110,14 +112,15 @@ export class LanceVectorStore { const rows = await this.table .query() .where(`file_path = '${escapeSqlString(filePath)}'`) - .select(['id', 'content_hash', 'start_line', 'end_line', 'embedding']) + .select(['id', 'content_hash', 'start_line', 'end_line', 'embedding', 'extra_metadata']) .toArray(); return rows.map((row) => ({ id: String(row.id), contentHash: String(row.content_hash), startLine: Number(row.start_line), endLine: Number(row.end_line), - embedding: Array.from(row.embedding as ArrayLike) + embedding: Array.from(row.embedding as ArrayLike), + extraMetadata: row.extra_metadata == null ? null : String(row.extra_metadata) })); } @@ -179,3 +182,17 @@ export class LanceVectorStore { function escapeSqlString(value: string): string { return value.replace(/'/g, "''"); } + +/** + * LanceDB aborts a merge holding two source rows with the same key, which would fail the whole + * indexing pass over one file. Callers are expected to emit unique ids; last write wins otherwise. + */ +function dedupeById(records: ChunkRecord[]): ChunkRecord[] { + const byId = new Map(); + for (const record of records) byId.set(record.id, record); + if (byId.size === records.length) return records; + logger.warn(`Collapsed ${records.length - byId.size} duplicate chunk id(s) before merge`, { + file: records[0]?.file_path + }); + return [...byId.values()]; +} diff --git a/tests/integration/mcp.test.ts b/tests/integration/mcp.test.ts index 9a5659d..099baf2 100644 --- a/tests/integration/mcp.test.ts +++ b/tests/integration/mcp.test.ts @@ -20,6 +20,7 @@ const ALL_TOOLS = [ 'find_references', 'get_file_context', 'get_repo_context', + 'get_task_context', 'index_status', 'list_indexed_repos', 'search_codebase', @@ -71,7 +72,7 @@ describe('MCP server without an index', () => { await rm(dbHome, { recursive: true, force: true }); }); - it('starts and exposes list/status plus the five retrieval tools', async () => { + it('starts and exposes list/status plus retrieval tools', async () => { const { tools } = await client.listTools(); expect(tools.map((t) => t.name).sort()).toEqual(ALL_TOOLS); }); @@ -89,6 +90,14 @@ describe('MCP server without an index', () => { expect(payload.ok).toBe(false); expect(payload.indexed).toBe(false); }); + + it('get_task_context returns an index hint for an unindexed repo', async () => { + const payload = parseToolText( + await client.callTool({ name: 'get_task_context', arguments: { task: 'add rate limiting' } }) + ); + expect(payload.ok).toBe(false); + expect(payload.indexed).toBe(false); + }); }); // Verifies the MCP server itself: a real client, spawned over stdio (same transport a host IDE uses), diff --git a/tests/unit/benchmark-metrics.test.ts b/tests/unit/benchmark-metrics.test.ts new file mode 100644 index 0000000..1ca215c --- /dev/null +++ b/tests/unit/benchmark-metrics.test.ts @@ -0,0 +1,57 @@ +import { describe, expect, it } from 'vitest'; +import { + meanReciprocalRank, + ndcgAtK, + percentile, + precisionAtK, + recallAtK, + tokenReductionPercent +} from '../../src/benchmark/metrics.js'; +import { workspaceScan } from '../../src/benchmark/workspaceScan.js'; + +describe('retrieval metrics', () => { + const retrieved = ['a.ts', 'b.ts', 'c.ts', 'd.ts']; + const relevant = new Set(['b.ts', 'z.ts']); + + it('computes precision and recall at K', () => { + expect(precisionAtK(retrieved, relevant, 2)).toBe(0.5); + expect(recallAtK(retrieved, relevant, 10)).toBe(0.5); + }); + + it('computes MRR from the first relevant hit', () => { + expect(meanReciprocalRank(retrieved, relevant)).toBe(0.5); + }); + + it('computes NDCG with graded relevance', () => { + const ndcg = ndcgAtK(['a.ts', 'b.ts'], { 'a.ts': 3, 'b.ts': 1 }, 2); + expect(ndcg).toBeGreaterThan(0.8); + expect(ndcg).toBeLessThanOrEqual(1); + }); + + it('computes token reduction and percentiles', () => { + expect(tokenReductionPercent(100, 40)).toBe(60); + expect(percentile([10, 20, 30, 40], 50)).toBe(20); + }); + + it('serializes a report without source content fields', () => { + const report = { + version: 1, + tasks: [{ id: 'x', retrievedFiles: ['a.ts'], missingRelevantFiles: ['b.ts'] }], + aggregate: { precisionAt5: 1 } + }; + const json = JSON.stringify(report); + expect(json).not.toContain('function '); + expect(json).toContain('"version":1'); + }); + + it('fails clearly instead of reporting a zero baseline when ripgrep is missing', () => { + const previous = process.env.CODE_INTEL_RG; + process.env.CODE_INTEL_RG = '/definitely/missing/code-intel-rg'; + try { + expect(() => workspaceScan(process.cwd(), 'searchCodebase')).toThrow(/Install ripgrep|CODE_INTEL_RG/); + } finally { + if (previous === undefined) delete process.env.CODE_INTEL_RG; + else process.env.CODE_INTEL_RG = previous; + } + }); +}); diff --git a/tests/unit/chunk-id-collision.test.ts b/tests/unit/chunk-id-collision.test.ts new file mode 100644 index 0000000..2afcda5 --- /dev/null +++ b/tests/unit/chunk-id-collision.test.ts @@ -0,0 +1,78 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { afterEach, beforeEach, describe, expect, it } from 'vitest'; +import { loadConfig } from '../../src/config/load.js'; +import { resolveRepoPaths } from '../../src/config/paths.js'; +import type { EmbeddingProvider } from '../../src/embeddings/EmbeddingProvider.js'; +import { Indexer } from '../../src/indexer/Indexer.js'; +import { computeRepoId } from '../../src/utils/repo-id.js'; +import type { LanceVectorStore } from '../../src/vector-store/LanceVectorStore.js'; +import type { ChunkRecord } from '../../src/vector-store/schema.js'; + +const TWO_SAME_NAMED_FUNCTIONS = `export function handle(a: number) { + return a + 1; +} + +export function handle(a: string) { + return a + '!'; +} +`; + +describe('chunk id collisions within one file', () => { + let repoRoot: string; + let dbHome: string; + const originalDbPath = process.env.CODE_INTEL_DB_PATH; + + beforeEach(async () => { + repoRoot = await mkdtemp(join(tmpdir(), 'code-intel-collision-repo-')); + dbHome = await mkdtemp(join(tmpdir(), 'code-intel-collision-db-')); + process.env.CODE_INTEL_DB_PATH = dbHome; + }); + + afterEach(async () => { + await rm(repoRoot, { recursive: true, force: true }); + await rm(dbHome, { recursive: true, force: true }); + if (originalDbPath === undefined) delete process.env.CODE_INTEL_DB_PATH; + else process.env.CODE_INTEL_DB_PATH = originalDbPath; + }); + + it('gives same-named siblings distinct ids so the merge batch stays unambiguous', async () => { + await writeFile(join(repoRoot, 'dup.ts'), TWO_SAME_NAMED_FUNCTIONS); + const config = loadConfig({ repoRoot }); + const paths = resolveRepoPaths(config, repoRoot, computeRepoId(repoRoot)); + const stored: ChunkRecord[] = []; + const vectorStore = { + getAllFileHashes: async () => new Map(), + getChunksForFile: async () => [], + upsertChunks: async (records: ChunkRecord[]) => { + stored.push(...records); + }, + deleteByIds: async () => undefined, + deleteByFile: async () => undefined, + renameFile: async () => undefined, + optimize: async () => undefined, + countRows: async () => stored.length + } as unknown as LanceVectorStore; + const embeddingProvider: EmbeddingProvider = { + embed: async () => [1, 0], + embedBatch: async (texts) => texts.map(() => [1, 0]), + dimensions: async () => 2, + modelName: () => 'test-model' + }; + + const summary = await new Indexer({ + repoRoot, + repoId: computeRepoId(repoRoot), + config, + vectorStore, + embeddingProvider, + paths + }).runFullIndex(); + + const duplicates = stored.filter((record) => record.symbol_name === 'handle'); + expect(duplicates).toHaveLength(2); + expect(new Set(duplicates.map((record) => record.id)).size).toBe(2); + expect(summary.chunksEmbedded).toBe(2); + }); +}); diff --git a/tests/unit/chunk-metadata.test.ts b/tests/unit/chunk-metadata.test.ts new file mode 100644 index 0000000..c581947 --- /dev/null +++ b/tests/unit/chunk-metadata.test.ts @@ -0,0 +1,46 @@ +import { describe, expect, it } from 'vitest'; +import { + buildChunkExtraMetadata, + extractImports, + isConfigPath, + isTestPath, + parseChunkExtraMetadata +} from '../../src/chunker/chunkMetadata.js'; + +describe('chunk metadata', () => { + it('detects tests and config from paths', () => { + expect(isTestPath('tests/unit/foo.test.ts')).toBe(true); + expect(isTestPath('src/foo.ts')).toBe(false); + expect(isConfigPath('package.json')).toBe(true); + expect(isConfigPath('vite.config.ts')).toBe(true); + expect(isConfigPath('src/search/searchCodebase.ts')).toBe(false); + }); + + it('extracts TS imports and exports', () => { + const content = ` +import { searchCodebase } from '../search/searchCodebase.js'; +export function buildServer() {} +`; + expect(extractImports(content, 'typescript')).toContain('../search/searchCodebase.js'); + const meta = buildChunkExtraMetadata('src/mcp/server.ts', content, content, 'typescript'); + expect(meta.exports).toContain('buildServer'); + expect(meta.isTest).toBe(false); + }); + + it('round-trips JSON extra_metadata', () => { + const meta = buildChunkExtraMetadata( + 'tests/unit/foo.test.ts', + 'import "./bar.js"', + 'searchCodebase()', + 'typescript' + ); + expect(meta.isTest).toBe(true); + const parsed = parseChunkExtraMetadata(JSON.stringify(meta)); + expect(parsed.isTest).toBe(true); + expect(parsed.imports.length).toBeGreaterThan(0); + }); + + it('returns empty metadata for invalid JSON', () => { + expect(parseChunkExtraMetadata('not-json').imports).toEqual([]); + }); +}); diff --git a/tests/unit/cursor-install.test.ts b/tests/unit/cursor-install.test.ts index 8f21359..0ff4dc2 100644 --- a/tests/unit/cursor-install.test.ts +++ b/tests/unit/cursor-install.test.ts @@ -38,6 +38,7 @@ describe('cursor-install', () => { expect(mcp.mcpServers['local-code-intelligence'].command).toBe('node'); const rule = await readFile(join(cursorHome, 'rules', LOCAL_CODE_INTEL_RULE_FILENAME), 'utf-8'); expect(rule).toContain('alwaysApply: true'); + expect(rule).toContain('get_task_context'); expect(rule).toContain('search_codebase'); const hooks = JSON.parse(await readFile(result.hooksPath, 'utf-8')) as { hooks: { preToolUse: { matcher: string }[]; sessionStart: { command: string }[] }; diff --git a/tests/unit/format-cli-failure.test.ts b/tests/unit/format-cli-failure.test.ts new file mode 100644 index 0000000..93f7319 --- /dev/null +++ b/tests/unit/format-cli-failure.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from 'vitest'; +import { formatCliFailure } from '../../src/cli/formatCliFailure.js'; +import { IndexerLockedError } from '../../src/indexer/lock.js'; +import { OllamaNotReachableError } from '../../src/embeddings/OllamaEmbeddingProvider.js'; +import { EmbeddingConfigurationError, OpenAICompatibleEmbeddingError } from '../../src/embeddings/OpenAICompatibleEmbeddingProvider.js'; + +describe('formatCliFailure', () => { + it('lists how to supply a missing company-proxy API key', () => { + const text = formatCliFailure( + new EmbeddingConfigurationError( + 'CODE_INTEL_EMBEDDING_API_KEY is required for provider "openai-compatible".' + ) + ); + expect(text).toContain('[FAIL] CODE_INTEL_EMBEDDING_API_KEY is required'); + expect(text).toContain('What you can do:'); + expect(text).toContain('export CODE_INTEL_EMBEDDING_API_KEY'); + expect(text).toContain('code-intel wizard'); + expect(text).toContain('--embedding-provider ollama'); + expect(text).toContain('code-intel doctor'); + }); + + it('lists how to start Ollama when the daemon is down', () => { + const text = formatCliFailure(new OllamaNotReachableError('Cannot reach Ollama. Try `ollama serve`.')); + expect(text).toContain('What you can do:'); + expect(text).toContain('ollama serve'); + }); + + it('lists how to recover from a live indexer lock', () => { + const text = formatCliFailure( + new IndexerLockedError('Another code-intel process (pid 12) is already indexing this repository.') + ); + expect(text).toContain('Wait for the other'); + expect(text).toContain('.lock'); + }); + + it('lists how to fix a missing index', () => { + const text = formatCliFailure(new Error('Run `code-intel setup --repo ` to index this workspace.')); + expect(text).toContain('code-intel setup --repo'); + expect(text).toContain('CODE_INTEL_EMBEDDING_API_KEY'); + }); + + it('lists how to recover from a proxy timeout', () => { + const text = formatCliFailure( + new OpenAICompatibleEmbeddingError('Embedding request failed (408 Request Timeout)', 408, true) + ); + expect(text).toContain('Retry in a minute'); + expect(text).toContain('timeout_ms'); + }); +}); diff --git a/tests/unit/freshness.test.ts b/tests/unit/freshness.test.ts new file mode 100644 index 0000000..00367b7 --- /dev/null +++ b/tests/unit/freshness.test.ts @@ -0,0 +1,34 @@ +import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { describe, expect, it } from 'vitest'; +import { + computeSampleFingerprint, + isContentSampleStale, + pickSamplePaths +} from '../../src/indexer/freshness.js'; +import { hashFileContent } from '../../src/hashing/hash.js'; + +describe('index freshness sample', () => { + it('picks a stable evenly spaced subset', () => { + const paths = Array.from({ length: 100 }, (_, i) => `f${String(i).padStart(3, '0')}.ts`); + const sample = pickSamplePaths(paths, 8); + expect(sample).toHaveLength(8); + expect(pickSamplePaths(paths, 8)).toEqual(sample); + }); + + it('detects a content change when the file count is unchanged', () => { + const dir = mkdtempSync(join(tmpdir(), 'code-intel-fresh-')); + mkdirSync(join(dir, 'src')); + writeFileSync(join(dir, 'src/a.ts'), 'export const a = 1;\n'); + writeFileSync(join(dir, 'src/b.ts'), 'export const b = 1;\n'); + const samplePaths = ['src/a.ts', 'src/b.ts']; + const fingerprint = computeSampleFingerprint(dir, samplePaths); + expect(fingerprint).toBeTruthy(); + expect(isContentSampleStale(dir, { samplePaths, sampleFingerprint: fingerprint! })).toBe(false); + + writeFileSync(join(dir, 'src/a.ts'), 'export const a = 2;\n'); + expect(isContentSampleStale(dir, { samplePaths, sampleFingerprint: fingerprint! })).toBe(true); + expect(hashFileContent('export const a = 2;\n')).not.toBe(hashFileContent('export const a = 1;\n')); + }); +}); diff --git a/tests/unit/onboard-cli.test.ts b/tests/unit/onboard-cli.test.ts index a2ee94f..850da1f 100644 --- a/tests/unit/onboard-cli.test.ts +++ b/tests/unit/onboard-cli.test.ts @@ -13,6 +13,8 @@ describe('onboard CLI', () => { it('exposes onboard and doctor --fix in help', async () => { const { stdout: rootHelp } = await execFileAsync(tsx, [cli, '--help'], { cwd: projectRoot }); expect(rootHelp).toContain('onboard'); + expect(rootHelp).toContain('context'); + expect(rootHelp).toContain('benchmark'); expect(rootHelp).toMatch(/pull the embedding model|index this repo|wire Cursor/i); const { stdout: doctorHelp } = await execFileAsync(tsx, [cli, 'doctor', '--help'], { diff --git a/tests/unit/resolve-import.test.ts b/tests/unit/resolve-import.test.ts new file mode 100644 index 0000000..c878906 --- /dev/null +++ b/tests/unit/resolve-import.test.ts @@ -0,0 +1,50 @@ +import { mkdtempSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { describe, expect, it } from 'vitest'; +import { candidateImportPaths, loadTsPathAliases } from '../../src/retrieval/resolveImport.js'; + +describe('candidateImportPaths', () => { + it('resolves a relative JS specifier to TypeScript candidates next to the importer', () => { + const paths = candidateImportPaths('src/mcp/server.ts', '../search/searchCodebase.js'); + expect(paths).toContain('src/search/searchCodebase.ts'); + expect(paths).toContain('src/search/searchCodebase.js'); + expect(paths.every((path) => path.startsWith('src/search/'))).toBe(true); + }); + + it('resolves ./foo to the same directory and index files', () => { + const paths = candidateImportPaths('src/auth/middleware.ts', './session'); + expect(paths).toContain('src/auth/session.ts'); + expect(paths).toContain('src/auth/session/index.ts'); + }); + + it('ignores package imports and paths that escape the repo', () => { + expect(candidateImportPaths('src/a.ts', 'zod')).toEqual([]); + expect(candidateImportPaths('src/a.ts', '../../outside')).toEqual([]); + }); + + it('resolves tsconfig-style path aliases', () => { + const aliases = [{ prefix: '@app/', targets: ['src/'] }]; + const paths = candidateImportPaths('src/mcp/server.ts', '@app/search/searchCodebase', aliases); + expect(paths).toContain('src/search/searchCodebase.ts'); + }); + + it('resolves Python dotted modules next to the importer and from repo-style packages', () => { + expect(candidateImportPaths('src/retrieval/task.py', '.score')).toContain('src/retrieval/score.py'); + expect(candidateImportPaths('pkg/mod.py', 'retrieval.score')).toContain('retrieval/score.py'); + expect(candidateImportPaths('pkg/mod.py', 'retrieval.score')).toContain('retrieval/score/__init__.py'); + }); + + it('loads TypeScript path aliases from tsconfig.json', () => { + const dir = mkdtempSync(join(tmpdir(), 'code-intel-alias-')); + writeFileSync( + join(dir, 'tsconfig.json'), + JSON.stringify({ compilerOptions: { baseUrl: '.', paths: { '@app/*': ['src/*'] } } }) + ); + const aliases = loadTsPathAliases(dir); + expect(aliases[0]?.prefix).toBe('@app/'); + expect(candidateImportPaths('src/a.ts', '@app/search/searchCodebase', aliases)).toContain( + 'src/search/searchCodebase.ts' + ); + }); +}); diff --git a/tests/unit/retrieval-confidence.test.ts b/tests/unit/retrieval-confidence.test.ts new file mode 100644 index 0000000..3ec9e65 --- /dev/null +++ b/tests/unit/retrieval-confidence.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from 'vitest'; +import { DEFAULT_CONFIG } from '../../src/config/defaults.js'; +import { confidenceFor, isDocHeavyPath, queryLooksLikeCodeChange } from '../../src/retrieval/confidence.js'; +import type { RetrievalCandidate } from '../../src/retrieval/types.js'; + +function candidate(file: string, total = 0.9): RetrievalCandidate { + return { + id: file, + file, + symbol: null, + symbolType: null, + parentSymbol: null, + startLine: 1, + endLine: 2, + content: 'x', + lastIndexedAt: null, + extra: { imports: [], exports: [], referencedSymbols: [], isTest: false, isConfig: false }, + sources: ['semantic'], + score: { + total, + semantic: total, + keyword: 0, + symbol: 0, + path: 0, + structural: 0, + dependency: 0, + reference: 0, + test: 0, + recency: 0 + }, + estimatedTokens: 4, + reason: 'semantic similarity' + }; +} + +describe('retrieval confidence', () => { + const threshold = DEFAULT_CONFIG.retrieval.confidenceThreshold; + + it('flags documentation paths', () => { + expect(isDocHeavyPath('CODE_INTEL_NEXT_PHASE.md')).toBe(true); + expect(isDocHeavyPath('examples/cursor/local-code-intel.SKILL.md')).toBe(true); + expect(isDocHeavyPath('src/retrieval/taskContext.ts')).toBe(false); + }); + + it('softens confidence when a code-change query ranks a doc first', () => { + expect(queryLooksLikeCodeChange('Add request IDs to MCP tool responses.')).toBe(true); + const result = confidenceFor( + [candidate('instructions.md'), candidate('src/mcp/server.ts', 0.4)], + threshold, + { query: 'Add request IDs to MCP tool responses.' } + ); + expect(result.score).toBeLessThan(threshold); + expect(result.reason).toMatch(/documentation/i); + }); + + it('keeps high confidence when the top hit is source', () => { + const result = confidenceFor([candidate('src/mcp/server.ts')], threshold, { + query: 'Add request IDs to MCP tool responses.' + }); + expect(result.score).toBe(0.9); + }); +}); diff --git a/tests/unit/retrieval-context.test.ts b/tests/unit/retrieval-context.test.ts new file mode 100644 index 0000000..91ee088 --- /dev/null +++ b/tests/unit/retrieval-context.test.ts @@ -0,0 +1,113 @@ +import { mkdtemp, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { afterEach, describe, expect, it } from 'vitest'; +import { + grantFilesystemFallback, + isFilesystemFallbackOpen +} from '../../src/retrieval/fallback.js'; +import { buildContextPackageForTest } from '../../src/retrieval/taskContext.js'; +import type { RetrievalCandidate } from '../../src/retrieval/types.js'; +import { shouldDenyTreeScan } from '../../src/cursor/treeScanPolicy.js'; + +function fakeCandidate(): RetrievalCandidate { + return { + id: '1', + file: 'src/search/searchCodebase.ts', + symbol: 'searchCodebase', + symbolType: 'function', + parentSymbol: null, + startLine: 1, + endLine: 20, + content: 'export async function searchCodebase() {}', + lastIndexedAt: null, + extra: { imports: [], exports: ['searchCodebase'], referencedSymbols: [], isTest: false, isConfig: false }, + sources: ['semantic'], + score: { + total: 0.8, + semantic: 0.8, + keyword: 0, + symbol: 1, + path: 0, + structural: 1, + dependency: 0, + reference: 0, + test: 0, + recency: 0 + }, + estimatedTokens: 12, + reason: 'semantic similarity' + }; +} + +describe('context package', () => { + it('groups chunks by file and reports empty-result confidence', () => { + const empty = buildContextPackageForTest( + 'missing thing', + [], + 0, + 0, + { score: 0, reason: 'Low-confidence retrieval. Recommended fallback: targeted repository search.' } + ); + expect(empty.files).toEqual([]); + expect(empty.confidence.score).toBe(0); + expect(empty.confidence.reason).toMatch(/Low-confidence/); + + const packed = buildContextPackageForTest('search', [fakeCandidate()], 4, 1, { + score: 0.8, + reason: 'ok' + }); + expect(packed.files).toHaveLength(1); + expect(packed.files[0]?.path).toBe('src/search/searchCodebase.ts'); + expect(packed.retrievalStats.candidatesConsidered).toBe(4); + }); + + it('orders packaged files by their best score, not insertion order', () => { + const best = fakeCandidate(); + const lower: RetrievalCandidate = { + ...fakeCandidate(), + id: '2', + file: 'src/unrelated.ts', + score: { ...fakeCandidate().score, total: 0.2 } + }; + const packed = buildContextPackageForTest('search', [lower, best], 2, 0, { + score: 0.8, + reason: 'ok' + }); + expect(packed.files.map((file) => file.path)).toEqual([ + 'src/search/searchCodebase.ts', + 'src/unrelated.ts' + ]); + }); +}); + +describe('filesystem fallback window', () => { + let dir: string; + + afterEach(async () => { + if (dir) await rm(dir, { recursive: true, force: true }); + }); + + it('opens a short window after a failed retrieval', async () => { + dir = await mkdtemp(join(tmpdir(), 'code-intel-fallback-')); + expect(isFilesystemFallbackOpen(dir)).toBe(false); + grantFilesystemFallback(dir, 'low confidence', '/repo'); + expect(isFilesystemFallbackOpen(dir)).toBe(true); + }); +}); + +describe('shouldDenyTreeScan fallback', () => { + const cwd = '/Users/me/proj'; + + it('allows workspace Grep/Glob when fallback is armed, but still denies explore', () => { + expect( + shouldDenyTreeScan( + { tool_name: 'Grep', cwd, workspace_roots: [cwd], tool_input: { pattern: 'foo' } }, + { allowFallback: true } + ) + ).toBe(false); + expect( + shouldDenyTreeScan({ hook_event_name: 'subagentStart', subagent_type: 'explore' }, { allowFallback: true }) + ).toBe(true); + }); +}); diff --git a/tests/unit/retrieval-intent.test.ts b/tests/unit/retrieval-intent.test.ts new file mode 100644 index 0000000..e077749 --- /dev/null +++ b/tests/unit/retrieval-intent.test.ts @@ -0,0 +1,40 @@ +import { describe, expect, it } from 'vitest'; +import { analyzeQuery } from '../../src/retrieval/intent.js'; + +describe('analyzeQuery', () => { + it('extracts quoted identifiers and identifier-like tokens', () => { + const intent = analyzeQuery('Add rate limiting to createUser and update the tests.'); + expect(intent.symbols).toContain('createUser'); + expect(intent.operations).toContain('find_implementation'); + expect(intent.operations).toContain('find_tests'); + }); + + it('detects configuration intent', () => { + const intent = analyzeQuery('Where is the production database connection configured?'); + expect(intent.operations).toContain('find_config'); + }); + + it('detects language from file extensions', () => { + const intent = analyzeQuery('Fix src/api/users.ts'); + expect(intent.files).toContain('src/api/users.ts'); + expect(intent.likelyLanguages).toContain('typescript'); + }); + + it('keeps normal context for a short known-symbol question unless mode is explicit', () => { + const intent = analyzeQuery('Where is AuthService?'); + expect(intent.requestedContext).toBe('normal'); + expect(intent.symbols).toContain('AuthService'); + expect(analyzeQuery('Where is AuthService?', 'minimal').requestedContext).toBe('minimal'); + }); + + it('does not add test expansion just because a task is a fix', () => { + expect(analyzeQuery('Fix stale index detection.').operations).not.toContain('find_tests'); + expect(analyzeQuery('Fix stale index detection and update tests.').operations).toContain('find_tests'); + }); + + it('keeps product acronyms as concepts instead of treating them as code symbols', () => { + const intent = analyzeQuery('Add request IDs to MCP tool responses.'); + expect(intent.symbols).not.toContain('MCP'); + expect(intent.symbols).not.toContain('IDs'); + }); +}); diff --git a/tests/unit/retrieval-score.test.ts b/tests/unit/retrieval-score.test.ts new file mode 100644 index 0000000..bbc2b18 --- /dev/null +++ b/tests/unit/retrieval-score.test.ts @@ -0,0 +1,158 @@ +import { describe, expect, it } from 'vitest'; +import { DEFAULT_CONFIG } from '../../src/config/defaults.js'; +import { + basenameMatchScore, + buildRetrievalScore, + candidateFromRecord, + combineScore, + pathScore, + recencyScore, + structuralScore, + symbolMatchScore +} from '../../src/retrieval/score.js'; +import type { ChunkSearchResult } from '../../src/vector-store/schema.js'; + +const weights = DEFAULT_CONFIG.search; + +function record(overrides: Partial = {}): ChunkSearchResult { + return { + id: '1', + repo_id: 'repo', + file_path: 'src/search/searchCodebase.ts', + absolute_path: '/repo/src/search/searchCodebase.ts', + language: 'typescript', + symbol_name: 'searchCodebase', + symbol_type: 'function', + parent_symbol: null, + start_line: 1, + end_line: 10, + content: 'export function searchCodebase() {}', + content_hash: 'content', + file_hash: 'file', + embedding: [], + last_indexed_at: null, + git_commit: null, + extra_metadata: null, + ...overrides + }; +} + +describe('retrieval scoring', () => { + it('combines semantic, keyword, and symbol with configured weights', () => { + const total = combineScore( + { + semantic: 1, + keyword: 0, + symbol: 0, + path: 0, + structural: 0, + dependency: 0, + reference: 0, + test: 0, + recency: 0 + }, + weights + ); + expect(total).toBeCloseTo(0.7, 5); + }); + + it('boosts keyword independently of semantic', () => { + const total = combineScore( + { + semantic: 0, + keyword: 1, + symbol: 0, + path: 0, + structural: 0, + dependency: 0, + reference: 0, + test: 0, + recency: 0 + }, + weights + ); + expect(total).toBeCloseTo(0.2, 5); + }); + + it('scores symbol matches when the query contains the identifier', () => { + expect(symbolMatchScore({ symbol_name: 'searchCodebase', parent_symbol: null }, 'where is searchcodebase')).toBe(1); + expect(symbolMatchScore({ symbol_name: 'run', parent_symbol: 'Indexer' }, 'indexer flow')).toBe(0.5); + expect(symbolMatchScore({ symbol_name: 'foo', parent_symbol: null }, 'bar')).toBe(0); + }); + + it('gives a path boost when the query tokens appear in the file path', () => { + expect(pathScore('src/search/searchCodebase.ts', 'search codebase ranking')).toBeGreaterThan(0); + expect(basenameMatchScore('src/search/searchCodebase.ts', 'where is searchCodebase implemented')).toBe(1); + expect(pathScore('src/unrelated.ts', 'authentication middleware')).toBe(0); + }); + + it('ranks exact symbols above generic semantic neighbors', () => { + const exact = candidateFromRecord( + record(), + { semantic: 0.2, keyword: 0, sources: ['semantic'], reason: 'semantic similarity' }, + weights, + 'where is searchcodebase implemented' + ); + const generic = candidateFromRecord( + record({ + id: '2', + file_path: 'src/other/highSimilarity.ts', + symbol_name: 'highSimilarity' + }), + { semantic: 0.9, keyword: 0, sources: ['semantic'], reason: 'semantic similarity' }, + weights, + 'where is searchcodebase implemented' + ); + expect(exact.score.total).toBeGreaterThan(generic.score.total); + expect(exact.reason).toMatch(/exact symbol/); + }); + + it('boosts test files only when the query asks for tests', () => { + const testRecord = record({ file_path: 'tests/searchCodebase.test.ts', symbol_name: 'coversSearch' }); + const normal = candidateFromRecord( + testRecord, + { semantic: 0.5, keyword: 0, sources: ['semantic'], reason: 'semantic similarity' }, + weights, + 'fix search ranking' + ); + const requested = candidateFromRecord( + testRecord, + { semantic: 0.5, keyword: 0, sources: ['semantic'], reason: 'semantic similarity' }, + weights, + 'find search ranking tests' + ); + expect(normal.score.test).toBe(0); + expect(requested.score.test).toBe(1); + expect(requested.score.total).toBeGreaterThan(normal.score.total); + }); + + it('gives structural score for named symbols', () => { + expect(structuralScore({ symbol_name: 'run', parent_symbol: 'Indexer' })).toBe(1); + expect(structuralScore({ symbol_name: null, parent_symbol: null })).toBe(0); + }); + + it('decays recency over a week', () => { + const now = Date.parse('2026-01-08T00:00:00.000Z'); + expect(recencyScore(new Date(now).toISOString(), now)).toBe(1); + expect(recencyScore('2026-01-01T00:00:00.000Z', now)).toBeCloseTo(0, 5); + }); + + it('keeps a rounded inspectable total', () => { + const score = buildRetrievalScore( + { + semantic: 1, + keyword: 1, + symbol: 1, + path: 0, + structural: 0, + dependency: 0, + reference: 0, + test: 0, + recency: 0 + }, + weights + ); + expect(score.total).toBe(1); + expect(score.semantic).toBe(1); + }); +}); diff --git a/tests/unit/retrieval-select.test.ts b/tests/unit/retrieval-select.test.ts new file mode 100644 index 0000000..3aa9373 --- /dev/null +++ b/tests/unit/retrieval-select.test.ts @@ -0,0 +1,93 @@ +import { describe, expect, it } from 'vitest'; +import type { RetrievalCandidate } from '../../src/retrieval/types.js'; +import { selectCandidates } from '../../src/retrieval/select.js'; + +function candidate( + overrides: Partial & { id: string; file: string; scoreTotal: number } +): RetrievalCandidate { + const estimatedTokens = overrides.estimatedTokens ?? 10; + return { + symbol: overrides.symbol ?? 'fn', + symbolType: 'function', + parentSymbol: null, + startLine: 1, + endLine: 10, + content: 'x'.repeat(estimatedTokens * 4), + lastIndexedAt: null, + extra: { imports: [], exports: [], referencedSymbols: [], isTest: false, isConfig: false }, + sources: overrides.sources ?? ['semantic'], + score: { + total: overrides.scoreTotal, + semantic: overrides.scoreTotal, + keyword: 0, + symbol: 0, + path: 0, + structural: 0, + dependency: 0, + reference: 0, + test: 0, + recency: 0 + }, + estimatedTokens, + reason: 'test', + id: overrides.id, + file: overrides.file + }; +} + +describe('selectCandidates', () => { + it('caps chunks per file and per symbol', () => { + const ranked = [ + candidate({ id: '1', file: 'a.ts', symbol: 'foo', scoreTotal: 1 }), + candidate({ id: '2', file: 'a.ts', symbol: 'foo', scoreTotal: 0.9 }), + candidate({ id: '3', file: 'a.ts', symbol: 'bar', scoreTotal: 0.8 }), + candidate({ id: '4', file: 'a.ts', symbol: 'baz', scoreTotal: 0.7 }), + candidate({ id: '5', file: 'a.ts', symbol: 'qux', scoreTotal: 0.6 }), + candidate({ id: '6', file: 'a.ts', symbol: 'zap', scoreTotal: 0.5 }) + ]; + const result = selectCandidates(ranked, { + limit: 10, + maxChunksPerFile: 4, + maxChunksPerSymbol: 2 + }); + expect(result.selected).toHaveLength(4); + expect(result.selected.filter((c) => c.symbol === 'foo')).toHaveLength(2); + expect(result.discarded.some((d) => d.reason === 'per_file_cap')).toBe(true); + }); + + it('respects a hard token budget but keeps the first useful chunk', () => { + const ranked = [ + candidate({ id: '1', file: 'a.ts', symbol: 'a', scoreTotal: 1, estimatedTokens: 80 }), + candidate({ id: '2', file: 'b.ts', symbol: 'b', scoreTotal: 0.9, estimatedTokens: 80 }) + ]; + const result = selectCandidates(ranked, { + limit: 10, + maxTokens: 50, + maxChunksPerFile: 4, + maxChunksPerSymbol: 2 + }); + expect(result.selected).toHaveLength(1); + expect(result.selected[0]?.id).toBe('1'); + expect(result.discarded.some((d) => d.reason === 'token_budget')).toBe(true); + }); + + it('keeps expansion-only files from crowding direct hits out of the top five', () => { + const ranked = [ + candidate({ id: 'e1', file: 'dep-a.ts', symbol: 'a', scoreTotal: 1, sources: ['expansion'] }), + candidate({ id: 'e2', file: 'dep-b.ts', symbol: 'b', scoreTotal: 0.95, sources: ['expansion'] }), + candidate({ id: 'e3', file: 'dep-c.ts', symbol: 'c', scoreTotal: 0.9, sources: ['expansion'] }), + candidate({ id: 'd1', file: 'implementation.ts', symbol: 'implementation', scoreTotal: 0.85 }), + candidate({ id: 'd2', file: 'tests.ts', symbol: 'tests', scoreTotal: 0.8 }) + ]; + const result = selectCandidates(ranked, { + limit: 5, + maxChunksPerFile: 4, + maxChunksPerSymbol: 2, + maxExpansionOnlyFilesInTopK: 2, + expansionTopK: 5 + }); + expect(result.selected.map((item) => item.file)).toContain('implementation.ts'); + expect(result.selected.filter((item) => item.sources[0] === 'expansion')).toHaveLength(2); + expect(result.discarded.some((item) => item.reason === 'expansion_top_k_cap')).toBe(true); + }); +}); diff --git a/tests/unit/vector-store.test.ts b/tests/unit/vector-store.test.ts index 161be76..985e327 100644 --- a/tests/unit/vector-store.test.ts +++ b/tests/unit/vector-store.test.ts @@ -58,6 +58,16 @@ describe('LanceVectorStore', () => { expect(chunks[0]?.contentHash).toBe('hash-2'); }); + it('collapses duplicate ids in one batch rather than failing the merge', async () => { + await store.upsertChunks([ + makeChunk({ content_hash: 'hash-first' }), + makeChunk({ content_hash: 'hash-last' }) + ]); + const chunks = await store.getChunksForFile('src/auth.ts'); + expect(chunks).toHaveLength(1); + expect(chunks[0]?.contentHash).toBe('hash-last'); + }); + it('tracks file hashes for incremental diffing', async () => { await store.upsertChunks([makeChunk({})]); const hashes = await store.getAllFileHashes(); diff --git a/tests/unit/watch-events.test.ts b/tests/unit/watch-events.test.ts new file mode 100644 index 0000000..cffde14 --- /dev/null +++ b/tests/unit/watch-events.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest'; +import { planWatchIndex, WATCH_FULL_INDEX_THRESHOLD } from '../../src/indexer/watch.js'; + +describe('planWatchIndex', () => { + const options = { allowSensitiveFiles: false, extraIgnorePatterns: [] as string[] }; + + it('plans a partial upsert for a single source edit', () => { + const plan = planWatchIndex('/repo', [{ type: 'update', path: '/repo/src/a.ts' }], options); + expect(plan.mode).toBe('partial'); + expect(plan.upserts).toEqual(['src/a.ts']); + expect(plan.deletes).toEqual([]); + }); + + it('records deletes even when the file is gone', () => { + const plan = planWatchIndex('/repo', [{ type: 'delete', path: '/repo/src/gone.ts' }], options); + expect(plan.deletes).toEqual(['src/gone.ts']); + }); + + it('falls back to a full scan when a burst exceeds the threshold', () => { + const events = Array.from({ length: WATCH_FULL_INDEX_THRESHOLD + 1 }, (_, i) => ({ + type: 'update', + path: `/repo/src/f${i}.ts` + })); + expect(planWatchIndex('/repo', events, options).mode).toBe('full'); + }); +});