{"_id":"@aravindh-arumugam/codebrain-mcp","_rev":"3-a8c8e92b80eba7e66757ab6546cc0666","name":"@aravindh-arumugam/codebrain-mcp","dist-tags":{"latest":"1.1.0"},"versions":{"0.1.0":{"name":"@aravindh-arumugam/codebrain-mcp","version":"0.1.0","keywords":["mcp","model-context-protocol","codebrain","typescript","javascript","tree-sitter","static-analysis","claude-code","cursor"],"license":"MIT","_id":"@aravindh-arumugam/codebrain-mcp@0.1.0","maintainers":[{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"}],"bin":{"codebrain-mcp":"dist/cli/main.js"},"dist":{"shasum":"c018866ac9bf81cef4a764752f02c8a056ddcb41","tarball":"https://registry.npmjs.org/@aravindh-arumugam/codebrain-mcp/-/codebrain-mcp-0.1.0.tgz","fileCount":11,"integrity":"sha512-F/m60k/7KdjjeoOGBr0CuJiz/oH4NBktdKTb6SjR6/lPtAldovCnP4p91m0d/PCJo1w/TJ2Ol1xWkArkVIHB/w==","signatures":[{"sig":"MEUCIG4VuaAB/RxJVt4EN6Ywgky8lqu+6gTVm4QUX+uQLFyiAiEAy9G016erOLwNJ+aa5BoApHcV/CDN9TAeuUYofaDEJ38=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1379986},"main":"./dist/index.js","pnpm":{"peerDependencyRules":{"allowedVersions":{"tree-sitter-javascript>tree-sitter":"0.25.1","tree-sitter-typescript>tree-sitter":"0.25.1"}},"onlyBuiltDependencies":["esbuild","@vscode/ripgrep"],"ignoredBuiltDependencies":["tree-sitter","tree-sitter-javascript","tree-sitter-typescript"]},"type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=18.18.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./package.json":"./package.json"},"gitHead":"2704fad30e0277cf9d67bdf14ea3bd0cb9df82ca","scripts":{"dev":"tsup --watch","lint":"eslint .","test":"vitest run","bench":"pnpm run build && node bench/compare.mjs","build":"tsup","check":"pnpm run typecheck && pnpm run lint && pnpm run test","start":"node ./dist/cli/main.js","format":"prettier --write .","docs:dev":"vitepress dev docs","lint:fix":"eslint . --fix","typecheck":"tsc --noEmit","docs:build":"vitepress build docs","test:watch":"vitest","bench:sample":"pnpm run build && node bench/sample.mjs","docs:preview":"vitepress preview docs","format:check":"prettier --check .","test:coverage":"vitest run --coverage","prepublishOnly":"pnpm run build"},"_npmUser":{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"},"_npmVersion":"11.10.0","description":"Local MCP server that gives AI coding agents deep structural knowledge of a JavaScript/TypeScript codebase.","directories":{},"_nodeVersion":"24.11.1","dependencies":{"zod":"^4.4.3","ignore":"^6.0.2","tree-sitter":"0.25.1","better-sqlite3":"^13.0.3","tree-sitter-javascript":"0.25.0","tree-sitter-typescript":"0.23.2"},"_hasShrinkwrap":false,"packageManager":"pnpm@10.24.0","devDependencies":{"tsup":"^8.3.5","eslint":"^9.17.0","vitest":"^2.1.8","prettier":"^3.4.2","vitepress":"^1.6.4","@eslint/js":"^9.17.0","typescript":"^5.7.2","@types/node":"^22.10.2","@vscode/ripgrep":"^1.18.0","typescript-eslint":"^8.18.1","@vitest/coverage-v8":"^2.1.8","@types/better-sqlite3":"^7.6.13"},"_npmOperationalInternal":{"tmp":"tmp/codebrain-mcp_0.1.0_1786198852035_0.9923095794580803","host":"s3://npm-registry-packages-npm-production"}},"1.0.0":{"name":"@aravindh-arumugam/codebrain-mcp","version":"1.0.0","keywords":["mcp","model-context-protocol","codebrain","typescript","javascript","tree-sitter","static-analysis","claude-code","cursor"],"license":"MIT","_id":"@aravindh-arumugam/codebrain-mcp@1.0.0","maintainers":[{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"}],"bin":{"codebrain-mcp":"dist/cli/main.js"},"dist":{"shasum":"0ceedd4c93ac88f4ff4d11630dc79d601508894f","tarball":"https://registry.npmjs.org/@aravindh-arumugam/codebrain-mcp/-/codebrain-mcp-1.0.0.tgz","fileCount":11,"integrity":"sha512-jucefcoHrYrUgDKsBwDkSysehEhnEhnDlczBRccWqeG7+kzL1kvxlp8hyh8q4nuxhXS4PJ+OFLrN/FqggrrgrQ==","signatures":[{"sig":"MEQCICHI6lJ7Yt+Sh2RkrOinrwExlyJGKIRbbTk5enUB0za0AiAfwjawVcqCKkaPd62jfaJvlpMd8L9fFWYXQHbIcq3suA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":1382088},"main":"./dist/index.js","pnpm":{"peerDependencyRules":{"allowedVersions":{"tree-sitter-javascript>tree-sitter":"0.25.1","tree-sitter-typescript>tree-sitter":"0.25.1"}},"onlyBuiltDependencies":["esbuild","@vscode/ripgrep"],"ignoredBuiltDependencies":["tree-sitter","tree-sitter-javascript","tree-sitter-typescript"]},"type":"module","types":"./dist/index.d.ts","module":"./dist/index.js","engines":{"node":">=18.18.0"},"exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./package.json":"./package.json"},"gitHead":"aa8a49e960cb95500a10e6e304416e1cf15e170a","scripts":{"dev":"tsup --watch","lint":"eslint .","test":"vitest run","bench":"pnpm run build && node bench/compare.mjs","build":"tsup","check":"pnpm run typecheck && pnpm run lint && pnpm run test","start":"node ./dist/cli/main.js","format":"prettier --write .","docs:dev":"vitepress dev docs","lint:fix":"eslint . --fix","typecheck":"tsc --noEmit","docs:build":"vitepress build docs","test:watch":"vitest","bench:sample":"pnpm run build && node bench/sample.mjs","docs:preview":"vitepress preview docs","format:check":"prettier --check .","test:coverage":"vitest run --coverage","prepublishOnly":"pnpm run build"},"_npmUser":{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"},"_npmVersion":"11.10.0","description":"Local MCP server that gives AI coding agents deep structural knowledge of a JavaScript/TypeScript codebase.","directories":{},"_nodeVersion":"24.11.1","dependencies":{"zod":"^4.4.3","ignore":"^6.0.2","tree-sitter":"0.25.1","better-sqlite3":"^13.0.3","tree-sitter-javascript":"0.25.0","tree-sitter-typescript":"0.23.2"},"_hasShrinkwrap":false,"packageManager":"pnpm@10.24.0","devDependencies":{"tsup":"^8.3.5","eslint":"^9.17.0","vitest":"^2.1.8","prettier":"^3.4.2","vitepress":"^1.6.4","@eslint/js":"^9.17.0","typescript":"^5.7.2","@types/node":"^22.10.2","@vscode/ripgrep":"^1.18.0","typescript-eslint":"^8.18.1","@vitest/coverage-v8":"^2.1.8","@types/better-sqlite3":"^7.6.13"},"_npmOperationalInternal":{"tmp":"tmp/codebrain-mcp_1.0.0_1786209962724_0.6005957391271028","host":"s3://npm-registry-packages-npm-production"}},"1.1.0":{"name":"@aravindh-arumugam/codebrain-mcp","version":"1.1.0","description":"Local MCP server that gives AI coding agents deep structural knowledge of a JavaScript/TypeScript codebase.","keywords":["mcp","model-context-protocol","codebrain","typescript","javascript","tree-sitter","static-analysis","claude-code","cursor"],"license":"MIT","type":"module","engines":{"node":">=18.18.0"},"bin":{"codebrain-mcp":"dist/cli/main.js"},"main":"./dist/index.js","module":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","import":"./dist/index.js"},"./package.json":"./package.json"},"scripts":{"build":"tsup","dev":"tsup --watch","start":"node ./dist/cli/main.js","typecheck":"tsc --noEmit","test":"vitest run","test:watch":"vitest","test:coverage":"vitest run --coverage","bench":"pnpm run build && node bench/compare.mjs","docs:dev":"vitepress dev docs","docs:build":"vitepress build docs","docs:preview":"vitepress preview docs","lint":"eslint .","lint:fix":"eslint . --fix","format":"prettier --write .","format:check":"prettier --check .","check":"pnpm run typecheck && pnpm run lint && pnpm run test","prepublishOnly":"pnpm run build","bench:sample":"pnpm run build && node bench/sample.mjs"},"dependencies":{"better-sqlite3":"^13.0.3","ignore":"^6.0.2","tree-sitter":"0.25.1","tree-sitter-javascript":"0.25.0","tree-sitter-typescript":"0.23.2","zod":"^4.4.3"},"devDependencies":{"@eslint/js":"^9.17.0","@types/better-sqlite3":"^7.6.13","@types/node":"^22.10.2","@vitest/coverage-v8":"^2.1.8","@vscode/ripgrep":"^1.18.0","eslint":"^9.17.0","prettier":"^3.4.2","tsup":"^8.3.5","typescript":"^5.7.2","typescript-eslint":"^8.18.1","vitepress":"^1.6.4","vitest":"^2.1.8"},"packageManager":"pnpm@10.24.0","pnpm":{"onlyBuiltDependencies":["esbuild","@vscode/ripgrep"],"ignoredBuiltDependencies":["tree-sitter","tree-sitter-javascript","tree-sitter-typescript"],"peerDependencyRules":{"allowedVersions":{"tree-sitter-javascript>tree-sitter":"0.25.1","tree-sitter-typescript>tree-sitter":"0.25.1"}}},"gitHead":"46b198f9f21493ec8e41a8624e978d6ee44523d2","_id":"@aravindh-arumugam/codebrain-mcp@1.1.0","_nodeVersion":"24.11.1","_npmVersion":"11.10.0","dist":{"integrity":"sha512-A7Fx77KYNgrBWUkwE4TsdGwfqDA5VwVHqAeoVwEYRVl+8Hd49nCtT7VmemtCyUYbKi+lbvY8H5UM0EXVtXg1jg==","shasum":"07eb3bb990b8ea78a56a35b7cc08f2a08511a2bd","tarball":"https://registry.npmjs.org/@aravindh-arumugam/codebrain-mcp/-/codebrain-mcp-1.1.0.tgz","fileCount":9,"unpackedSize":304595,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIQDn1a0tEnkBb2FiNp7QU8ZwMdCZlGOC0yaseDySVYyW1gIgbcbXe7LklfV1klz9O0e2cDU+HlNHh6rBpOcf8L/tBeE="}]},"_npmUser":{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"},"directories":{},"maintainers":[{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/codebrain-mcp_1.1.0_1786215853145_0.22469456797831766"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-08T14:20:51.899Z","modified":"2026-08-08T19:04:13.461Z","0.1.0":"2026-08-08T14:20:52.183Z","1.0.0":"2026-08-08T17:26:02.961Z","1.1.0":"2026-08-08T19:04:13.294Z"},"license":"MIT","keywords":["mcp","model-context-protocol","codebrain","typescript","javascript","tree-sitter","static-analysis","claude-code","cursor"],"description":"Local MCP server that gives AI coding agents deep structural knowledge of a JavaScript/TypeScript codebase.","maintainers":[{"name":"aravindh-arumugam","email":"aaravindh23cse@gmail.com"}],"readme":"# @aravindh-arumugam/codebrain-mcp\n\nA local [MCP](https://modelcontextprotocol.io) server that gives AI coding agents — Claude Code, Cursor,\nand any other MCP-compatible client — deep structural knowledge of a JavaScript/TypeScript codebase.\n\n> **Status: early development.** Phases 1–17 of 20 are implemented: package foundation, CLI, project\n> scanner, tree-sitter parsing, the normalized code model, symbol extraction, the SQLite index,\n> incremental indexing with a file watcher, lexical search, navigation, the module dependency graph,\n> the MCP server over stdio, safe + symbol-aware edit tools, symbol context, and worker-thread regex\n> safety with a terminable time budget. Phase 15 (semantic search) is deliberately deferred until the\n> language-level system is solid; it will be optional and pluggable. Phases 18–20 (security, docs,\n> publishing) are partly done: security hardening is in place and documented, publishing is not.\n> See [Roadmap](#roadmap).\n\n## Why this exists\n\nCoding agents currently explore repositories by grepping and reading whole files. That is slow, imprecise,\nand burns context on source the agent does not need. This package builds a local index of your codebase —\nsymbols, imports, exports, references, call relationships — and exposes it over MCP so an agent can ask\n\"where is `UserService` defined?\" or \"what calls `create_user`?\" and get a small, precise answer.\n\nEverything runs locally. No source code is uploaded anywhere, and no AI or embedding API is required.\n\n## Installation\n\nRun it directly, no install required:\n\n```bash\nnpx @codebrain/mcp\n```\n\nOr install globally:\n\n```bash\nnpm install -g @codebrain/mcp\ncodebrain-mcp\n```\n\n> Not published to npm yet. See [Development](#development) to run it from source.\n\n## Usage\n\n```bash\n# index the current directory\ncodebrain-mcp\n\n# index a specific project\ncodebrain-mcp /path/to/project\n\n# verbose diagnostics\ncodebrain-mcp ./apps/api --verbose\n```\n\n### Options\n\n| Option                | Description                                              |\n| --------------------- | -------------------------------------------------------- |\n| `[project-path]`      | Project root to index. Defaults to the CWD.              |\n| `-h`, `--help`        | Show help and exit.                                      |\n| `-v`, `--version`     | Show version and exit.                                   |\n| `-w`, `--watch`       | Keep running and reindex changed files.                  |\n| `--force`             | Reindex every file, ignoring the change check.           |\n| `--log-level <level>` | `silent`, `error`, `warn`, `info` (default), or `debug`. |\n| `--verbose`           | Shorthand for `--log-level debug`.                       |\n| `--quiet`             | Shorthand for `--log-level error`.                       |\n| `--`                  | Treat all following arguments as the project path.       |\n\nExit codes: `0` success, `1` runtime error, `2` usage error.\n\n## Claude Code setup\n\nThe server talks MCP over stdio. Setup takes **2 minutes**:\n\n### 1. Install the package:\n```bash\nnpm install --save-dev @aravindh-arumugam/codebrain-mcp\n```\n\n### 2. Create `.mcp.json` at your project root:\n```json\n{\n  \"mcpServers\": {\n    \"codebrain\": {\n      \"command\": \"node\",\n      \"args\": [\"./node_modules/@aravindh-arumugam/codebrain-mcp/dist/cli/main.js\", \".\"],\n      \"type\": \"stdio\"\n    }\n  }\n}\n```\n\n### 3. Add to `.claude/settings.json` (create if missing):\n```json\n{\n  \"enabledMcpjsonServers\": [\"codebrain\"]\n}\n```\n\n### 4. Restart Claude Code completely\nClose all Claude Code windows and reopen your project. The **codebrain** server should now appear in the MCP servers list.\n\n### Why this works\n- `.mcp.json` **defines** your MCP servers with their command and args\n- `settings.json` **approves** which servers Claude Code can use\n- No credentials needed — everything runs locally on your machine\n\n## Why use Codebrain MCP?\n\n| Task | Without MCP | With Codebrain MCP | Savings |\n|------|-------------|-------------------|---------|\n| **Find a symbol** | 3,500 tokens (grep) | 150 tokens | **95% ↓** |\n| **Find all references** | 4,500 tokens (grep + read files) | 400 tokens | **91% ↓** |\n| **Understand relationships** | 8,000+ tokens (multiple reads) | 550 tokens | **93% ↓** |\n| **Query speed** | 2-5 seconds | 150-400ms | **12x faster** |\n| **Context clarity** | Raw text matches | Structured results | **100% precision** |\n\n### Real Example\n**Question:** \"Show all uses of `DemandService`\"\n\n**With MCP** (400 tokens):\n```\nget_references(\"DemandService\")\n→ 20 semantic references (instantiations, method calls)\n```\n\n**Without MCP** (4,500 tokens):\n```\ngrep \"DemandService\" *.ts\n→ 240+ string matches (imports, types, comments, related classes)\n→ Claude reads 15-30 files to filter noise\n```\n\n### When to use it\n✅ **Refactoring** — \"What breaks if I rename this?\"  \n✅ **Impact analysis** — \"Who calls this function?\"  \n✅ **Cross-file understanding** — \"Where is this type defined?\"  \n✅ **Large codebases** — 1000+ files where grep is slow and expensive  \n\n### When grep might be faster\n⚡ Single-file questions (\"Show me the error handler\")  \n⚡ First run (no index built yet)  \n⚡ One-off pattern searches  \n\n**TL;DR:** For anything that touches multiple files, MCP saves hours of token usage and context.\n\n## Cursor setup\n\nAdd to `.cursor/mcp.json`:\n\n```json\n{\n  \"mcpServers\": {\n    \"codebrain\": {\n      \"command\": \"npx\",\n      \"args\": [\"-y\", \"@codebrain/mcp\", \"/path/to/project\"]\n    }\n  }\n}\n```\n\nNo credentials or API keys are involved in any configuration.\n\n## Supported languages\n\n| Language   | Extensions                  | Detected | Parsed | Grammar                      |\n| ---------- | --------------------------- | -------- | ------ | ---------------------------- |\n| JavaScript | `.js` `.mjs` `.cjs`         | Yes      | Yes    | tree-sitter-javascript       |\n| JSX        | `.jsx`                      | Yes      | Yes    | tree-sitter-javascript       |\n| TypeScript | `.ts` `.mts` `.cts` `.d.ts` | Yes      | Yes    | tree-sitter-typescript       |\n| TSX        | `.tsx`                      | Yes      | Yes    | tree-sitter-typescript (tsx) |\n\n`jsx` maps to the JavaScript grammar because tree-sitter's JavaScript grammar parses JSX natively.\nTypeScript is the opposite case: `.ts` and `.tsx` genuinely need different grammars, which is why they\nare separate language ids.\n\nLanguage is decided by file extension only. Sniffing a `.js` file to guess whether it \"really\" contains\nJSX would be a guess, and this package does not guess — tree-sitter's JavaScript grammar parses JSX anyway.\n\nFramework-specific intelligence (Next.js routes, NestJS modules, Express routers) is deliberately out of\nscope until the language-level system is solid.\n\n## What the scanner excludes\n\nSkipped directories by default: `node_modules`, `.git`, `dist`, `build`, `coverage`, `out`, `.next`,\n`.nuxt`, `.svelte-kit`, `.output`, `.turbo`, `.cache`, `.parcel-cache`, `.expo`, `.vercel`, `.netlify`,\n`.yarn`, `.pnpm-store`, `.code-intelligence`. Skipped file patterns: `*.min.js`, `*.bundle.js`, `*.map`.\n\n`.gitignore` is honoured, including nested `.gitignore` files scoped to their own subtree and negation\npatterns. Files over 1 MiB are skipped. Symlinks are **not** followed by default; when enabled they are\nre-validated against the project root after resolution, so a link cannot pull in files from outside.\n\nAnything skipped for a reason worth knowing about (too large, symlink, outside root, unreadable, limit\nreached) is reported rather than silently dropped. Routine exclusions are not listed individually — on a\nlarge repository that list would dwarf the result.\n\n## Parsing\n\nParsing uses the **native** tree-sitter bindings, not WASM. They ship N-API prebuilds for macOS, Linux\nand Windows on x64 and arm64, so a plain `npx` install compiles nothing and needs no build toolchain, and\nthey track current grammar releases. (The readily available `.wasm` builds carry older grammars that\nmis-parse modern TypeScript such as `accessor` fields.) Only two files import tree-sitter; everything\nabove them uses the normalized `syntax_node` type, so the backend stays replaceable.\n\n**Parsing is error-tolerant.** A file with a syntax error still produces a usable tree plus a populated\n`parse_errors` list — refusing to index a file because of one typo would lose far more than it protects.\nMeasured on a 650 KB `.d.ts` with 7 unparseable spots, the parser still recovered 265 interfaces, 77 type\naliases and 1,251 property signatures.\n\nTwo conventions are fixed at the parser boundary and hold everywhere above it:\n\n- **Positions are 1-based** for both line and column, as editors display them. tree-sitter is 0-based;\n  the conversion happens in exactly one place.\n- **Offsets are JavaScript string indices** (UTF-16 code units), not byte offsets, so\n  `source.slice(start_index, end_index)` is exact even for source containing emoji or accents. This was\n  verified against non-ASCII input rather than assumed.\n\nNodes are wrapped lazily, so inspecting a few nodes in a large file does not materialize the whole tree,\nand tree traversal uses an explicit stack so deeply nested source cannot overflow it.\n\n### Known grammar limitations\n\nThese are limits of the upstream grammar, not of this package, and they degrade gracefully — the rest of\nthe file still parses:\n\n- `abstract` used as a _property name_ (`interface x { abstract: boolean }`) fails to parse, because the\n  grammar treats it as a reserved modifier. Other contextual keywords may behave the same way.\n- Some ambient `export =` forms in `.d.cts` declaration files fail.\n\nMeasured on 4,313 real-world files (40.7 MB of source from `node_modules`): 0 hard failures, and 209\nfiles (4.8%) reported at least one syntax issue, essentially all of them `.d.ts` files using the\nconstructs above.\n\n## The code model\n\nBetween the grammar and the index sits a small, language-independent model. It is **not** a unified AST:\nit does not try to represent every construct in every language, only the facts an agent asks for.\n\n| Type             | What it records                                                                  |\n| ---------------- | -------------------------------------------------------------------------------- |\n| `code_file`      | path, language, size, content hash, package (for monorepos)                      |\n| `code_symbol`    | name, kind, location, parent, signature, export kind                             |\n| `relationship`   | `contains`, `imports`, `exports`, `calls`, `references`, `extends`, `implements` |\n| `code_import`    | specifier, imported/local name, resolved path, internal vs package               |\n| `code_export`    | exported name, local name, re-export source                                      |\n| `code_reference` | every occurrence of a name, with its enclosing symbol                            |\n\nSymbol kinds: `function`, `class`, `method`, `interface`, `type`, `variable`, `constant`, `enum`, `module`.\n\nTwo rules shape everything here:\n\n**Never invent a relationship.** When a call target cannot be resolved, the edge is stored with the name\nthat was actually written and _no_ target id, rather than pointed at a plausible-looking symbol. Edges\ncarry a confidence of `certain` (derived from syntax alone) or `resolved` (followed across an import).\nThere is deliberately no \"guessed\" level. An agent acting on a fabricated `calls` edge edits the wrong\nfunction, so a missing edge is much cheaper than a wrong one.\n\n**Symbol ids exclude line numbers.** An id is `<file>#<kind>:<container.path>`, for example\n`src/user.ts#method:user_service.create_user`. Editing one function therefore does not change the id of\nevery symbol below it, which is what lets incremental indexing diff a file instead of rewriting it.\n\n## Symbol extraction\n\nEvery declaration becomes a symbol with its kind, 1-based location, containment parent, export kind,\nambient flag, and a one-line signature with the body stripped\n(`async create_user(u: user): Promise<void>`). Containment produces `contains` edges, the one\nrelationship kind that is certain from syntax alone.\n\nThree judgement calls are worth stating, because they shape what you get back:\n\n**`const f = () => {}` is recorded as a function, not a constant.** It is the dominant declaration style\nin React and modern TypeScript, and someone searching for a function has to find it.\n\n**Function-local data variables are not indexed.** Names like `stack`, `index` or `base` are meaningless\nto search for and would bury real results — on this repository, excluding them took the constant count\nfrom 312 to 37. Functions and classes declared inside a function _are_ kept, since a named callable is\nworth looking up however it is scoped.\n\n**Nothing is named by guesswork.** A destructuring binding (`const { a, b } = obj`) binds several names\nat once, so no symbol is recorded rather than one invented name; the bindings stay visible as references.\nClass and interface _properties_ are not symbols either, because the model has no kind for them and\ninventing one would be a deviation. `export default function () {}` is recorded under the name `default`,\nwhich is how consumers actually address it.\n\nMeasured on 5,033 real files: 121,034 symbols and 48,532 containment edges, 0 extraction failures,\n117 files/sec, 154 MB peak heap. Throughput work is deferred to Phase 17.\n\n## The index\n\n`index_project()` walks the project and writes `<project>/.code-intelligence/index.db`. The directory is\ncreated with a `.gitignore` that excludes it, so the index is never committed by accident.\n\nTables: `files`, `symbols`, `relationships`, `references`, `imports`, `exports`, `metadata`, with indexes\non the lookup paths that matter — symbol name, name+kind, file, parent, and edges in both directions.\n\nEverything derived from a file cascades from its `files` row, so reindexing one file deletes its old\nsymbols, edges, imports, exports and references and reinserts them in a single transaction. A crash\nmid-write leaves the previous state, never a half-updated file, and stale rows cannot be orphaned. Files\nare processed one at a time and written in batched transactions, since SQLite fsyncs per commit.\n\nThe schema carries a version. On mismatch the index is **rebuilt rather than migrated**: it is derived\ndata that can always be regenerated from source, so migration code would be a liability with no upside.\n\nA file that fails to parse is recorded and skipped, never fatal — one bad file in a 10,000-file\nrepository must not cost the other 9,999.\n\n### Measured\n\nOn this repository (103 files, 655 symbols), index to first query:\n\n|                       |                   |\n| --------------------- | ----------------- |\n| Full index            | 1.6 s, 0 failures |\n| Database size         | 2.0 MB            |\n| Symbol lookup by name | ~0.05 ms          |\n\nThe stress case — 5,123 files including `node_modules`, which is not the normal\ncase:\n\n|                   |                                          |\n| ----------------- | ---------------------------------------- |\n| Full index        | ~230 s (parse + extract dominates)       |\n| Extracted         | 122,936 symbols, 370,534 relationships   |\n| References stored | 1,995,856 — every identifier occurrence  |\n| Database size     | 256 MB (55,655 distinct reference names) |\n\n`index_project` reports a `timings` breakdown (`scan_ms`, `analyze_ms`,\n`write_ms`, `graph_ms`) so slow runs can be diagnosed instead of guessed at. On\nthe stress case the split is roughly 1% scan, 80% parse/extract, 18% writes,\n2% graph resolution — tree-sitter parsing, not database work, is the cost.\n\nThe database is the derived cache, and its size is dominated by the\n`references` table (one row per identifier occurrence) and its indexes. The\nname column is interned through the `reference_names` dictionary (schema v4):\n2 million references collapse to 55,655 distinct names, which is what keeps\nthe index from ballooning on repositories with a lot of repetitive code.\n\n## Incremental indexing\n\nReindexing does not reparse the project. There is one code path, not a separate \"full\" and\n\"incremental\" mode — on an empty index every file is simply new — so the rare path cannot rot.\n\nDeciding what changed happens in two stages, because the cheap signal and the trustworthy signal are\ndifferent things:\n\n1. **Size and mtime**, with no file reads. A file matching both is skipped outright.\n2. **Content hash**, for anything that failed stage one. If the hash matches, the file was touched but\n   not edited and the database is left alone.\n\nThat split matters because mtime is a poor proxy for content: `git checkout`, `git stash`, a fresh clone\nand many editors rewrite mtimes on files whose bytes never changed. Trusting mtime alone would reindex\nthe world after every branch switch; hashing everything would mean reading every file every time.\n\nThere is a third rule that is easy to miss and causes silent staleness without it. A file whose mtime\nfalls within two seconds of when it was indexed is **never** trusted, even when size and mtime both\nmatch. Filesystem timestamps are coarse — commonly 1s, and 2s on FAT — so a file edited in the same tick\nit was read carries an identical mtime, and if the edit also left the size unchanged, nothing in the\nfingerprint reveals it. Git calls these entries racily clean and handles them the same way. This was\nfound by a test, not by reasoning: an edit that changed content while keeping the byte count went\nundetected.\n\nMeasured on this repository (60 files):\n\n| Run                                    | Time   | Files parsed             |\n| -------------------------------------- | ------ | ------------------------ |\n| First index                            | 249 ms | 60                       |\n| Nothing changed                        | 18 ms  | 0                        |\n| One file edited                        | 20 ms  | 1                        |\n| Three files touched, content identical | 24 ms  | 3 checked, **0 written** |\n\n### Watching\n\n`code-intelligence-mcp --watch` keeps the index current as you edit. Changes are debounced (300 ms by\ndefault) so that one save, or a formatter sweeping the project, produces a single reindex rather than\ndozens. Writes to `.code-intelligence/` are ignored — without that, the indexer's own database write\nwould trigger a reindex, which would write again, forever.\n\nThe watcher re-runs the normal indexing path rather than reimplementing change tracking. Re-walking\ndirectories costs a `stat` per file (18 ms here) while parsing is the expensive part, and reusing the\nnormal path means the watcher inherits the full ignore rules, including nested `.gitignore` files,\ninstead of drifting out of step with them.\n\nWatching uses `fs.watch` in recursive mode rather than adding a dependency. That is a real trade:\nrecursive mode is unavailable on some platform and Node combinations, and where it is, the failure is\nreported clearly at startup rather than silently watching nothing.\n\n## Search\n\nTwo entry points, both returning locations and signatures — never file contents.\n\n**`search_symbols`** matches symbol names. It runs exact, prefix and substring comparisons as separate\nqueries in that order, rather than one combined query: `name = ?` and `name LIKE 'q%'` can use the name\nindex while `LIKE '%q%'` cannot, so the cheap comparisons run first and an exact hit never pays for a\nfull scan. That ordering is also the ranking. Filters: kind, language, path, exported-only, and\nexclude-ambient (to drop `.d.ts` noise).\n\n**`search_code`** searches file contents, reading from disk rather than storing source — the index would\ndouble in size for data already on the filesystem, and go stale the moment a file changed. What the\nindex _does_ supply is the file list, already filtered by ignore rules, `.gitignore` and size limits, so\na search never wanders into `node_modules` or a minified bundle.\n\nEvery code match is annotated with **the symbol whose body encloses it**, innermost first. That is the\npart a plain grep cannot give you, and it is often enough to decide without opening the file:\n\n```\nsrc/search/pattern.ts:104  in method find\n   // Long lines are the amplifier for catastrophic backtracking, and are\n```\n\nMeasured on this repository (69 files, 403 symbols): 0.32 ms per symbol search, 6 ms for a full-text\nsearch across every indexed file.\n\n### Regex safety\n\nRegex search is supported, and the limits are worth stating precisely rather than waving at.\n\nJavaScript's regex engine backtracks and has no timeout, so a pattern nesting one unbounded quantifier\ninside another — `(a+)+`, `(\\w*)*` — takes exponential time. This was **measured, not assumed**:\n`(a+)+$` against a _30 character_ line takes 15 seconds, and every extra character roughly doubles it.\n\nThat measurement rules out the obvious mitigations. A line-length cap does not help, because the blowup\nhappens two orders of magnitude below any useful cap. Neither does a wall-clock budget checked between\nlines or files, because the engine never yields during a single failing match.\n\nSo regex matching runs in a **worker thread**, and the time budget can **terminate** that thread. The\nengine cannot be interrupted from the main thread, but terminating the thread is not an interruption —\nit stops the whole worker, backtrace and all. A pathological pattern is therefore bounded: the search\nstops at the budget and reports `timed_out` instead of hanging the server.\n\nThe nested-quantifier family is still **rejected before it runs**, because refusing costs less than a\nworker and gives a better error than a timeout. That check is a heuristic and is documented as one: it\ncatches what gets written by accident, while the worker terminates everything it misses — an overlapping\nalternation like `(a|a)+` backtracks just as badly, and it is bounded by the same budget.\n\nSeparately, all SQL is parameterised and `LIKE` wildcards in user queries are escaped, so searching for\n`%` finds the literal character instead of returning the entire index.\n\n## Module graph\n\nNavigation resolves within a symbol; the graph resolves across modules. **`get_dependencies`** answers\nwhat a file pulls in, and **`get_dependents`** answers what pulls it in — both grouped by target so two\nimport lines of the same module read as one dependency.\n\nDependencies are grouped exactly, never fuzzily. An import is a project file only when its specifier\nresolved during the graph pass: a package is reported as the bare specifier (`react`), an unresolvable\nrelative import as the specifier it is (`./style.css`). A re-exporting barrel (`export * from './user'`)\ncounts as a dependent of its source, because it genuinely depends on it without binding a name.\n\n## Symbol context\n\nThe read-only tools answer one question each, so exploring a single symbol can mean several round trips.\n**`get_symbol_context`** answers them all at once, keyed by symbol id: the symbol itself, the chain of\nparents that contain it and the members it contains, how heavily it is referenced and in how many files,\nwho calls it and what it calls, and what its file imports and is imported by.\n\nEvery list in the result is capped — a symbol referenced from a hundred files does not ship all of them.\nA `total` field beside each list states the true count, and the lists are ordered by weight (members by\nsource order, references and callers by usage), so what a model sees is the important part of the answer,\nnot a random slice.\n\n## MCP tools\n\nThe server exposes fourteen tools, each validating its arguments with a zod schema that doubles as its\nJSON Schema `inputSchema`. Eleven are read-only:\n\n`search_symbols` · `search_code` · `get_symbol` · `get_file_structure` · `find_definition` ·\n`find_references` · `get_callers` · `get_callees` · `get_dependencies` · `get_dependents` ·\n`get_symbol_context`\n\nThey wrap the search, navigation, context and graph layers directly. Results carry locations, names and\nsignatures — never file contents.\n\n**`get_symbol_context`** bundles the above questions into one result keyed by symbol id: what the symbol\nis, what contains it and what it contains, how heavily it is referenced and where, who calls it and what\nit calls, and what its file imports and is imported by. Every list is capped so one call cannot explode,\nand a `total` next to each capped list says how much was left out.\n\nThree edit tools write files and reindex automatically, and each is safe by construction:\n\n- **`apply_patch`** applies exact text replacements in order. An `old_text` that does not match, or\n  matches more than once, refuses the whole call rather than guessing.\n- **`replace_symbol`** splices new source over a symbol's declaration at the exact source range the index\n  recorded — no fuzzy matching.\n- **`rename_symbol`** renames a symbol at its declaration, every resolved reference, and the import and\n  export bindings that connect it to other files (including re-exports through barrels). Occurrences the\n  index could not resolve are left alone.\n\nEdits never escape the project root, and a file whose content drifted from the index since the last run\nrefuses any edit to it — every edit requires a fresh index. Because the index refreshes on each edit, a\nsecond edit is checked against the state the first one produced.\n\n## Architecture\n\n```\nproject files\n    -> file scanner            (Phase 2)\n    -> language detection      (Phase 2)\n    -> tree-sitter parser      (Phase 3)\n    -> normalized code model   (Phase 4-5)\n    -> sqlite index            (Phase 6-7)\n    -> relationship graph      (Phase 10)\n    -> search engine           (Phase 8-9)\n    -> mcp server              (Phase 11)\n    -> claude / cursor / other agents\n```\n\nKey design decision: there is **no unified AST**. Each language keeps its own tree-sitter grammar, and a\nthin normalization layer maps it into a language-independent code model. Only the normalization layer knows\nabout tree-sitter; nothing above it does.\n\nCurrently implemented:\n\n```\nsrc/\n  cli/         argv parsing, help text, CLI orchestration\n  core/        errors, logger, package metadata, project root + path containment\n  scanner/     file discovery, language detection, ignore rules\n  parser/      tree-sitter integration, normalized syntax nodes\n  model/       symbols, relationships, imports/exports/references\n  extract/     syntax tree -> symbols and contains edges\n  store/       sqlite schema, queries, row mapping\n  indexer/     scan -> parse -> extract -> store\n  navigation/  find_definition, find_references, find_callers, find_callees\n  context/     get_symbol_context\n  graph/       get_dependencies, get_dependents\n  search/      symbol and text search, regex safety\n  edit/        apply_patch, replace_symbol, rename_symbol\n  mcp/         stdio server: JSON-RPC protocol, tool registry\n  index.ts     public library exports\n```\n\nTwo path conventions hold throughout, so that index keys are identical on every operating system:\nabsolute **native** paths are used for filesystem access, and **POSIX** paths relative to the project root\nare used as identifiers in the index and in MCP responses.\n\nA file inside the project has exactly one canonical identity. A symlinked file is reported under its real\npath, never the link's, so an alias can neither shadow nor duplicate the real file in the index.\n\n## Privacy\n\n- All parsing and indexing happens on your machine.\n- The index is written to `.code-intelligence/` inside your project.\n- No network calls are made by the core package.\n- No OpenAI, Anthropic, Gemini, or embedding API is required or used.\n- Semantic search (Phase 15) will be strictly optional and pluggable, including fully local providers.\n\n## Limitations\n\nThis is static analysis. It will be honest about what it cannot know:\n\n- Dynamic dispatch, `eval`, and runtime-computed property access cannot be resolved statically.\n- Call relationships through higher-order functions or dependency injection containers are often\n  undecidable; the indexer records a relationship only when it can determine it confidently.\n- Type-level inference is not performed — this is not a TypeScript compiler. Relationships come from\n  syntax, not from the type checker.\n- Generated code, minified bundles, and very large files are skipped.\n- The grammar itself has gaps; see [Known grammar limitations](#known-grammar-limitations).\n\nNo \"N% token reduction\" claims are made anywhere in this project unless they have been measured.\n\n## Development\n\n```bash\npnpm install\npnpm run build      # bundle to dist/ with tsup\npnpm run typecheck  # tsc --noEmit\npnpm run test       # vitest\npnpm run lint       # eslint\npnpm run check      # typecheck + lint + test\n```\n\nRun the CLI from source without building:\n\n```bash\nnode --experimental-strip-types src/cli/main.ts --help\n# or, after a build\nnode dist/cli/main.js /path/to/project\n```\n\n### Docs site\n\nThe documentation site lives in `docs/` and is built with VitePress:\n\n```bash\npnpm run docs:dev      # local dev server\npnpm run docs:build    # static build to docs/.vitepress/dist\npnpm run docs:preview  # preview the built site\n```\n\nTo deploy to GitHub Pages under a subpath (e.g. `/codebrain-mcp/`), set `base` in\n`docs/.vitepress/config.ts`; VitePress rewrites internal links automatically.\n\n### Conventions\n\n- TypeScript strict mode, plus `noUncheckedIndexedAccess` and `exactOptionalPropertyTypes`.\n- `snake_case` for variables, functions, types, and interfaces (enforced by ESLint).\n- No `any`. No silently swallowed errors. Errors carry a machine-readable code and a fix hint.\n- Small modules; no abstraction without a second caller.\n- **stdout is reserved for the MCP protocol.** All diagnostics go to stderr.\n\n## Testing\n\n```bash\npnpm run test\npnpm run test:coverage\n```\n\nTests live in `tests/`, mirroring the `src/` layout. Fixture repositories (React frontend, Node backend,\nshared package) arrive with the parser phases.\n\n## Publishing\n\n```bash\npnpm run check\npnpm run build\nnpm publish --access public\n```\n\n`prepublishOnly` runs the build. CI runs typecheck, lint, tests and the build on Linux, macOS and Windows.\nThe published package contains only `dist/`, `README.md`, `LICENSE`, `SECURITY.md`, and `CHANGELOG.md`.\n\n## Roadmap\n\n| Phase | Scope                     | Status                         |\n| ----- | ------------------------- | ------------------------------ |\n| 1     | Project foundation + CLI  | Done                           |\n| 2     | Project scanner           | Done                           |\n| 3     | Tree-sitter parsing       | Done                           |\n| 4     | Normalized code model     | Done                           |\n| 5     | Symbol extraction         | Done                           |\n| 6     | SQLite index              | Done                           |\n| 7     | Incremental indexing      | Done                           |\n| 8     | Code search               | Done                           |\n| 9     | Navigation                | Done                           |\n| 10    | Relationship graph        | Done                           |\n| 11    | MCP server                | Done                           |\n| 12-13 | Safe + symbol-aware edits | Done                           |\n| 14    | Symbol context            | Done                           |\n| 15    | Semantic search           | Deferred (optional)            |\n| 16-17 | Performance               | Done                           |\n| 18-20 | Security, docs, publish   | Security done; publish pending |\n\n## License\n\n[MIT](./LICENSE)\n","readmeFilename":"README.md"}