{"_id":"@adborroto/semantic-search-mcp","_rev":"4-60d6fd0a15afbcf5f8d6b05700b12dd3","name":"@adborroto/semantic-search-mcp","dist-tags":{"latest":"0.3.0"},"versions":{"0.1.0":{"name":"@adborroto/semantic-search-mcp","version":"0.1.0","keywords":["mcp","model-context-protocol","semantic-search","vector-search","embeddings","rag","retrieval","lancedb","transformers","local-first","offline","cli"],"author":{"url":"https://github.com/adborroto","name":"adborroto"},"license":"MIT","_id":"@adborroto/semantic-search-mcp@0.1.0","maintainers":[{"name":"adborroto","email":"adborroto90@gmail.com"}],"homepage":"https://github.com/adborroto/semantic-search-mcp#readme","bugs":{"url":"https://github.com/adborroto/semantic-search-mcp/issues"},"bin":{"semantic-search":"src/index.js"},"dist":{"shasum":"19c8b8d3d65ac2e6aedca4cb19a3f69e743e04d3","tarball":"https://registry.npmjs.org/@adborroto/semantic-search-mcp/-/semantic-search-mcp-0.1.0.tgz","fileCount":23,"integrity":"sha512-DZ8OF2v3ofFiN48o/Q4t2l216JCsvKfQmG8Z4VN/T4sN0s37I0sJdv/a/CO/gJ/SETGZr4id/Yai6bmnwdQXzQ==","signatures":[{"sig":"MEYCIQDFFL922sSOri5WgkreiYVnt7rXNWOEgj9ge+0j0+tLDAIhAPMxDj5IdTkZwoymjhW3XwqsJL83INkriWtqRJxUP10i","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":81230},"main":"src/index.js","type":"module","engines":{"node":">=22"},"gitHead":"7168f90eda1d008ae961c750bcc03fe19fde1c7a","scripts":{"lint":"eslint .","test":"node --test \"test/*.test.js\"","test:unit":"SS_SKIP_INTEGRATION=1 node --test \"test/*.test.js\""},"_npmUser":{"name":"adborroto","email":"adborroto90@gmail.com"},"overrides":{"sharp":"^0.35.3"},"repository":{"url":"git+https://github.com/adborroto/semantic-search-mcp.git","type":"git"},"_npmVersion":"11.16.0","description":"RAG-lite: local semantic search over your files, as a CLI and an MCP server. No LLM, no API keys, no servers.","directories":{},"_nodeVersion":"26.3.0","allowScripts":{"sharp@0.35.3":true,"protobufjs@7.6.5":true,"onnxruntime-node@1.24.0":true},"dependencies":{"zod":"^4.4.3","ignore":"^7.0.6","mammoth":"^1.12.0","commander":"^15.0.0","pdf-parse":"^2.4.5","@lancedb/lancedb":"^0.31.0","@huggingface/transformers":"^3.8.1","@modelcontextprotocol/sdk":"^1.29.0"},"_hasShrinkwrap":false,"devDependencies":{"eslint":"^9.39.0"},"_npmOperationalInternal":{"tmp":"tmp/semantic-search-mcp_0.1.0_1787062180043_0.7024918865668544","host":"s3://npm-registry-packages-npm-production"}},"0.2.0":{"name":"@adborroto/semantic-search-mcp","version":"0.2.0","keywords":["mcp","model-context-protocol","semantic-search","vector-search","embeddings","rag","retrieval","lancedb","transformers","local-first","offline","cli"],"author":{"url":"https://github.com/adborroto","name":"adborroto"},"license":"MIT","_id":"@adborroto/semantic-search-mcp@0.2.0","maintainers":[{"name":"adborroto","email":"adborroto90@gmail.com"}],"homepage":"https://github.com/adborroto/semantic-search-mcp#readme","bugs":{"url":"https://github.com/adborroto/semantic-search-mcp/issues"},"bin":{"semantic-search":"src/index.js"},"dist":{"shasum":"252cfdf6d04077efbed6fd1d7020d77969bab7e8","tarball":"https://registry.npmjs.org/@adborroto/semantic-search-mcp/-/semantic-search-mcp-0.2.0.tgz","fileCount":23,"integrity":"sha512-zIp4qtxXiIQla9ZGgfvKoagZfGLRmW+7lsB5IRSjHiQUhk5KU3lJhOog8bpvI4NxBEZFz+4THXL60OXJsLvMGQ==","signatures":[{"sig":"MEYCIQCP+LjpMAcVuih08tkFxiRiA8qeDg4VyMV/TguAMXQeLwIhANAYOt9EYg9D1K8GdyZ8ywj/+nancYhF+1ctht3hnLUZ","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":83214},"main":"src/index.js","type":"module","engines":{"node":">=22"},"gitHead":"69942f9dbbaee083fa408a2dfb578a536e2404cc","scripts":{"lint":"eslint .","test":"node --test \"test/*.test.js\"","test:unit":"SS_SKIP_INTEGRATION=1 node --test \"test/*.test.js\""},"_npmUser":{"name":"adborroto","email":"adborroto90@gmail.com"},"overrides":{"sharp":"^0.35.3"},"repository":{"url":"git+https://github.com/adborroto/semantic-search-mcp.git","type":"git"},"_npmVersion":"11.16.0","description":"RAG-lite: local semantic search over your files, as a CLI and an MCP server. No LLM, no API keys, no servers.","directories":{},"_nodeVersion":"26.3.0","allowScripts":{"sharp@0.35.3":true,"protobufjs@7.6.5":true,"onnxruntime-node@1.24.0":true},"dependencies":{"zod":"^4.4.3","ignore":"^7.0.6","mammoth":"^1.12.0","commander":"^15.0.0","pdf-parse":"^2.4.5","@lancedb/lancedb":"^0.31.0","@huggingface/transformers":"^3.8.1","@modelcontextprotocol/sdk":"^1.29.0"},"_hasShrinkwrap":false,"devDependencies":{"eslint":"^9.39.0"},"_npmOperationalInternal":{"tmp":"tmp/semantic-search-mcp_0.2.0_1787063852739_0.8436190732377218","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@adborroto/semantic-search-mcp","version":"0.2.1","keywords":["mcp","model-context-protocol","semantic-search","vector-search","embeddings","rag","retrieval","lancedb","transformers","local-first","offline","cli"],"author":{"url":"https://github.com/adborroto","name":"adborroto"},"license":"MIT","_id":"@adborroto/semantic-search-mcp@0.2.1","maintainers":[{"name":"adborroto","email":"adborroto90@gmail.com"}],"homepage":"https://github.com/adborroto/semantic-search-mcp#readme","bugs":{"url":"https://github.com/adborroto/semantic-search-mcp/issues"},"bin":{"semantic-search":"src/index.js"},"dist":{"shasum":"f2c77b70915e5500c8a819c36df55c0b6e5b3203","tarball":"https://registry.npmjs.org/@adborroto/semantic-search-mcp/-/semantic-search-mcp-0.2.1.tgz","fileCount":23,"integrity":"sha512-cnTont2ZCL4fkI2YfT2oyp3zRnyJSHD3z7yAXkfS7Q8yoMuVMb8Q1zya9b0F/XAxeOc2qTnE3rnsa5ugd/RNIw==","signatures":[{"sig":"MEUCIQD81oNMYVPPRf+mDgbdkqEmt9ROvRS802hXm1LxXZ19UAIgDykMWXcVoMDuemddupYfYJQd55qmv5s1okJluvcXD54=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":84666},"main":"src/index.js","type":"module","engines":{"node":">=22"},"gitHead":"7bd4ec3137f8c4d800eb2e219466d5986f0c1035","scripts":{"lint":"eslint .","test":"node --test \"test/*.test.js\"","test:unit":"SS_SKIP_INTEGRATION=1 node --test \"test/*.test.js\""},"_npmUser":{"name":"adborroto","email":"adborroto90@gmail.com"},"overrides":{"sharp":"^0.35.3"},"repository":{"url":"git+https://github.com/adborroto/semantic-search-mcp.git","type":"git"},"_npmVersion":"11.16.0","description":"RAG-lite: local semantic search over your files, as a CLI and an MCP server. No LLM, no API keys, no servers.","directories":{},"_nodeVersion":"26.3.0","allowScripts":{"sharp@0.35.3":true,"protobufjs@7.6.5":true,"onnxruntime-node@1.24.0":true},"dependencies":{"zod":"^4.4.3","ignore":"^7.0.6","mammoth":"^1.12.0","commander":"^15.0.0","pdf-parse":"^2.4.5","@lancedb/lancedb":"^0.31.0","@huggingface/transformers":"^3.8.1","@modelcontextprotocol/sdk":"^1.29.0"},"_hasShrinkwrap":false,"devDependencies":{"eslint":"^9.39.0"},"_npmOperationalInternal":{"tmp":"tmp/semantic-search-mcp_0.2.1_1787066071527_0.8385216945424085","host":"s3://npm-registry-packages-npm-production"}},"0.3.0":{"name":"@adborroto/semantic-search-mcp","version":"0.3.0","description":"RAG-lite: local semantic search over your files, as a CLI and an MCP server. No LLM, no API keys, no servers.","main":"src/index.js","type":"module","bin":{"semantic-search":"src/index.js"},"engines":{"node":">=22"},"scripts":{"test":"node --test \"test/*.test.js\"","lint":"eslint .","test:unit":"SS_SKIP_INTEGRATION=1 node --test \"test/*.test.js\""},"keywords":["mcp","model-context-protocol","semantic-search","vector-search","embeddings","rag","retrieval","lancedb","transformers","local-first","offline","cli"],"author":{"name":"adborroto","url":"https://github.com/adborroto"},"license":"MIT","repository":{"type":"git","url":"git+https://github.com/adborroto/semantic-search-mcp.git"},"homepage":"https://github.com/adborroto/semantic-search-mcp#readme","bugs":{"url":"https://github.com/adborroto/semantic-search-mcp/issues"},"dependencies":{"@huggingface/transformers":"^3.8.1","@lancedb/lancedb":"^0.31.0","@modelcontextprotocol/sdk":"^1.29.0","commander":"^15.0.0","ignore":"^7.0.6","mammoth":"^1.12.0","pdf-parse":"^2.4.5","zod":"^4.4.3"},"devDependencies":{"eslint":"^9.39.0"},"overrides":{"sharp":"^0.35.3"},"allowScripts":{"protobufjs@7.6.5":true,"onnxruntime-node@1.24.0":true,"sharp@0.35.3":true},"gitHead":"2d183de37342fbbc4ff127d511945a788945f69f","_id":"@adborroto/semantic-search-mcp@0.3.0","_nodeVersion":"26.3.0","_npmVersion":"11.16.0","dist":{"integrity":"sha512-FtK74PZj87Gn7oEgRBTwbYkuRyDA1YcZCmB2dK+gl68EqgI/0y4Icc8SLq9AXotK9mPRh6oCewRrZVQOfdeDUw==","shasum":"171e8fab45bf9acc55bb707ab6760261656c070d","tarball":"https://registry.npmjs.org/@adborroto/semantic-search-mcp/-/semantic-search-mcp-0.3.0.tgz","fileCount":23,"unpackedSize":104704,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIDpM2uivPgrY/8okAqj/hjr66rgywmwqk8XsiU3HoY+vAiAYApJBZI1mVtZTT8CH0xsrUQpKdyhonUsk+I//JkcrUQ=="}]},"_npmUser":{"name":"adborroto","email":"adborroto90@gmail.com"},"directories":{},"maintainers":[{"name":"adborroto","email":"adborroto90@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/semantic-search-mcp_0.3.0_1787067990366_0.17765323250022025"},"_hasShrinkwrap":false}},"time":{"created":"2026-08-18T14:09:39.865Z","modified":"2026-08-18T15:46:30.766Z","0.1.0":"2026-08-18T14:09:40.172Z","0.2.0":"2026-08-18T14:37:32.917Z","0.2.1":"2026-08-18T15:14:31.688Z","0.3.0":"2026-08-18T15:46:30.551Z"},"bugs":{"url":"https://github.com/adborroto/semantic-search-mcp/issues"},"author":{"name":"adborroto","url":"https://github.com/adborroto"},"license":"MIT","homepage":"https://github.com/adborroto/semantic-search-mcp#readme","keywords":["mcp","model-context-protocol","semantic-search","vector-search","embeddings","rag","retrieval","lancedb","transformers","local-first","offline","cli"],"repository":{"type":"git","url":"git+https://github.com/adborroto/semantic-search-mcp.git"},"description":"RAG-lite: local semantic search over your files, as a CLI and an MCP server. No LLM, no API keys, no servers.","maintainers":[{"name":"adborroto","email":"adborroto90@gmail.com"}],"readme":"# semantic-search-mcp\n\n[![CI](https://github.com/adborroto/semantic-search-mcp/actions/workflows/ci.yml/badge.svg)](https://github.com/adborroto/semantic-search-mcp/actions/workflows/ci.yml)\n[![Security](https://github.com/adborroto/semantic-search-mcp/actions/workflows/security.yml/badge.svg)](https://github.com/adborroto/semantic-search-mcp/actions/workflows/security.yml)\n[![npm](https://img.shields.io/npm/v/@adborroto/semantic-search-mcp)](https://www.npmjs.com/package/@adborroto/semantic-search-mcp)\n[![node](https://img.shields.io/node/v/@adborroto/semantic-search-mcp)](https://nodejs.org)\n[![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](./LICENSE)\n\nA tiny, self-contained **RAG-lite retrieval engine**: it indexes files on disk and answers\n\"what's semantically relevant to this query\" — nothing more. It does **not** call an LLM and\ndoes **not** generate answers. It hands back the most relevant text chunks (file, line, score)\nso that whatever consumes it — a human, a script, or an LLM through MCP — can decide what to do\nwith them.\n\nEverything runs locally and offline after the first run:\n\n- **Embeddings**: [`@huggingface/transformers`](https://github.com/huggingface/transformers.js)\n  running `Xenova/all-MiniLM-L6-v2` with int8-quantized weights on CPU. No GPU, no API key, no\n  network calls at query time.\n- **Vector store**: [`@lancedb/lancedb`](https://lancedb.github.io/lancedb/) — an embedded,\n  file-backed vector database. No server process, no Docker.\n- **Interfaces**: a CLI and a stdio [MCP](https://modelcontextprotocol.io/) server, so any\n  MCP-aware agent (Claude Code, Cursor, Zed, …) can search your corpus directly.\n\n## Quickstart\n\n```bash\nnpm install -g @adborroto/semantic-search-mcp\n\nsemantic-search add ~/code/my-project      # add a folder to the corpus\nsemantic-search index                      # embed it (incremental on later runs)\nsemantic-search search \"how does the retry logic work\"\n```\n\nThat's the whole setup. There is no config file to write by hand — `add` creates and manages it\nfor you. To try it without installing anything:\n\n```bash\nnpx @adborroto/semantic-search-mcp add ~/code/my-project\n```\n\n> **Heads up on install size: ~950MB of dependencies**, plus a ~25MB embedding model\n> downloaded on first use. Almost all of it is native binaries you can't avoid at this layer —\n> `@lancedb/lancedb` (~430MB including its platform binary) and the ONNX runtime (~300MB, which\n> ships builds for every platform in one package). Both are cached once; everything after the\n> first run is offline.\n\n## Requirements\n\n- Node.js **>= 22** (`node:sqlite`, used by the fallback backend, is only stable from 22).\n- ~950MB disk for dependencies and ~25MB for the embedding model, plus roughly 1–3 KB per\n  indexed chunk.\n- No GPU, no external services, no database server.\n\n## Why \"RAG-lite\"\n\nA full RAG pipeline is: retrieve chunks → feed them to an LLM → LLM writes an answer. This\nproject stops at step one. That keeps it simple, fast, cheap to run, and easy to reason about —\nand it composes cleanly with whatever LLM or agent framework you're already using, instead of\nbundling its own opinionated generation layer.\n\n## Managing the corpus\n\n```bash\nsemantic-search add ~/code/api ~/notes     # add one or more folders\nsemantic-search list                       # show what's configured\nsemantic-search remove api                 # by folder name...\nsemantic-search remove ~/notes             # ...or by path\nsemantic-search config                     # where config + index actually live\n```\n\n`add` validates that each path is a real directory, resolves it to an absolute path, and skips\nduplicates (including the same directory reached through a symlink). `remove` also purges that\nfolder's chunks from the index, so its content stops appearing in results — pass `--keep-index`\nif you want to drop it from the corpus but keep it searchable.\n\n### Where things are stored\n\nConfig and index follow the XDG base directory spec, so they survive upgrades and are shared\nby every install method:\n\n| What | Location |\n|---|---|\n| Config | `~/.config/semantic-search/config.json` |\n| Index + model cache | `~/.local/share/semantic-search/` |\n\nOverride any of it with `SS_CONFIG_PATH`, `SS_INDEX_DIR`, `SS_MODEL_CACHE_DIR`, or the standard\n`XDG_CONFIG_HOME` / `XDG_DATA_HOME`. `SS_STORE_BACKEND=sqlite` forces the fallback backend.\n\n> **The index contains verbatim text of everything you indexed.** If you point this at private\n> code, `~/.local/share/semantic-search/` holds that content in plaintext. Never commit it, and\n> don't attach it to a bug report.\n\nEvery option is documented in [`src/config.js`](./src/config.js) — chunk sizing, ignore\npatterns, model name, top-k, concurrency. Editing `config.json` directly still works for those;\n`add`/`remove` preserve any keys they don't own.\n\n## Usage\n\n### Index\n\n```bash\nsemantic-search index                      # all configured folders\nsemantic-search index ~/code/one-project   # just this folder, ignoring config\nsemantic-search index --force              # reprocess everything\n```\n\nIndexing is **incremental**: unchanged files are skipped by modification time, files whose\n*content* didn't actually change (just touched) skip re-embedding, and files deleted from disk\nget pruned from the index. Only what actually changed gets reprocessed.\n\nWith several folders configured, `index` walks them in sequence with a per-folder header and a\ncombined total:\n\n```\n[1/3] my-api  /home/me/code/my-api  ─────────────────────────────\n  ↺ indexed   src/auth/middleware.js  (8 chunks)\n  2 indexed  1,203 skipped  16 chunks  4.1s\n\n[2/3] my-app  /home/me/code/my-app  ─────────────────────────────\n  ...\n\n──────────────────────────────────────────────────────────────\ntotal  5 indexed  3,891 skipped  0 deleted  41 chunks  12.3s\n```\n\nEach `index <path>` call only prunes stale entries for files *under that path*, so indexing\nfolder B never touches folder A's entries.\n\nUseful flags: `--max-files <n>` stops after N new files (bounds memory on huge corpora),\n`--concurrency <n>` sets parallelism, `--verbose` logs each file to stderr.\n\n### Search\n\n```bash\nsemantic-search search \"how does the retry logic work\" -k 5\n```\n\nPrints a table of file path, line number, score, and a text preview.\n\nRetrieval is **hybrid**: the query goes to two independent arms — a vector search over the\nembeddings, and a BM25 full-text search over the same chunks — and the two rankings are fused\nwith [Reciprocal Rank Fusion](https://plg.uwaterloo.ca/~gvcormac/cormacksigir09-rrf.pdf). The\narms fail differently: the vector arm misses exact identifiers, error strings and config keys it\nhas no semantic handle on; the lexical arm misses paraphrases. Running both is a recall fix, and\nfusing on *rank* rather than score keeps an unbounded BM25 score from drowning out cosine\nsimilarity.\n\nSet `\"hybridSearch\": false` in `config.json` for vector-only retrieval, and `\"rrfK\"` to tune\nRRF's rank-smoothing constant (default 60, from the paper).\n\n### What gets indexed\n\nPoint it at a folder and **everything inside it is indexed, recursively**. There is no\nallow-list of \"supported\" file extensions — `.dart`, `.kt`, `.java`, `.tsx`, `.sql`, `.erb` and\nanything else textual are all indexed as-is, with `.pdf` and `.docx` going through a parser\nfirst.\n\nFour things are excluded:\n\n1. **Whatever git ignores**, if the folder is a git repo. `.gitignore` is honored at any depth,\n   along with `.git/info/exclude`, your global excludes file, and negation patterns\n   (`!keep.this`). This is delegated to `git ls-files` rather than reimplemented, so it matches\n   git exactly — which means generated and vendored output your project already ignores stays\n   out of the index without you maintaining a second list.\n2. **Your `.indexignore` rules** (see below), for content that *is* committed but shouldn't be\n   searchable — fixtures, snapshots, a checked-in secrets template.\n3. **Binary files**, by extension (images, archives, fonts, compiled objects, model weights) and\n   by content — a NUL byte in the first 4KB means binary, the same heuristic `grep -I` uses.\n   This is a safety measure to keep non-text bytes out of the tokenizer, not a judgement about\n   what is worth indexing.\n4. **Files over 500,000 bytes** (`maxFileSizeBytes`), which is the main guard against a\n   generated single-line megabyte file exhausting memory.\n\nSymlinks are skipped rather than followed, so a link planted inside a folder can't pull outside\ncontent into the index.\n\nFor folders that are *not* git repos there is no `.gitignore` to lean on, so a small built-in\nlist (`node_modules/`, `.git/`, `dist/`, `build/`, `coverage/`, `vendor/`, …) still applies.\n\nTo exclude more, drop a gitignore-style `.indexignore` in either place:\n\n- **inside a folder you index** — patterns are relative to that folder;\n- **next to your config** (`~/.config/semantic-search/.indexignore`) — applies everywhere.\n\nSee [`.indexignore.example`](./.indexignore.example) for a starting point covering iOS, Android,\nFlutter, Ruby, and JVM build artifacts.\n\n## MCP server\n\n```bash\nsemantic-search mcp\n```\n\nStarts a stdio MCP server exposing six tools.\n\n**`search(query, k?)`** — semantic search, returns raw JSON:\n```\n[{ filePath, text, score, offset, startLine }, ...]\n```\n\n**`gather(query, k?, contextLines?)`** — same search, returned as a single formatted markdown\nblock ready to drop into a context window:\n\n````markdown\n### [1/5]  my-api  ·  src/auth/session.js  ·  line 42  ·  score 0.923\n```\n...chunk text...\n```\n````\n\n`contextLines` (default 0) reads N extra lines around each chunk from the source file — useful\nwhen a chunk boundary cuts off context you need.\n\n**`list_folders()`** — every configured folder with its name and absolute path. A good first call\nso the agent knows what corpus exists.\n\n**`cat_file(filePath, startLine?, endLine?)`** — read a file by absolute path, as returned by\n`search`/`gather`. Confined to the configured folders (see [Security](#security)).\n\n**`grep(pattern, folder?, fileGlob?, caseSensitive?, maxResults?)`** — literal or regex search\nacross the corpus, for when you need exact matches rather than similarity. Filtered against the\nexact same file list the indexer would index, so gitignored and `.indexignore`d files can't leak\nthrough exact-match search.\n\n```\nmy-api  ·  src/auth/session.js:42  export function createSession(user) {\n```\n\n**`index(root?, force?, maxFiles?, concurrency?)`** — trigger an incremental reindex, so an\nagent can refresh the corpus without shelling out.\n\nAll search tools share the same ranking and file-resolution code as the CLI; neither\nreimplements it.\n\n### Registering with an MCP client\n\nClaude Code:\n\n```bash\nclaude mcp add --scope user semantic-search -- semantic-search mcp\nclaude mcp list   # should show \"✔ Connected\"\n```\n\nAny client that takes a JSON server definition:\n\n```json\n{\n  \"mcpServers\": {\n    \"semantic-search\": {\n      \"command\": \"semantic-search\",\n      \"args\": [\"mcp\"]\n    }\n  }\n}\n```\n\nPrefer a global install over `npx` here: a bare `npx` re-resolves the package every time the\nserver launches, adding startup latency and picking up upgrades unannounced. If you do use\n`npx`, pin the version — `npx -y @adborroto/semantic-search-mcp@0.1.0 mcp`.\n\nNew MCP servers are usually only picked up when a session starts, so start a fresh session\nafter registering.\n\n## Security\n\nThis is a **local, single-user tool** with a simple trust model: anything inside a configured\nfolder is readable by any MCP client that can reach the server.\n\n- `cat_file` refuses paths outside the configured folders, resolving symlinks first so a link\n  planted inside a folder can't be used to escape it.\n- `grep` is filtered against the same file list the indexer builds — git's ignore rules plus\n  your `.indexignore` — so files deliberately excluded from indexing don't leak through\n  exact-match search instead.\n- Subprocesses are spawned with argv arrays (never a shell), so patterns can't inject commands.\n\nGiven that, **don't point it at a corpus you wouldn't hand to your LLM provider** — chunks are\nreturned to whatever client asked for them. See [SECURITY.md](./SECURITY.md).\n\n## How it works\n\n### File discovery\n\nThe rule is \"index everything under the folder\", and the only interesting part is what *not* to\nindex. Rather than reimplementing git's ignore semantics — nested `.gitignore` files, negations,\n`info/exclude`, the global excludes file — a git root is enumerated with:\n\n```\ngit ls-files -z --cached --others --exclude-standard\n```\n\nTracked files plus untracked-but-not-ignored ones, scoped to the directory it runs in. Anything\ngit ignores is absent by construction. Non-git folders fall back to a plain recursive walk with\nthe built-in pattern list.\n\nThe same function backs both the indexer and the MCP `grep` tool\n([`src/ignoreRules.js`](./src/ignoreRules.js)). That's deliberate: `grep` shells out to real\n`grep -r`, which happily reports hits inside gitignored build output, so it filters its results\nagainst the indexer's own file list. If the two derived their rules separately they would drift,\nand the ignore list would stop being a boundary.\n\n### Hybrid retrieval\n\nA query runs through two arms in parallel:\n\n- **Vector** — embed the query, take the nearest neighbours by cosine distance, then reorder that\n  list with a small lexical boost for chunks containing the literal query terms.\n- **Lexical** — BM25 over the same chunk text, via a LanceDB full-text index (the `sqlite`\n  fallback computes BM25 in JS, since `node:sqlite` isn't guaranteed to ship FTS5).\n\nThe two rankings are fused with RRF: each list contributes `1 / (60 + rank)` to every chunk it\nreturns, and the contributions are summed. Fusing on rank rather than score is the point —\ncosine sits in `[-1, 1]` while BM25 is unbounded above, so adding or averaging raw scores lets\none arm silently overwhelm the other depending on corpus size.\n\nWhy two arms at all: a lexical boost applied to the vector arm's output can only *reorder* what\nthe vector query already returned. A chunk whose sole signal is an exact term match — an error\ncode, a symbol name, a config key with no semantic neighbourhood — was unreachable if it fell\noutside the vector pool. The lexical arm retrieves it independently. That's a recall fix, not a\nreranking one, and it is why search scores now look like `0.03` rather than `0.9`: they are RRF\nsums, not cosine similarities. Only their ordering is meaningful.\n\nThe full-text index is rebuilt at the end of each indexing run, because an FTS index doesn't\ncover rows added after it was built — otherwise the chunks a run just wrote would be invisible\nto the lexical arm.\n\n### Chunking\n\nText is split into paragraphs, then packed greedily into chunks of about **200 tokens** with\n**~35 tokens of overlap**, counted with the embedding model's *real* tokenizer rather than a\ncharacter-count approximation. This isn't arbitrary: `all-MiniLM-L6-v2` has a **256 token**\nwindow and silently truncates anything longer, so chunks are sized to fit inside it with margin\nfor the `[CLS]`/`[SEP]` tokens. Overlap is additionally capped so that overlap plus the next\nparagraph can never breach that limit — otherwise a chunk's tail would be dropped at embed time\nwhile still being returned by `search`.\n\nA single paragraph larger than the hard limit (a minified bundle, one giant log line) falls back\nto word-level packing with the same overlap logic, and any single \"word\" over 500 chars is\nsliced first, so nothing huge is ever handed to the tokenizer in one piece.\n\nToken counts are computed **once per paragraph/word and cached** for reuse during overlap\ncalculation. An earlier version re-tokenized on every overlap lookup, which was fine on small\ninputs but caused runaway CPU and multi-GB memory growth on large repositories. If you extend\nthe chunker, preserve that property.\n\n### Incremental reindexing\n\nThere's no separate manifest — the vector store *is* the manifest. Every stored chunk carries\nits source file's `mtimeMs` and a `sha256` content hash. On each run:\n\n1. If a file's on-disk `mtime` matches what's stored, skip it without reading the file.\n2. If `mtime` changed but the content hash is identical (a `touch`), skip re-embedding.\n3. Otherwise, delete that file's old chunks and insert freshly embedded ones.\n4. After the listing, any indexed path no longer on disk (and under the root being indexed) is\n   pruned.\n\n### Storage backends\n\nThe default is LanceDB: embedded, file-backed, real vector search. A `node:sqlite` +\nbrute-force cosine fallback ([`src/store/sqliteFallbackStore.js`](./src/store/sqliteFallbackStore.js))\nimplements the same interface ([`src/store/vectorStore.js`](./src/store/vectorStore.js)) for\nenvironments where LanceDB's native binding doesn't load — sandboxed containers, unusual\narchitectures. Switch with `SS_STORE_BACKEND=sqlite`.\n\nThe fallback does a full table scan per search: fine for tens of thousands of chunks, not\nbeyond. LanceDB's default metric is L2, not cosine, so this project explicitly sets\n`.distanceType('cosine')` on every query, since embeddings are compared as normalized vectors.\n\n## Project structure\n\n```\nsrc/\n  config.js            Defaults + config file resolution (XDG) — the only source of tunables\n  configFile.js        Read/modify/write the config file (backs add/remove/list)\n  embeddings.js        transformers.js pipeline + tokenizer (lazy singletons)\n  chunker.js           Token-aware paragraph packing with overlap\n  ignoreRules.js       What is indexable: git ignore rules + .indexignore + binary filter,\n                       shared by the indexer and grep so they can't drift apart\n  safePath.js          Path confinement for the MCP file-reading tools\n  version.js           Version read from package.json\n  extractors/          text (anything not binary), pdf (pdf-parse), docx (mammoth)\n  store/\n    vectorStore.js        Storage interface + backend selector\n    lancedbStore.js       LanceDB implementation (default)\n    sqliteFallbackStore.js node:sqlite + manual cosine fallback\n  indexer.js           List + extract + chunk + embed + incremental upsert/prune\n  search.js            Hybrid retrieval: vector + BM25 arms fused with RRF — shared by CLI and MCP\n  mcp-server.js        MCP stdio server: the six tools above\n  index.js             CLI entrypoint (commander)\nscripts/index-all.sh   Batched indexing for very large corpora on constrained hosts (Linux)\n```\n\n## Development\n\n```bash\ngit clone https://github.com/adborroto/semantic-search-mcp.git\ncd semantic-search-mcp\nnpm install\nnpm test              # unit + end-to-end (node:test, no framework)\nnpm run test:unit     # skip the slow end-to-end test\nnpm run lint\n```\n\nA `config.json` in the checkout root takes precedence over the XDG location, so you can develop\nagainst a scratch corpus without touching your real setup. Tests always write to temp\ndirectories. See [CONTRIBUTING.md](./CONTRIBUTING.md).\n\n## Out of scope (by design)\n\n- **Answer generation.** This returns chunks, not answers. Feed them to an LLM yourself.\n- **Reranking with a second model.** Hybrid retrieval plus RRF is dependency-free and gets\n  most of the way there — but it is not a cross-encoder reranker.\n- **A web UI.** CLI and MCP only.\n- **Massive-scale corpora.** Built for a personal or team-sized corpus of docs and code — tens\n  of thousands of chunks, not millions. Both backends assume that scale.\n\n## License\n\n[MIT](./LICENSE)\n","readmeFilename":"README.md"}