{"_id":"@cargo-cult/pi","_rev":"7-d20897edcd13f56114f67370cb092e82","name":"@cargo-cult/pi","dist-tags":{"latest":"0.47.0"},"versions":{"0.40.0":{"name":"@cargo-cult/pi","version":"0.40.0","keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@cargo-cult/pi@0.40.0","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"homepage":"https://github.com/TUstudents/pi-mono#readme","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"bin":{"pi-pods":"dist/cli.js"},"dist":{"shasum":"814520421fdc177f8758a21b06c8a5893981a8df","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.40.0.tgz","fileCount":43,"integrity":"sha512-xun9ZMMGwp4SULVn+sZ4bmy0Y9mhfE+Vn7aB1Ln04b3i/UwXx3j7ki+GZ28+Y4+U2m7Ykq610SZSPen+yRVBow==","signatures":[{"sig":"MEUCIBWtxU2RfbtDva0eyaceBhL5u8YHekZfidoryUPACp8sAiEAgdaCGlpPoU0nOsMlt54+VqEJ53b1jFCCLMhG0JCtl4s=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.40.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":292097},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"94ba25b6aee13ac8f98935b3536dab5192ed0399","scripts":{"build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","clean":"rm -rf dist","prepublishOnly":"npm run clean && npm run build"},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"repository":{"url":"git+https://github.com/TUstudents/pi-mono.git","type":"git","directory":"packages/pods"},"_npmVersion":"10.9.4","description":"CLI tool for managing vLLM deployments on GPU pods","directories":{},"_nodeVersion":"22.21.1","dependencies":{"chalk":"^5.5.0","@cargo-cult/pi-agent-core":"^0.40.0"},"_hasShrinkwrap":false,"devDependencies":{},"_npmOperationalInternal":{"tmp":"tmp/pi_0.40.0_1767983662318_0.11079059632843125","host":"s3://npm-registry-packages-npm-production"}},"0.40.1":{"name":"@cargo-cult/pi","version":"0.40.1","keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@cargo-cult/pi@0.40.1","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"homepage":"https://github.com/TUstudents/pi-mono#readme","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"bin":{"pi-pods":"dist/cli.js"},"dist":{"shasum":"e75c23426372b18b3a659b07404bd089975ae6a0","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.40.1.tgz","fileCount":43,"integrity":"sha512-7LuYtKb24hcaUKNSf4CdIPywBggZxLJhMcdSgUMbs76H44roTaIcII4qso1s95Ix41oN+Hb8yb/cYT9Or8bVAQ==","signatures":[{"sig":"MEUCIEHhVRIjiOtuN8ed6kmXdUpWbHdLLQjlqpa4wk1GE55sAiEAuq0BiJjKU2ohbDLTehwtpeq2svexdnQ66PKFvLVXKdk=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.40.1","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":292097},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"4d64ff64d64172c9c7ac87db756bf6044503396f","scripts":{"build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","clean":"rm -rf dist","prepublishOnly":"npm run clean && npm run build"},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"repository":{"url":"git+https://github.com/TUstudents/pi-mono.git","type":"git","directory":"packages/pods"},"_npmVersion":"10.9.4","description":"CLI tool for managing vLLM deployments on GPU pods","directories":{},"_nodeVersion":"22.21.1","dependencies":{"chalk":"^5.5.0","@cargo-cult/pi-agent-core":"^0.40.1"},"_hasShrinkwrap":false,"devDependencies":{},"_npmOperationalInternal":{"tmp":"tmp/pi_0.40.1_1767984141592_0.6558326999830075","host":"s3://npm-registry-packages-npm-production"}},"0.42.0":{"name":"@cargo-cult/pi","version":"0.42.0","keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@cargo-cult/pi@0.42.0","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"homepage":"https://github.com/TUstudents/pi-mono#readme","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"bin":{"pi-pods":"dist/cli.js"},"dist":{"shasum":"704d253c1f69d26a1d13d5038d94b774ea079fcc","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.42.0.tgz","fileCount":43,"integrity":"sha512-DOEyOSdcan2Jwh39/7X6PT8py1O3i4+IYMFM185cQeL5KQ14sGUFX/Bg1G7+agk11X3ujALy1oobteIKrVBaOw==","signatures":[{"sig":"MEQCIFFQPAb6MvPxohfP33rPbV3iXKpYaKZEp8BO1iJJxrnnAiA4380LClyprg+h2XS6q7zGYF+9CekXcoOAnBhwVksJ7Q==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.42.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":292097},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"84996508517bf91a4cbe8fbaa09d5b12b19f38cf","scripts":{"build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","clean":"rm -rf dist","prepublishOnly":"npm run clean && npm run build"},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"repository":{"url":"git+https://github.com/TUstudents/pi-mono.git","type":"git","directory":"packages/pods"},"_npmVersion":"10.9.4","description":"CLI tool for managing vLLM deployments on GPU pods","directories":{},"_nodeVersion":"22.21.1","dependencies":{"chalk":"^5.5.0","@cargo-cult/pi-agent-core":"^0.42.0"},"_hasShrinkwrap":false,"devDependencies":{},"_npmOperationalInternal":{"tmp":"tmp/pi_0.42.0_1767985885790_0.7384846441199766","host":"s3://npm-registry-packages-npm-production"}},"0.42.1":{"name":"@cargo-cult/pi","version":"0.42.1","keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@cargo-cult/pi@0.42.1","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"homepage":"https://github.com/TUstudents/pi-mono#readme","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"bin":{"pi-pods":"dist/cli.js"},"dist":{"shasum":"d27de75d8a4e89175e0d2b55318f99e5e347c933","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.42.1.tgz","fileCount":43,"integrity":"sha512-uTEj/4xjkEdlrvOe2BBX+tDQTTHE16JFEEosaytx6kLWl1X8OIqmM/B9Yom0W9reqK3F825CUnfnqI6dVJZlkw==","signatures":[{"sig":"MEUCIDmclXKcF31nff+eFjL5xid3XB/vwbFOnaPNpsV1olssAiEAs3592jIo94w4g2gHD9fx8vS3lHiXg+dec+IZQOCHhaw=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.42.1","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":292097},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"5198ee434057587f815953b57e154b8f71e1db3c","scripts":{"build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","clean":"rm -rf dist","prepublishOnly":"npm run clean && npm run build"},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"repository":{"url":"git+https://github.com/TUstudents/pi-mono.git","type":"git","directory":"packages/pods"},"_npmVersion":"10.9.4","description":"CLI tool for managing vLLM deployments on GPU pods","directories":{},"_nodeVersion":"22.21.1","dependencies":{"chalk":"^5.5.0","@cargo-cult/pi-agent-core":"^0.42.1"},"_hasShrinkwrap":false,"devDependencies":{},"_npmOperationalInternal":{"tmp":"tmp/pi_0.42.1_1767986276179_0.9494570281667192","host":"s3://npm-registry-packages-npm-production"}},"0.45.3":{"name":"@cargo-cult/pi","version":"0.45.3","keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@cargo-cult/pi@0.45.3","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"homepage":"https://github.com/TUstudents/pi-mono#readme","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"bin":{"pi-pods":"dist/cli.js"},"dist":{"shasum":"3b1bc79c23ff50cae087d8cd83c928e112ce2b3f","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.45.3.tgz","fileCount":43,"integrity":"sha512-Srbc/m9V6tSrBnh+7ULcrYoKI0ulNu71FJeojW/C4v3WsDfjCDkjbzFqb22nduoha66HJzTaBSJZbUCr8fxmKg==","signatures":[{"sig":"MEYCIQCUALT+CD6nIACspm8mlO7qtVZiRYkrDAz36c4tk/RghQIhANObf2+CNr4V9Az15aqu2jIdA7jXJomEejuzqTnzVGG9","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.45.3","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":292097},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"c565ce93e848a6a833849cebebec6178585abed7","scripts":{"build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","clean":"rm -rf dist","prepublishOnly":"npm run clean && npm run build"},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"repository":{"url":"git+https://github.com/TUstudents/pi-mono.git","type":"git","directory":"packages/pods"},"_npmVersion":"10.9.4","description":"CLI tool for managing vLLM deployments on GPU pods","directories":{},"_nodeVersion":"22.21.1","dependencies":{"chalk":"^5.5.0","@cargo-cult/pi-agent-core":"^0.45.3"},"_hasShrinkwrap":false,"devDependencies":{},"_npmOperationalInternal":{"tmp":"tmp/pi_0.45.3_1768326678658_0.75356944394093","host":"s3://npm-registry-packages-npm-production"}},"0.45.7":{"name":"@cargo-cult/pi","version":"0.45.7","keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","_id":"@cargo-cult/pi@0.45.7","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"homepage":"https://github.com/TUstudents/pi-mono#readme","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"bin":{"pi-pods":"dist/cli.js"},"dist":{"shasum":"2d5a98b9e8be7f46ac49c637e0257b9ff137f00f","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.45.7.tgz","fileCount":43,"integrity":"sha512-3ggrnaVftLuXkp2t26VEWdyb1IF7Q0n6S2w89PCKq80A/eoLRT45XRn1c0yJaFhgcnrgU+NG2aoSaNOZ/Og3bw==","signatures":[{"sig":"MEUCICOP+j+AHy8bwa7LcxmYmh/TJDdAIgvBAY+/euzgxDZXAiEAj0uME8eyabuU80IxvRLxGW2HbuOLunrL7X1Z3sWwGik=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.45.7","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"unpackedSize":292097},"type":"module","engines":{"node":">=20.0.0"},"gitHead":"0854b993f8933023232e238f1e8eb14c6768817f","scripts":{"build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","clean":"rm -rf dist","prepublishOnly":"npm run clean && npm run build"},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"repository":{"url":"git+https://github.com/TUstudents/pi-mono.git","type":"git","directory":"packages/pods"},"_npmVersion":"10.9.4","description":"CLI tool for managing vLLM deployments on GPU pods","directories":{},"_nodeVersion":"22.21.1","dependencies":{"chalk":"^5.5.0","@cargo-cult/pi-agent-core":"^0.45.7"},"_hasShrinkwrap":false,"devDependencies":{},"_npmOperationalInternal":{"tmp":"tmp/pi_0.45.7_1768473931881_0.7075208044257508","host":"s3://npm-registry-packages-npm-production"}},"0.47.0":{"name":"@cargo-cult/pi","version":"0.47.0","description":"CLI tool for managing vLLM deployments on GPU pods","type":"module","bin":{"pi-pods":"dist/cli.js"},"scripts":{"clean":"rm -rf dist","build":"tsgo -p tsconfig.build.json && chmod +x dist/cli.js && cp src/models.json dist/ && cp -r scripts dist/","prepublishOnly":"npm run clean && npm run build"},"keywords":["llm","vllm","gpu","ai","cli"],"author":{"name":"Mario Zechner"},"license":"MIT","repository":{"type":"git","url":"git+https://github.com/TUstudents/pi-mono.git","directory":"packages/pods"},"engines":{"node":">=20.0.0"},"dependencies":{"@cargo-cult/pi-agent-core":"^0.47.0","chalk":"^5.5.0"},"devDependencies":{},"_id":"@cargo-cult/pi@0.47.0","gitHead":"639988d4bcbe25693775ce39cd905382eb58b7d4","bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"homepage":"https://github.com/TUstudents/pi-mono#readme","_nodeVersion":"22.21.1","_npmVersion":"10.9.4","dist":{"integrity":"sha512-UU96gHPa6MPGiZcwIJvIRjjijnyOD7WQYSuMyYigNGOVgCCmshJAkKs7XONFlsBpngsum15JtuIAkuzTGyG7CQ==","shasum":"0fee6ac51ff457d620c4a6f5a182ec53fef3290b","tarball":"https://registry.npmjs.org/@cargo-cult/pi/-/pi-0.47.0.tgz","fileCount":43,"unpackedSize":292097,"attestations":{"url":"https://registry.npmjs.org/-/npm/v1/attestations/@cargo-cult%2fpi@0.47.0","provenance":{"predicateType":"https://slsa.dev/provenance/v1"}},"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEYCIQCv8F7LF0GZIxW7u+iMocN3lQKAAmC8pYd/2twpzoW/DgIhALzBmNpLNAPrRfQ3/q0cXzgE17JJI1M8rJLjdjLkX0EC"}]},"_npmUser":{"name":"tustudents","email":"python@atoms.eu"},"directories":{},"maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/pi_0.47.0_1768544539459_0.6463611228451693"},"_hasShrinkwrap":false}},"time":{"created":"2026-01-09T18:34:22.233Z","modified":"2026-01-16T06:22:19.965Z","0.40.0":"2026-01-09T18:34:22.476Z","0.40.1":"2026-01-09T18:42:21.738Z","0.42.0":"2026-01-09T19:11:25.960Z","0.42.1":"2026-01-09T19:17:56.341Z","0.45.3":"2026-01-13T17:51:18.807Z","0.45.7":"2026-01-15T10:45:32.020Z","0.47.0":"2026-01-16T06:22:19.670Z"},"bugs":{"url":"https://github.com/TUstudents/pi-mono/issues"},"author":{"name":"Mario Zechner"},"license":"MIT","homepage":"https://github.com/TUstudents/pi-mono#readme","keywords":["llm","vllm","gpu","ai","cli"],"repository":{"type":"git","url":"git+https://github.com/TUstudents/pi-mono.git","directory":"packages/pods"},"description":"CLI tool for managing vLLM deployments on GPU pods","maintainers":[{"name":"tustudents","email":"python@atoms.eu"}],"readme":"# pi\n\nDeploy and manage LLMs on GPU pods with automatic vLLM configuration for agentic workloads.\n\n## Installation\n\n```bash\nnpm install -g @mariozechner/pi\n```\n\n## What is pi?\n\n`pi` simplifies running large language models on remote GPU pods. It automatically:\n- Sets up vLLM on fresh Ubuntu pods\n- Configures tool calling for agentic models (Qwen, GPT-OSS, GLM, etc.)\n- Manages multiple models on the same pod with \"smart\" GPU allocation\n- Provides OpenAI-compatible API endpoints for each model\n- Includes an interactive agent with file system tools for testing\n\n## Quick Start\n\n```bash\n# Set required environment variables\nexport HF_TOKEN=your_huggingface_token      # Get from https://huggingface.co/settings/tokens\nexport PI_API_KEY=your_api_key              # Any string you want for API authentication\n\n# Setup a DataCrunch pod with NFS storage (models path auto-extracted)\npi pods setup dc1 \"ssh root@1.2.3.4\" \\\n  --mount \"sudo mount -t nfs -o nconnect=16 nfs.fin-02.datacrunch.io:/your-pseudo /mnt/hf-models\"\n\n# Start a model (automatic configuration for known models)\npi start Qwen/Qwen2.5-Coder-32B-Instruct --name qwen\n\n# Send a single message to the model\npi agent qwen \"What is the Fibonacci sequence?\"\n\n# Interactive chat mode with file system tools\npi agent qwen -i\n\n# Use with any OpenAI-compatible client\nexport OPENAI_BASE_URL='http://1.2.3.4:8001/v1'\nexport OPENAI_API_KEY=$PI_API_KEY\n```\n\n## Prerequisites\n\n- Node.js 18+\n- HuggingFace token (for model downloads)\n- GPU pod with:\n  - Ubuntu 22.04 or 24.04\n  - SSH root access\n  - NVIDIA drivers installed\n  - Persistent storage for models\n\n## Supported Providers\n\n### Primary Support\n\n**DataCrunch** - Best for shared model storage\n- NFS volumes sharable across multiple pods in same region\n- Models download once, use everywhere\n- Ideal for teams or multiple experiments\n\n**RunPod** - Good persistent storage\n- Network volumes persist independently\n- Cannot share between running pods simultaneously\n- Good for single-pod workflows\n\n### Also Works With\n- Vast.ai (volumes locked to specific machine)\n- Prime Intellect (no persistent storage)\n- AWS EC2 (with EFS setup)\n- Any Ubuntu machine with NVIDIA GPUs, CUDA driver, and SSH\n\n## Commands\n\n### Pod Management\n\n```bash\npi pods setup <name> \"<ssh>\" [options]        # Setup new pod\n  --mount \"<mount_command>\"                   # Run mount command during setup\n  --models-path <path>                        # Override extracted path (optional)\n  --vllm release|nightly|gpt-oss              # vLLM version (default: release)\n\npi pods                                       # List all configured pods\npi pods active <name>                         # Switch active pod\npi pods remove <name>                         # Remove pod from local config\npi shell [<name>]                             # SSH into pod\npi ssh [<name>] \"<command>\"                   # Run command on pod\n```\n\n**Note**: When using `--mount`, the models path is automatically extracted from the mount command's target directory. You only need `--models-path` if not using `--mount` or to override the extracted path.\n\n#### vLLM Version Options\n\n- `release` (default): Stable vLLM release, recommended for most users\n- `nightly`: Latest vLLM features, needed for newest models like GLM-4.5\n- `gpt-oss`: Special build for OpenAI's GPT-OSS models only\n\n### Model Management\n\n```bash\npi start <model> --name <name> [options]  # Start a model\n  --memory <percent>      # GPU memory: 30%, 50%, 90% (default: 90%)\n  --context <size>        # Context window: 4k, 8k, 16k, 32k, 64k, 128k\n  --gpus <count>          # Number of GPUs to use (predefined models only)\n  --pod <name>            # Target specific pod (overrides active)\n  --vllm <args...>        # Pass custom args directly to vLLM\n\npi stop [<name>]          # Stop model (or all if no name given)\npi list                   # List running models with status\npi logs <name>            # Stream model logs (tail -f)\n```\n\n### Agent & Chat Interface\n\n```bash\npi agent <name> \"<message>\"               # Single message to model\npi agent <name> \"<msg1>\" \"<msg2>\"         # Multiple messages in sequence\npi agent <name> -i                        # Interactive chat mode\npi agent <name> -i -c                     # Continue previous session\n\n# Standalone OpenAI-compatible agent (works with any API)\npi-agent --base-url http://localhost:8000/v1 --model llama-3.1 \"Hello\"\npi-agent --api-key sk-... \"What is 2+2?\"  # Uses OpenAI by default\npi-agent --json \"What is 2+2?\"            # Output event stream as JSONL\npi-agent -i                                # Interactive mode\n```\n\nThe agent includes tools for file operations (read, list, bash, glob, rg) to test agentic capabilities, particularly useful for code navigation and analysis tasks.\n\n## Predefined Model Configurations\n\n`pi` includes predefined configurations for popular agentic models, so you do not have to specify `--vllm` arguments manually. `pi` will also check if the model you selected can actually run on your pod with respect to the number of GPUs and available VRAM. Run `pi start` without additional arguments to see a list of predefined models that can run on the active pod.\n\n### Qwen Models\n```bash\n# Qwen2.5-Coder-32B - Excellent coding model, fits on single H100/H200\npi start Qwen/Qwen2.5-Coder-32B-Instruct --name qwen\n\n# Qwen3-Coder-30B - Advanced reasoning with tool use\npi start Qwen/Qwen3-Coder-30B-A3B-Instruct --name qwen3\n\n# Qwen3-Coder-480B - State-of-the-art on 8xH200 (data-parallel mode)\npi start Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8 --name qwen-480b\n```\n\n### GPT-OSS Models\n```bash\n# Requires special vLLM build during setup\npi pods setup gpt-pod \"ssh root@1.2.3.4\" --models-path /workspace --vllm gpt-oss\n\n# GPT-OSS-20B - Fits on 16GB+ VRAM\npi start openai/gpt-oss-20b --name gpt20\n\n# GPT-OSS-120B - Needs 60GB+ VRAM\npi start openai/gpt-oss-120b --name gpt120\n```\n\n### GLM Models\n```bash\n# GLM-4.5 - Requires 8-16 GPUs, includes thinking mode\npi start zai-org/GLM-4.5 --name glm\n\n# GLM-4.5-Air - Smaller version, 1-2 GPUs\npi start zai-org/GLM-4.5-Air --name glm-air\n```\n\n### Custom Models with --vllm\n\nFor models not in the predefined list, use `--vllm` to pass arguments directly to vLLM:\n\n```bash\n# DeepSeek with custom settings\npi start deepseek-ai/DeepSeek-V3 --name deepseek --vllm \\\n  --tensor-parallel-size 4 --trust-remote-code\n\n# Mistral with pipeline parallelism\npi start mistralai/Mixtral-8x22B-Instruct-v0.1 --name mixtral --vllm \\\n  --tensor-parallel-size 8 --pipeline-parallel-size 2\n\n# Any model with specific tool parser\npi start some/model --name mymodel --vllm \\\n  --tool-call-parser hermes --enable-auto-tool-choice\n```\n\n## DataCrunch Setup\n\nDataCrunch offers the best experience with shared NFS storage across pods:\n\n### 1. Create Shared Filesystem (SFS)\n- Go to DataCrunch dashboard → Storage → Create SFS\n- Choose size and datacenter\n- Note the mount command (e.g., `sudo mount -t nfs -o nconnect=16 nfs.fin-02.datacrunch.io:/hf-models-fin02-8ac1bab7 /mnt/hf-models-fin02`)\n\n### 2. Create GPU Instance\n- Create instance in same datacenter as SFS\n- Share the SFS with the instance\n- Get SSH command from dashboard\n\n### 3. Setup with pi\n```bash\n# Get mount command from DataCrunch dashboard\npi pods setup dc1 \"ssh root@instance.datacrunch.io\" \\\n  --mount \"sudo mount -t nfs -o nconnect=16 nfs.fin-02.datacrunch.io:/your-pseudo /mnt/hf-models\"\n\n# Models automatically stored in /mnt/hf-models (extracted from mount command)\n```\n\n### 4. Benefits\n- Models persist across instance restarts\n- Share models between multiple instances in same datacenter\n- Download once, use everywhere\n- Pay only for storage, not compute time during downloads\n\n## RunPod Setup\n\nRunPod offers good persistent storage with network volumes:\n\n### 1. Create Network Volume (optional)\n- Go to RunPod dashboard → Storage → Create Network Volume\n- Choose size and region\n\n### 2. Create GPU Pod\n- Select \"Network Volume\" during pod creation (if using)\n- Attach your volume to `/runpod-volume`\n- Get SSH command from pod details\n\n### 3. Setup with pi\n```bash\n# With network volume\npi pods setup runpod \"ssh root@pod.runpod.io\" --models-path /runpod-volume\n\n# Or use workspace (persists with pod but not shareable)\npi pods setup runpod \"ssh root@pod.runpod.io\" --models-path /workspace\n```\n\n\n## Multi-GPU Support\n\n### Automatic GPU Assignment\nWhen running multiple models, pi automatically assigns them to different GPUs:\n```bash\npi start model1 --name m1  # Auto-assigns to GPU 0\npi start model2 --name m2  # Auto-assigns to GPU 1\npi start model3 --name m3  # Auto-assigns to GPU 2\n```\n\n### Specify GPU Count for Predefined Models\nFor predefined models with multiple configurations, use `--gpus` to control GPU usage:\n```bash\n# Run Qwen on 1 GPU instead of all available\npi start Qwen/Qwen2.5-Coder-32B-Instruct --name qwen --gpus 1\n\n# Run GLM-4.5 on 8 GPUs (if it has an 8-GPU config)\npi start zai-org/GLM-4.5 --name glm --gpus 8\n```\n\nIf the model doesn't have a configuration for the requested GPU count, you'll see available options.\n\n### Tensor Parallelism for Large Models\nFor models that don't fit on a single GPU:\n```bash\n# Use all available GPUs\npi start meta-llama/Llama-3.1-70B-Instruct --name llama70b --vllm \\\n  --tensor-parallel-size 4\n\n# Specific GPU count\npi start Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8 --name qwen480 --vllm \\\n  --data-parallel-size 8 --enable-expert-parallel\n```\n\n## API Integration\n\nAll models expose OpenAI-compatible endpoints:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n    base_url=\"http://your-pod-ip:8001/v1\",\n    api_key=\"your-pi-api-key\"\n)\n\n# Chat completion with tool calling\nresponse = client.chat.completions.create(\n    model=\"Qwen/Qwen2.5-Coder-32B-Instruct\",\n    messages=[\n        {\"role\": \"user\", \"content\": \"Write a Python function to calculate fibonacci\"}\n    ],\n    tools=[{\n        \"type\": \"function\",\n        \"function\": {\n            \"name\": \"execute_code\",\n            \"description\": \"Execute Python code\",\n            \"parameters\": {\n                \"type\": \"object\",\n                \"properties\": {\n                    \"code\": {\"type\": \"string\"}\n                },\n                \"required\": [\"code\"]\n            }\n        }\n    }],\n    tool_choice=\"auto\"\n)\n```\n\n## Standalone Agent CLI\n\n`pi` includes a standalone OpenAI-compatible agent that can work with any API:\n\n```bash\n# Install globally to get pi-agent command\nnpm install -g @mariozechner/pi\n\n# Use with OpenAI\npi-agent --api-key sk-... \"What is machine learning?\"\n\n# Use with local vLLM\npi-agent --base-url http://localhost:8000/v1 \\\n         --model meta-llama/Llama-3.1-8B-Instruct \\\n         --api-key dummy \\\n         \"Explain quantum computing\"\n\n# Interactive mode\npi-agent -i\n\n# Continue previous session\npi-agent --continue \"Follow up question\"\n\n# Custom system prompt\npi-agent --system-prompt \"You are a Python expert\" \"Write a web scraper\"\n\n# Use responses API (for GPT-OSS models)\npi-agent --api responses --model openai/gpt-oss-20b \"Hello\"\n```\n\nThe agent supports:\n- Session persistence across conversations\n- Interactive TUI mode with syntax highlighting\n- File system tools (read, list, bash, glob, rg) for code navigation\n- Both Chat Completions and Responses API formats\n- Custom system prompts\n\n## Tool Calling Support\n\n`pi` automatically configures appropriate tool calling parsers for known models:\n\n- **Qwen models**: `hermes` parser (Qwen3-Coder uses `qwen3_coder`)\n- **GLM models**: `glm4_moe` parser with reasoning support\n- **GPT-OSS models**: Uses `/v1/responses` endpoint, as tool calling (function calling in OpenAI parlance) is currently a [WIP with the `v1/chat/completions` endpoint](https://docs.vllm.ai/projects/recipes/en/latest/OpenAI/GPT-OSS.html#tool-use).\n- **Custom models**: Specify with `--vllm --tool-call-parser <parser> --enable-auto-tool-choice`\n\nTo disable tool calling:\n```bash\npi start model --name mymodel --vllm --disable-tool-call-parser\n```\n\n## Memory and Context Management\n\n### GPU Memory Allocation\nControls how much GPU memory vLLM pre-allocates:\n- `--memory 30%`: High concurrency, limited context\n- `--memory 50%`: Balanced (default)\n- `--memory 90%`: Maximum context, low concurrency\n\n### Context Window\nSets maximum input + output tokens:\n- `--context 4k`: 4,096 tokens total\n- `--context 32k`: 32,768 tokens total\n- `--context 128k`: 131,072 tokens total\n\nExample for coding workload:\n```bash\n# Large context for code analysis, moderate concurrency\npi start Qwen/Qwen2.5-Coder-32B-Instruct --name coder \\\n  --context 64k --memory 70%\n```\n\n**Note**: When using `--vllm`, the `--memory`, `--context`, and `--gpus` parameters are ignored. You'll see a warning if you try to use them together.\n\n## Session Persistence\n\nThe interactive agent mode (`-i`) saves sessions for each project directory:\n\n```bash\n# Start new session\npi agent qwen -i\n\n# Continue previous session (maintains chat history)\npi agent qwen -i -c\n```\n\nSessions are stored in `~/.pi/sessions/` organized by project path and include:\n- Complete conversation history\n- Tool call results\n- Token usage statistics\n\n## Architecture & Event System\n\nThe agent uses a unified event-based architecture where all interactions flow through `AgentEvent` types. This enables:\n- Consistent UI rendering across console and TUI modes\n- Session recording and replay\n- Clean separation between API calls and UI updates\n- JSON output mode for programmatic integration\n\nEvents are automatically converted to the appropriate API format (Chat Completions or Responses) based on the model type.\n\n### JSON Output Mode\n\nUse `--json` flag to output the event stream as JSONL (JSON Lines) for programmatic consumption:\n```bash\npi-agent --api-key sk-... --json \"What is 2+2?\"\n```\n\nEach line is a complete JSON object representing an event:\n```jsonl\n{\"type\":\"user_message\",\"text\":\"What is 2+2?\"}\n{\"type\":\"assistant_start\"}\n{\"type\":\"assistant_message\",\"text\":\"2 + 2 = 4\"}\n{\"type\":\"token_usage\",\"inputTokens\":10,\"outputTokens\":5,\"totalTokens\":15,\"cacheReadTokens\":0,\"cacheWriteTokens\":0}\n```\n\n## Troubleshooting\n\n### OOM (Out of Memory) Errors\n- Reduce `--memory` percentage\n- Use smaller model or quantized version (FP8)\n- Reduce `--context` size\n\n### Model Won't Start\n```bash\n# Check GPU usage\npi ssh \"nvidia-smi\"\n\n# Check if port is in use\npi list\n\n# Force stop all models\npi stop\n```\n\n### Tool Calling Issues\n- Not all models support tool calling reliably\n- Try different parser: `--vllm --tool-call-parser mistral`\n- Or disable: `--vllm --disable-tool-call-parser`\n\n### Access Denied for Models\nSome models (Llama, Mistral) require HuggingFace access approval. Visit the model page and click \"Request access\".\n\n### vLLM Build Issues\nIf using `--vllm nightly` fails, try:\n- Use `--vllm release` for stable version\n- Check CUDA compatibility with `pi ssh \"nvidia-smi\"`\n\n### Agent Not Finding Messages\nIf the agent shows configuration instead of your message, ensure quotes around messages with special characters:\n```bash\n# Good\npi agent qwen \"What is this file about?\"\n\n# Bad (shell might interpret special chars)\npi agent qwen What is this file about?\n```\n\n## Advanced Usage\n\n### Working with Multiple Pods\n```bash\n# Override active pod for any command\npi start model --name test --pod dev-pod\npi list --pod prod-pod\npi stop test --pod dev-pod\n```\n\n### Custom vLLM Arguments\n```bash\n# Pass any vLLM argument after --vllm\npi start model --name custom --vllm \\\n  --quantization awq \\\n  --enable-prefix-caching \\\n  --max-num-seqs 256 \\\n  --gpu-memory-utilization 0.95\n```\n\n### Monitoring\n```bash\n# Watch GPU utilization\npi ssh \"watch -n 1 nvidia-smi\"\n\n# Check model downloads\npi ssh \"du -sh ~/.cache/huggingface/hub/*\"\n\n# View all logs\npi ssh \"ls -la ~/.vllm_logs/\"\n\n# Check agent session history\nls -la ~/.pi/sessions/\n```\n\n## Environment Variables\n\n- `HF_TOKEN` - HuggingFace token for model downloads\n- `PI_API_KEY` - API key for vLLM endpoints\n- `PI_CONFIG_DIR` - Config directory (default: `~/.pi`)\n- `OPENAI_API_KEY` - Used by `pi-agent` when no `--api-key` provided\n\n## License\n\nMIT","readmeFilename":"README.md"}