{"_id":"@dmallory42/pi-read-url","_rev":"3-7b5d0bbf81ea5084f71d4f25dbc63bc8","name":"@dmallory42/pi-read-url","dist-tags":{"latest":"0.1.2"},"versions":{"0.1.0":{"name":"@dmallory42/pi-read-url","version":"0.1.0","keywords":["pi-package","pi","curl","web","markdown"],"license":"MIT","_id":"@dmallory42/pi-read-url@0.1.0","maintainers":[{"name":"dmallory42","email":"d.mallory42@gmail.com"}],"homepage":"https://github.com/dmallory42/pi-read-url#readme","bugs":{"url":"https://github.com/dmallory42/pi-read-url/issues"},"pi":{"extensions":["./index.ts"]},"dist":{"shasum":"34023d92240468460734a9723b65c55e63bc0b36","tarball":"https://registry.npmjs.org/@dmallory42/pi-read-url/-/pi-read-url-0.1.0.tgz","fileCount":5,"integrity":"sha512-H9Fwnsog4cLpUaiHmNxcVeaTYnQt4F0FIZ8t4Xrro2+NWUmOT2sGyEAhH0ydJDDfyaLYwVgAOifTb55UbCHX5A==","signatures":[{"sig":"MEUCIQCAS0mI010vLFqkLcea79T1VwGEFl33b3KIwB2yuEyfsgIgUmYsCuff6KRIruCh42l3cDqmrNNJSt2lWrWubbg+ZJw=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":17764},"type":"module","engines":{"node":">=20.11.0"},"gitHead":"b5a931fa326d2bd918e80762cd6b6b71380d5eba","scripts":{"test":"bash ./scripts/smoke-test.sh","build":"echo 'No build step required'","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"dmallory42","email":"d.mallory42@gmail.com"},"repository":{"url":"git+https://github.com/dmallory42/pi-read-url.git","type":"git"},"_npmVersion":"11.12.1","description":"Pi extension that reads public web pages via system curl and returns cleaned markdown.","directories":{},"_nodeVersion":"24.15.0","dependencies":{"jsdom":"^26.1.0","turndown":"^7.2.0","@mozilla/readability":"^0.5.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.5.0","@types/node":"^20.12.0","@types/jsdom":"^21.1.7","@types/turndown":"^5.0.5","@sinclair/typebox":"^0.34.41","@mariozechner/pi-coding-agent":"^0.70.2"},"peerDependencies":{"@sinclair/typebox":"*","@mariozechner/pi-coding-agent":"*"},"_npmOperationalInternal":{"tmp":"tmp/pi-read-url_0.1.0_1777099539835_0.9722279522526955","host":"s3://npm-registry-packages-npm-production"}},"0.1.1":{"name":"@dmallory42/pi-read-url","version":"0.1.1","keywords":["pi-package","pi","curl","web","markdown"],"license":"MIT","_id":"@dmallory42/pi-read-url@0.1.1","maintainers":[{"name":"dmallory42","email":"d.mallory42@gmail.com"}],"homepage":"https://github.com/dmallory42/pi-read-url#readme","bugs":{"url":"https://github.com/dmallory42/pi-read-url/issues"},"pi":{"extensions":["./index.ts"]},"dist":{"shasum":"1a8e0209ca6420820d9fc32ff9e07f811b81a062","tarball":"https://registry.npmjs.org/@dmallory42/pi-read-url/-/pi-read-url-0.1.1.tgz","fileCount":5,"integrity":"sha512-PFkzL2CG63iiJxW3/YYI/rROcDTJYt4Y3NnDUJxCBSZaOZY3jXtio7mB6MQvVAFP6dgKGbm+7gPzrpq3VZnE0Q==","signatures":[{"sig":"MEYCIQCkAY9Pyv449Y3H0sKdNE//PgA+pThMWxJEt1ddQHRUGQIhAKQn7X00eHo+ZXAyGNCXUOZZgJnjHx2vJTf1lDsRwQDZ","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":18655},"type":"module","engines":{"node":">=20.11.0"},"gitHead":"e661194c710e3f54b43f90c26647bd541cee8a14","scripts":{"test":"bash ./scripts/smoke-test.sh","build":"echo 'No build step required'","typecheck":"tsc -p tsconfig.json --noEmit"},"_npmUser":{"name":"dmallory42","email":"d.mallory42@gmail.com"},"repository":{"url":"git+https://github.com/dmallory42/pi-read-url.git","type":"git"},"_npmVersion":"11.12.1","description":"Pi extension that reads public web pages via system curl and returns cleaned markdown.","directories":{},"_nodeVersion":"24.15.0","dependencies":{"jsdom":"^26.1.0","turndown":"^7.2.0","@mozilla/readability":"^0.5.0"},"publishConfig":{"access":"public"},"_hasShrinkwrap":false,"devDependencies":{"typescript":"^5.5.0","@types/node":"^20.12.0","@types/jsdom":"^21.1.7","@types/turndown":"^5.0.5","@sinclair/typebox":"^0.34.41","@mariozechner/pi-coding-agent":"^0.70.2"},"peerDependencies":{"@sinclair/typebox":"*","@mariozechner/pi-coding-agent":"*"},"_npmOperationalInternal":{"tmp":"tmp/pi-read-url_0.1.1_1777100944372_0.5681048826298278","host":"s3://npm-registry-packages-npm-production"}},"0.1.2":{"name":"@dmallory42/pi-read-url","version":"0.1.2","type":"module","description":"Pi extension for extracting public HTML page URLs into clean markdown via system curl.","keywords":["pi-package","pi","pi-extension","read-url","curl","web","markdown","content-extraction","html","readability"],"license":"MIT","repository":{"type":"git","url":"git+https://github.com/dmallory42/pi-read-url.git"},"homepage":"https://github.com/dmallory42/pi-read-url#readme","bugs":{"url":"https://github.com/dmallory42/pi-read-url/issues"},"publishConfig":{"access":"public"},"pi":{"extensions":["./index.ts"]},"scripts":{"build":"echo 'No build step required'","typecheck":"tsc -p tsconfig.json --noEmit","test":"bash ./scripts/smoke-test.sh"},"dependencies":{"@mozilla/readability":"^0.5.0","jsdom":"^26.1.0","turndown":"^7.2.0"},"peerDependencies":{"@mariozechner/pi-coding-agent":"*","@sinclair/typebox":"*"},"devDependencies":{"@mariozechner/pi-coding-agent":"^0.70.2","@sinclair/typebox":"^0.34.41","@types/jsdom":"^21.1.7","@types/node":"^20.12.0","@types/turndown":"^5.0.5","typescript":"^5.5.0"},"engines":{"node":">=20.11.0"},"gitHead":"6e0212f822f505f67d9c0d6bb36cb6bf198b1d86","_id":"@dmallory42/pi-read-url@0.1.2","_nodeVersion":"24.15.0","_npmVersion":"11.12.1","dist":{"integrity":"sha512-BYYfVjGSiOnf+5McMSeML3M0WYODE7otpDCx2zkK2zappqhHAqnyQuiZ+WuZhVl8pKU8vBiRAAKkZlrYUXpq+A==","shasum":"66748208f993c913ca12f23ad2a51fae1705d28f","tarball":"https://registry.npmjs.org/@dmallory42/pi-read-url/-/pi-read-url-0.1.2.tgz","fileCount":5,"unpackedSize":18763,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEQCIG5OVu2/6UWmH5Ut1Z2bTGOy25bJNwPMnhXD6ZR64ImPAiAY1beyRjRX7C66uW8KZI3/G2OdkBz3tLTDn2WkjC/pYg=="}]},"_npmUser":{"name":"dmallory42","email":"d.mallory42@gmail.com"},"directories":{},"maintainers":[{"name":"dmallory42","email":"d.mallory42@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/pi-read-url_0.1.2_1777101250828_0.298167107003527"},"_hasShrinkwrap":false}},"time":{"created":"2026-04-25T06:45:39.746Z","modified":"2026-04-25T07:14:11.066Z","0.1.0":"2026-04-25T06:45:39.991Z","0.1.1":"2026-04-25T07:09:04.550Z","0.1.2":"2026-04-25T07:14:10.964Z"},"bugs":{"url":"https://github.com/dmallory42/pi-read-url/issues"},"license":"MIT","homepage":"https://github.com/dmallory42/pi-read-url#readme","keywords":["pi-package","pi","pi-extension","read-url","curl","web","markdown","content-extraction","html","readability"],"repository":{"type":"git","url":"git+https://github.com/dmallory42/pi-read-url.git"},"description":"Pi extension for extracting public HTML page URLs into clean markdown via system curl.","maintainers":[{"name":"dmallory42","email":"d.mallory42@gmail.com"}],"readme":"# @dmallory42/pi-read-url\n\nA small [pi](https://github.com/badlogic/pi-mono) extension that adds a `read_url` tool for turning public HTML page URLs into clean, readable markdown using the machine's system `curl`.\n\nFast, local, and lightweight — built for content extraction, not browser automation.\n\n## What it does\n\n- fetches public HTML page URLs via the user's system `curl`\n- extracts the main readable content with Mozilla Readability\n- converts extracted HTML to markdown with Turndown\n- keeps output compact by default to save tokens\n- supports `maxChars` for even smaller extracts\n- optionally includes metadata like site name, byline, excerpt, and HTTP status\n- rejects obvious non-page URLs like PDFs, media files, and common downloads\n- returns friendlier errors for DNS issues, blocked pages, timeouts, and unsupported content types\n\n## Install\n\n### From npm\n\n```bash\npi install npm:@dmallory42/pi-read-url\n```\n\n### From git\n\n```bash\npi install git:github.com/dmallory42/pi-read-url\n```\n\nThen reload pi:\n\n```text\n/reload\n```\n\n## Usage\n\nAsk pi naturally:\n\n```text\nRead https://example.com\nUse read_url to extract the main content from https://example.com\nUse read_url on https://example.com with maxChars 4000\nUse read_url on https://example.com and includeMetadata true\nUse read_url on https://example.com and focus on the author bio\n```\n\n## Tool behavior\n\n`read_url` is built for extracting content from public HTML pages by URL.\n\nIt works best for:\n\n- blogs\n- docs\n- static content pages\n- article-like HTML pages\n\nIt is not intended for:\n\n- PDFs or office documents\n- images, media, and downloadable files\n- login-gated pages\n- JS-heavy SPAs\n- aggressively bot-protected sites\n- interactive browsing tasks like clicking, login flows, or form submission\n\n## Parameters\n\n### `url`\n\nThe HTTP(S) page URL to fetch.\n\n### `objective`\n\nAn optional focus hint to help frame what the caller cares about in the returned extract.\n\n### `maxChars`\n\nOptional character cap for the returned markdown. Useful when you want to save tokens.\n\n### `includeMetadata`\n\nOptional boolean. When enabled, the output also includes metadata like HTTP status, site name, byline, and excerpt.\n\n## Why\n\nSometimes you just want the content of a page URL in a form the model can use:\n\n- without sending it through a third-party fetch service\n- without hand-rolling `curl | grep | sed` pipelines\n- without jumping to a full browser automation stack\n\n## Development\n\nTypecheck:\n\n```bash\nnpm run typecheck\n```\n\nRun the local smoke test:\n\n```bash\nnpm test\n```\n\nRelease notes and workflow live in [`RELEASING.md`](./RELEASING.md).\n\n## License\n\nMIT\n","readmeFilename":"README.md"}