{"_id":"@davo20019/seo-audit","_rev":"8-76b81de56203744366dc338882cfbd43","name":"@davo20019/seo-audit","dist-tags":{"latest":"0.8.0"},"versions":{"0.2.0":{"name":"@davo20019/seo-audit","version":"0.2.0","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.2.0","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"67961ffbdda5d246d6ebadc8ba8b3ce8326a3e91","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.2.0.tgz","fileCount":37,"integrity":"sha512-glS2jooroOdv7b4/34RQxA7z67cD87/1fXJ/WBCpohW6yfneQQAVk+Uw5MdH00JVIoTIK0OLF+BpLo/PHX2X+Q==","signatures":[{"sig":"MEYCIQCTixL7v3VOirO1lBhAXmmrd8u61F630uqzPL9yEUJjkwIhANbQpyKLyKlsKaJfGdBdNqeb4Bd45ZUZLnzoLda3WYTp","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":177697},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"4cbfcc44fc7a1ff46fe0aefb64082e57f2a8759d","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.2.0_1777138048869_0.9579442360744481","host":"s3://npm-registry-packages-npm-production"}},"0.2.1":{"name":"@davo20019/seo-audit","version":"0.2.1","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.2.1","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"b85e98729f0a7ee64515a477fad4f4dc4aef1d90","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.2.1.tgz","fileCount":49,"integrity":"sha512-od4Hy9Nj6hdr9quGm3zQnvgYx4TRGMxhqDLb9bJtimAkGgmII4ZS+wk/nLlhiaJqIrWsdhFviJajNuanvn0daw==","signatures":[{"sig":"MEUCIQCYeWtmtlUzf6ug9s1WlvLpKdaTVOuFhlFpXz8YlA+v4AIgad4cRQ6u95YZjt3iUJHbNk3bsCs3bhCRYuvNI/nZTh4=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":251744},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"d15c2b7d029d9fcc9810123bf7e364520ee4a4b7","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.2.1_1777150228966_0.25177640314845706","host":"s3://npm-registry-packages-npm-production"}},"0.3.0":{"name":"@davo20019/seo-audit","version":"0.3.0","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.3.0","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"f3629960ee5a70e877f63fcaed54760f57bd4b37","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.3.0.tgz","fileCount":51,"integrity":"sha512-VAsgR2o4XV9cKoIra4BfnBMqR3Ds04XqGsppXVjEPD8qsYSbqAsHZu90//6TGZ8lignP365X6rvMLQGtzq2YqA==","signatures":[{"sig":"MEUCIQC5mAna6gDtbOmhfMWLlpx9xBoY9tSLYnoCCLb3zR4CxQIgMJCP8hEZtA0RYixE45XeTj3xzNN3hPcNGHaf29dqj4U=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":274031},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"77a215dba7931d02351dfdd72baf2b13c2d97a4e","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.3.0_1777160456206_0.42531553303145286","host":"s3://npm-registry-packages-npm-production"}},"0.4.0":{"name":"@davo20019/seo-audit","version":"0.4.0","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.4.0","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"93c23b8c1e5e538940f9bce7f1bb672983c7d63a","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.4.0.tgz","fileCount":53,"integrity":"sha512-I9Cqx3Hnsj4jFlzRZSRlwIP18t0i2FvAMo+i6pHQEg/D/zR1OFfM1kWlPcjYIA78JoZ5KRKh+ZOnGH0NrP+F7A==","signatures":[{"sig":"MEUCIQCdEFgLfHDqhq+QpuO6f2zZF2oPVrng7y1v/EDMexoAHAIgBvUws3quQVRAAinCDNgZmTDj6IABRW7emF4RzA2eXdo=","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":287828},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"2b0a89fa415bbd952abd67ca8f5baf48c4a017c8","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.4.0_1777212950997_0.9093890435068677","host":"s3://npm-registry-packages-npm-production"}},"0.5.0":{"name":"@davo20019/seo-audit","version":"0.5.0","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.5.0","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"9d2c303ce5f787492c040633ad1fdff713914732","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.5.0.tgz","fileCount":57,"integrity":"sha512-wAbHu3An4zrLwjkajtvyUUQKwLStPPtFVVh9KooQErOJjR28BTYVbDREC3BzF0ta1N+iOZIRqRjLpxd1umJXwQ==","signatures":[{"sig":"MEQCIF4qvcmIqArW435PXYDu24FwUWs3Ab+90ZzvzJEr65g3AiB/J2pMVSVdes46hPZ1Nu31rvk4xB0NQ9JTyzP0JUnqHw==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":309795},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"96a9796029a7ff55d898232cb539ff9ab59e28b1","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.5.0_1777220094113_0.39050256161867325","host":"s3://npm-registry-packages-npm-production"}},"0.6.0":{"name":"@davo20019/seo-audit","version":"0.6.0","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.6.0","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"cb78c77747e9ddef8d777e401e1362984adaf23b","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.6.0.tgz","fileCount":57,"integrity":"sha512-Zwm6kxTaL7BegA36lcvwJm+6fNtmkT+Agsm8fK96DBoTExqF9Tdugma+Mipf5cqoM7vf+8iBT0TlRLT0npGilQ==","signatures":[{"sig":"MEQCIHTyn3xQNFsHgYzU05Xq0HK3cp6ZPZ2WYTQ6FA0ejuQOAiArQxmLOCoeeKS9BujLLY6tD/Q355+S7WXrnTMTuUjxHg==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":317313},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"9fbf88a5941b0c19e5cee605b155e0333c3ebeea","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.6.0_1777224294244_0.38968483721882397","host":"s3://npm-registry-packages-npm-production"}},"0.7.0":{"name":"@davo20019/seo-audit","version":"0.7.0","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"author":{"name":"David Loor"},"license":"MIT","_id":"@davo20019/seo-audit@0.7.0","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"homepage":"https://github.com/davo20019/seo-analysis#readme","bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"bin":{"seo-audit":"dist/cli.js"},"dist":{"shasum":"8ac3986e59d5b18f1529074ec19ce99878918b5c","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.7.0.tgz","fileCount":59,"integrity":"sha512-Nj1ocxAOyZuJEuklPx4oT/8hCvxhzFW1sKGIltNdOQbWQmNu0HOcCwZAP9jJ7eIrEx0a5FmpC76cYHXD6hTaRA==","signatures":[{"sig":"MEQCIFe0GAo6DiyA3sqReO9MKE+4Zv6z4TxolGFMQ6Zdg/ndAiBk4cGtf3NxHYSA2TvHumwUvBZdxoAcHPStOJYUlj7DWA==","keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U"}],"unpackedSize":332498},"main":"./dist/index.js","type":"module","types":"./dist/index.d.ts","engines":{"node":">=20"},"exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"gitHead":"6d91556ea42c07b149761fe1ae4004515a150ffa","scripts":{"dev":"tsx src/cli.ts","test":"vitest run","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"repository":{"url":"git+https://github.com/davo20019/seo-analysis.git","type":"git"},"_npmVersion":"10.9.2","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","directories":{},"_nodeVersion":"22.13.0","dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"_hasShrinkwrap":false,"devDependencies":{"tsx":"^4.21.0","vitest":"^4.1.5","lighthouse":"^12.8.2","typescript":"^6.0.2","@types/node":"^25.6.0"},"_npmOperationalInternal":{"tmp":"tmp/seo-audit_0.7.0_1777251811106_0.3269382339396487","host":"s3://npm-registry-packages-npm-production"}},"0.8.0":{"name":"@davo20019/seo-audit","version":"0.8.0","type":"module","description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","license":"MIT","author":{"name":"David Loor"},"homepage":"https://github.com/davo20019/seo-analysis#readme","repository":{"type":"git","url":"git+https://github.com/davo20019/seo-analysis.git"},"bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"main":"./dist/index.js","types":"./dist/index.d.ts","exports":{".":{"types":"./dist/index.d.ts","default":"./dist/index.js"}},"bin":{"seo-audit":"dist/cli.js"},"engines":{"node":">=20"},"scripts":{"dev":"tsx src/cli.ts","build":"tsc -p tsconfig.json && node scripts/prepare-cli.cjs","start":"node dist/cli.js","test":"vitest run","test:watch":"vitest","prepublishOnly":"npm test && npm run build"},"dependencies":{"cheerio":"^1.2.0","playwright":"^1.59.1"},"devDependencies":{"@types/node":"^25.6.0","lighthouse":"^12.8.2","tsx":"^4.21.0","typescript":"^6.0.2","vitest":"^4.1.5"},"_id":"@davo20019/seo-audit@0.8.0","gitHead":"902c09fdaa3829f109591a2d5b41e0d9a3810ba2","_nodeVersion":"22.13.0","_npmVersion":"10.9.2","dist":{"integrity":"sha512-NdMaWXjmlOsE84VzC582OZTN9wdm4R7f+AUgd/TAS7mklhZH1I+fDwMKfB+xUnOHlASlclmcNasEgPVavHU+pA==","shasum":"90ec24f5ac869612e410321ae633eda921d711e0","tarball":"https://registry.npmjs.org/@davo20019/seo-audit/-/seo-audit-0.8.0.tgz","fileCount":69,"unpackedSize":372666,"signatures":[{"keyid":"SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U","sig":"MEUCIA6TFsBW+70+IbBxlXQMJP5w2iHwH3w5QAorcc9SOiRlAiEAwXLeqoYGeeSeddREpHxhG5hWqW1JxELPUZlSgy6cUvc="}]},"_npmUser":{"name":"davo20019","email":"davo20019@gmail.com"},"directories":{},"maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"_npmOperationalInternal":{"host":"s3://npm-registry-packages-npm-production","tmp":"tmp/seo-audit_0.8.0_1777295685037_0.1989127033268674"},"_hasShrinkwrap":false}},"time":{"created":"2026-04-25T17:27:28.787Z","modified":"2026-04-27T13:14:45.353Z","0.2.0":"2026-04-25T17:27:29.017Z","0.2.1":"2026-04-25T20:50:29.130Z","0.3.0":"2026-04-25T23:40:56.344Z","0.4.0":"2026-04-26T14:15:51.185Z","0.5.0":"2026-04-26T16:14:54.251Z","0.6.0":"2026-04-26T17:24:54.387Z","0.7.0":"2026-04-27T01:03:31.247Z","0.8.0":"2026-04-27T13:14:45.184Z"},"bugs":{"url":"https://github.com/davo20019/seo-analysis/issues"},"author":{"name":"David Loor"},"license":"MIT","homepage":"https://github.com/davo20019/seo-analysis#readme","keywords":["seo","audit","crawler","lighthouse","schema","json-ld","core-web-vitals","crux","sitemap","playwright","headless","cli","open-source"],"repository":{"type":"git","url":"git+https://github.com/davo20019/seo-analysis.git"},"description":"Crawl-based SEO audit CLI with JS rendering, JSON-LD validation, CrUX field data, sitemap walking, HTML/PDF reports, and crawl diffing.","maintainers":[{"name":"davo20019","email":"davo20019@gmail.com"}],"readme":"# SEO Analysis Tool\n\nThis project is a crawl-based SEO CLI for auditing websites.\n\nIt focuses on technical SEO issues that can be derived from the crawl itself, with optional headless-Chromium rendering (`--render`), Google CrUX field data (`--crux`), and Lighthouse audits for a small set of pages. Optional traffic enrichment from Google Search Console (`--gsc`) and Google Analytics 4 (`--ga4`) ranks issues by real-world impact.\n\nEvery successful audit is auto-persisted to `~/.config/seo-audit/crawls/`; `seo-audit diff <url>` compares the two most recent crawls of a host, and `--fail-on <severity>` gates CI/cron jobs against regressions vs. the previous persisted crawl. Opt out of persistence with `--no-persist` or `SEO_AUDIT_NO_PERSIST=1`.\n\n## What It Checks\n\n- missing, short, long, duplicate, and multiple title tags\n- missing, short, long, and duplicate meta descriptions\n- suspicious metadata values like `[object Object]`, `undefined`, or `null`\n- canonical issues, including missing, invalid, cross-host, and URL-mismatch canonicals\n- HTTP pages instead of HTTPS\n- locale-aware `html lang` checks\n- hreflang extraction and validation\n- hreflang duplicate values, missing `x-default`, missing self-reference, missing return links, redirecting targets, and locale-target mismatches\n- missing and multiple H1 headings\n- `noindex` directives\n- images without alt text\n- exact missing Open Graph fields\n- missing or invalid JSON-LD structured data\n- low body word count\n- pages with no crawlable internal links\n- internal links with missing anchor text\n- internal links with generic anchor text like `read more` or `click here`\n- broken internal links, redirecting internal links, and redirect chains\n- weak internal-link support, including pages with only one incoming internal link\n- orphan candidates based on the crawled internal-link graph\n- sitemap inclusion checks for crawled pages\n- sitemap-vs-canonical mismatches\n- missing or blocking `robots.txt`\n- missing `sitemap.xml`\n- missing or empty `llms.txt`\n- optional Lighthouse audits for performance, accessibility, best practices, and SEO\n- images missing explicit width and height attributes\n- images without `loading=\"lazy\"` hints\n- images served in legacy formats (jpg/png/gif) instead of webp/avif\n- missing or zoom-blocking viewport meta (mobile audit)\n- missing Strict-Transport-Security, missing Content-Type, or overly defensive Cache-Control on HTTP responses\n- `X-Robots-Tag` noindex/nofollow directives delivered via HTTP response headers\n- HTTP `Link: rel=\"canonical\"` mismatches with the HTML `<link rel=\"canonical\">`, multiple/invalid canonical Link values\n- HTML responses served without `Content-Encoding` (gzip/br/zstd) compression\n- compressed responses missing `Vary: Accept-Encoding` (shared-cache hazard)\n- URLs explicitly disallowed by `robots.txt` for the configured user agent\n- JSON-LD validation against rich-result requirements: Product, Article (BlogPosting/NewsArticle), FAQPage, BreadcrumbList, Organization, LocalBusiness\n- JSON-LD Product offers without `price` or `priceCurrency`\n- nested sitemap-index resolution (walks one level of nested sitemaps, capped at 50 children)\n- sitemap entries with `lastmod` older than 12 months\n- real-user Core Web Vitals from Google CrUX (LCP/INP/CLS p75) when `--crux` is enabled and `CRUX_API_KEY` is set\n- optional AI-agent readiness scoring (`--agent-readiness`) covering AI-bot rules, llms.txt depth, `llms-full.txt`, markdown content negotiation, well-known endpoints (`agent-skills`, `api-catalog`, `mcp/server-card`, OAuth discovery), Web Bot Auth, and `Link:` headers\n- optional Google Search Console enrichment (`--gsc`) merging clicks/impressions/CTR/avg-position per crawled URL, plus a \"Priority issues\" summary that ranks high/medium-severity issues by traffic exposure\n- optional Google Analytics 4 enrichment (`--ga4`) merging sessions/pageviews/users/engagement-rate per crawled URL; feeds the \"Priority issues\" summary as a fallback when GSC isn't available\n- crawl persistence + diffing: every audit auto-saves to `~/.config/seo-audit/crawls/<host>/<timestamp>.json`; `seo-audit diff <url>` auto-picks the two most recent crawls, and `--fail-on <severity>` gates CI/cron against regressions vs. the previous persisted crawl\n- near-duplicate content detection (MinHash, Jaccard ≥ 0.85 over 5-word shingles): clusters of pages with substantially similar body text get a \"Content duplicates\" summary section + medium-severity `CONTENT_NEAR_DUPLICATE` per-page issue. Skip with `--no-content-dedup`.\n- internal link equity (PageRank, damping 0.85, 20 iterations): per-page `pageRank` score plus an \"underlinked important pages\" highlight for high-content pages with below-median rank — the \"your money page gets 1 internal link\" insight. Skip with `--no-link-graph`.\n\n## Install\n\nRun instantly with `npx` (no install needed):\n\n```bash\nnpx @davo20019/seo-audit https://example.com\n```\n\nInstall globally:\n\n```bash\nnpm install -g @davo20019/seo-audit\nseo-audit https://example.com --max-pages 25\n```\n\nAdd to a project (use as a library too):\n\n```bash\nnpm install @davo20019/seo-audit\n```\n\n```ts\nimport { analyzeSite } from \"@davo20019/seo-audit\";\n\nconst report = await analyzeSite(\"https://example.com\", {\n  maxPages: 50,\n  onProgress: (event) => {\n    // Structured progress events for agents, jobs, and custom UIs.\n    if (event.phase === \"page-complete\") {\n      console.error(`crawled ${event.crawledPages}/${event.maxPages ?? \"all\"}`);\n    }\n  },\n});\n```\n\n## Quick Start (from source)\n\n```bash\ngit clone https://github.com/davo20019/seo-analysis.git\ncd seo-analysis\nnpm install\nnpm run dev -- https://example.com\n```\n\nBuild the CLI:\n\n```bash\nnpm run build\nnpm run start -- https://example.com --max-pages 20\n```\n\nWrite JSON output to a file:\n\n```bash\nnpm run dev -- https://example.com --json --output report.json\n```\n\nInteractive terminal runs show a single updating crawl progress line on stderr.\nIt is hidden automatically in CI/non-TTY runs and can be disabled with\n`--no-progress` or `SEO_AUDIT_NO_PROGRESS=1`, so JSON/stdout output remains\nmachine-readable for agents and scripts.\n\n## Useful Options\n\n| Flag | Description |\n|---|---|\n| `--output <path>` | Write JSON report to a file (default: stdout). |\n| `--extract <json>` | Inline JSON of extraction rules. Mutually exclusive with `--extract-file`. |\n| `--extract-file <path>` | JSON file of extraction rules. |\n| `--no-progress` | Disable the interactive stderr crawl progress line. |\n| `--no-persist` | Skip persisting the crawl to `~/.config/seo-audit/crawls/`. Default: every successful audit is persisted. Set `SEO_AUDIT_NO_PERSIST=1` to default the same. |\n| `--fail-on <severity>` | Exit non-zero if issues at `<severity>` increased. Fresh-audit mode: compares to the previous persisted crawl. Diff mode: compares the two passed report files. One of: `high`, `medium`, `low`. |\n\n```bash\n# Faster crawl with retries and sitemap seeding\nnpm run dev -- https://example.com --max-pages 50 --concurrency 6 --retries 2\n\n# Crawl every URL surfaced by discovered sitemap files\nnpm run dev -- https://example.com --full-sitemap --concurrency 12\n\n# Sample representative sitemap URLs up to --max-pages\nnpm run dev -- https://example.com --sample-sitemap --max-pages 25\n\n# Limit the crawl to a site section\nnpm run dev -- https://example.com --include-path '^/blog'\n\n# Skip utility or archive paths\nnpm run dev -- https://example.com --exclude-path '/tag/' --exclude-path '/page/[0-9]+'\n\n# Add Lighthouse for a few representative pages\nnpm run dev -- https://example.com --lighthouse --lighthouse-pages 3\n\n# Override the User-Agent string sent by the crawler\nnpm run dev -- https://example.com --user-agent \"Mozilla/5.0 (compatible; MyCrawler/1.0)\"\n```\n\n```bash\n# Render pages with headless Chromium (Playwright) — needed for SPAs,\n# JS-challenge sites (Cloudflare turnstile), and pages whose final DOM\n# depends on JS. Slower and heavier than the default static fetch.\nnpm run dev -- https://example.com --render --max-pages 5\n```\n\nNotes on `--render`:\n- First `npm install` auto-downloads Chromium (~300MB). To skip (e.g. CI), set `PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD=1` and run `npx playwright install chromium` later.\n- Rendering is slower than raw fetch (typical: 2–10s per page). Use `--max-pages` to scope.\n- Render uses `--retries` (default 3) with exponential backoff on transient failures (navigation timeouts, ad-script hangs).\n- Known limitation: `redirectChain` is not captured for rendered pages in v1. The `finalUrl` is still accurate.\n\n### Custom extractions\n\nDefine CSS-selector-based field extractions and pull them off every crawled\npage — useful for content audits, schema/data validation at scale, migration\nQA, and competitive teardowns.\n\n```bash\nseo-audit https://example.com \\\n  --extract '{\"h1\":\"h1\",\"price\":\"[itemprop=price]@content\"}' \\\n  --json --output report.json\n```\n\nOr with a config file (recommended for repeatable audits):\n\n```bash\necho '{\"h1\":\"h1\",\"author\":\"meta[name=author]@content\"}' > extractions.json\nseo-audit https://example.com --extract-file extractions.json --json\n```\n\n**Selector grammar** (suffix optional):\n\n| Suffix    | Returns                                       |\n|-----------|-----------------------------------------------|\n| *(none)*  | Trimmed text content                          |\n| `@attr`   | Attribute value (e.g. `meta[name=author]@content`) |\n| `#html`   | Inner HTML                                    |\n\n**Object form** unlocks `all` (multi-match → array) and `required` (emits a\nlow-severity `EXTRACTION_MISSING_REQUIRED` issue if no match — integrates\nwith `--fail-on low`):\n\n```json\n{\n  \"h1\": \"h1\",\n  \"price\": \"[itemprop=price]@content\",\n  \"intro\": \"article > p:first-of-type#html\",\n  \"faqQuestions\": { \"selector\": \".faq h3\", \"all\": true },\n  \"title\": { \"selector\": \"title\", \"required\": true }\n}\n```\n\nPer-page values appear under `pages[].extracted` in the JSON report.\nA summary appears at `extractionSummary`. The HTML/text/PDF reports show\nmatch coverage and missing-required counts only — full per-page detail\nstays in JSON for downstream tools (jq, spreadsheets, CI gates).\n\n### Log-file analysis (`seo-audit logs`)\n\nParse a web-server / CDN access log, verify bot identities via reverse-DNS,\nand join the results against the most recent persisted crawl. No server\naccess required — point the tool at a log file or pipe one in.\n\n```bash\n# Local file\nseo-audit logs ./access.log --site https://example.com\n\n# Stdin\nzcat cloudflare-logs-*.gz | seo-audit logs - --site https://example.com --format cloudflare\n\n# JSON output for downstream agents / scripts\nseo-audit logs ./access.log --site https://example.com --json --output logs.json\n```\n\n**Findings (when a persisted crawl exists for the host):**\n\n- **Orphan pages** — bot-visited URLs not in your internal link graph.\n- **Stale priorities** — top-PageRank URLs that bots haven't crawled in 30+ days.\n- **Status mismatches** — pages your audit recorded as 200 that bots saw return 4xx/5xx.\n\n**Supported formats:** Apache/Nginx Combined Log Format, generic JSON\n(one object per line), Cloudflare Logpush JSON, Fastly Real-Time JSON.\nAuto-detected; override with `--format`.\n\n**Bot verification** is on by default and uses reverse-DNS → forward-DNS\nsuffix matching with an in-memory cache. Disable with `--no-verify-bots`.\nIn sandboxed/air-gapped environments without DNS, the tool emits a\n`LOG_DNS_UNAVAILABLE` issue and continues with all hits flagged as\nunverified instead of stalling.\n\n**Privacy:** no raw IPs are written to any output; the DNS cache is\nin-memory only; no telemetry.\n\n```bash\n# Query Google's CrUX API for real-user Core Web Vitals (requires CRUX_API_KEY env var)\nCRUX_API_KEY=your-google-api-key npm run dev -- https://example.com --crux --max-pages 5\n```\n\n```bash\n# Score the site for AI-agent readiness (llms.txt depth, AI-bot policy, well-known endpoints)\nnpm run dev -- https://example.com --agent-readiness --max-pages 25\n```\n\n```bash\n# Enrich with Google Search Console traffic data (priority-rank issues by impressions)\nGOOGLE_APPLICATION_CREDENTIALS=./gsc-service-account.json \\\n  npm run dev -- https://example.com --gsc --max-pages 100\n```\n\nThe `--gsc` flag pulls clicks, impressions, CTR, and average position from Search Console for every crawled URL and adds a **Priority issues** section to the summary — high/medium-severity issues sorted by impressions, so the audit answers \"which problem affects pages that actually get traffic?\" rather than just \"what problems exist?\".\n\n**Auth** uses a Google Cloud service account (no OAuth browser flow, no token caching). The fastest way to set it up is the bundled wizard:\n\n```bash\nseo-audit --gsc-setup https://your-site.com\n```\n\nThe wizard:\n- Detects `gcloud` and (if present) creates the service account, downloads the key to `~/.config/seo-audit/gsc-key.json`, and chmods it `600`.\n- Falls back to printed step-by-step instructions if `gcloud` isn't available.\n- Prints the service-account email to grant in Search Console (Settings → Users and permissions → Add user → Restricted).\n- Verifies the credentials by exchanging them for a real access token before declaring success.\n\nAfter setup, every subsequent run is silent:\n\n```bash\nexport GOOGLE_APPLICATION_CREDENTIALS=~/.config/seo-audit/gsc-key.json\nseo-audit https://your-site.com --gsc\n```\n\nIf you prefer manual setup or are running in CI, the auth layer also accepts:\n- `GOOGLE_APPLICATION_CREDENTIALS=/path/to/key.json` (file path)\n- `GOOGLE_APPLICATION_CREDENTIALS_JSON='{...inline json...}'` (CI-friendly, one secret)\n- `--gsc-service-account-key-file /path/to/key.json` (CLI override)\n\nThe same credentials work for any future Google integration (e.g. a future `--ga4`).\n\nOptional flags: `--gsc-property` to override property auto-detection (URL-prefix or `sc-domain:example.com`), `--gsc-days` to change the lookback window (default 90).\n\n### Google Analytics 4 enrichment (`--ga4`)\n\nMerges per-page sessions, pageviews, users, and engagement rate from the GA4\nData API into the crawl report. Useful when you want to know which pages get\nreal traffic — not just search impressions — and prioritize technical-issue\nremediation accordingly.\n\n**Setup** (one-time, ~30 seconds if `--gsc-setup` has already run):\n\n1. If you haven't already run `seo-audit --gsc-setup`, run it now. It creates\n   a service account and downloads its JSON key. The same SA is reused for\n   GA4 — no second key needed.\n2. Open your GA4 property → **Admin → Property Access Management**.\n3. Click **+** → **Add users**. Paste the service-account email\n   (visible in `~/.config/seo-audit/gsc-key.json` under `client_email`).\n4. Set **Direct roles** to **Viewer**. Save.\n5. Run:\n\n   ```sh\n   GOOGLE_APPLICATION_CREDENTIALS=~/.config/seo-audit/gsc-key.json \\\n     seo-audit https://your-site.com --gsc --ga4\n   ```\n\nThe CLI auto-detects the GA4 property by matching the crawl origin against\neach accessible property's web data stream `defaultUri`. If multiple\nproperties match (e.g., separate prod/staging streams for the same domain),\nthe audit fails with the candidate list — pass `--ga4-property properties/N`\nto disambiguate.\n\n**Priority issues** are ranked by GSC impressions when `--gsc` is enabled,\nwith GA4 sessions as a fallback when GSC is unavailable. The HTML report\nshows a \"via GSC\" / \"via GA4\" badge per row.\n\nThe `--agent-readiness` flag adds an opinionated rubric (inspired by [Cloudflare's agent-readiness framework](https://blog.cloudflare.com/agent-readiness/)) covering four buckets:\n\n- **Discoverability** — robots.txt, sitemap.xml, `Link:` HTTP headers (RFC 8288).\n- **Content accessibility** — llms.txt presence and content depth (H1, sections, links, size), `llms-full.txt`, and markdown content negotiation (`Accept: text/markdown`).\n- **Bot access control** — explicit rules for known AI user agents (GPTBot, ClaudeBot, PerplexityBot, Google-Extended, CCBot, Bytespider, Applebot-Extended, …), `Content-Signal` directives (`search`, `ai-train`, `ai-input`), and Web Bot Auth (`/.well-known/http-message-signatures-directory`).\n- **Capabilities & protocols** — well-known endpoints (`/.well-known/agent-skills/index.json`, `/.well-known/api-catalog`, `/.well-known/mcp/server-card.json`, OAuth discovery) plus structured-data coverage from an agent perspective (Organization/WebSite on the homepage, Article schema on article-like paths).\n\nEach bucket is scored 0–100 and averaged into a single `score` (also 0–100). The full breakdown — including which probes succeeded — lands in the JSON, text, and HTML reports as a separate `Agent Readiness` section.\n\n```bash\n# Generate a polished HTML report (open in any browser, email to a client)\nnpm run dev -- https://example.com --max-pages 25 --html-report report.html\n\n# Generate a PDF report (uses Playwright/Chromium under the hood)\nnpm run dev -- https://example.com --max-pages 25 --pdf-report report.pdf\n\n# Generate both at once\nnpm run dev -- https://example.com --max-pages 25 --html-report report.html --pdf-report report.pdf\n```\n\n## Diff Reports\n\nCompare two crawl reports to see what changed between runs:\n\n```bash\n# Compare last week's crawl to this week's\nnpm run dev -- diff old-report.json new-report.json\n\n# Output as JSON for piping into another tool\nnpm run dev -- diff old.json new.json --json --output diff.json\n\n# Fail (non-zero exit) if any high-severity issues were added — useful in CI\nnpm run dev -- diff old.json new.json --fail-on high\n```\n\nThe diff highlights:\n- New issue codes (didn't appear in old)\n- Resolved issue codes (appeared in old, not in new)\n- Counts that increased or decreased\n- HTTP status changes on common pages\n- Pages added or removed from the crawl\n\n## GitHub Action\n\nRun SEO audits in CI without writing any glue code:\n\n```yaml\n# .github/workflows/seo.yml\nname: SEO Audit\non: [pull_request]\n\njobs:\n  audit:\n    runs-on: ubuntu-latest\n    steps:\n      - uses: actions/checkout@v4\n      - uses: davo20019/seo-analysis@v1\n        with:\n          url: https://staging.mysite.com\n          max-pages: 50\n          fail-on: high   # block PRs that introduce high-severity issues\n```\n\n### Inputs\n\n| Input | Default | Description |\n|---|---|---|\n| `url` | (required) | URL to crawl |\n| `max-pages` | `50` | Maximum pages to crawl |\n| `concurrency` | (CLI default) | Pages fetched in parallel |\n| `render` | `false` | Use Playwright for JS rendering |\n| `fail-on` | (none) | Fail the workflow if issues at this severity exist (`high`, `medium`, `low`) |\n| `crux` | `false` | Query Google CrUX for real-user Core Web Vitals |\n| `crux-api-key` | (none) | API key for CrUX (use a repo secret) |\n| `agent-readiness` | `false` | Score AI-agent readiness (llms.txt depth, AI-bot rules, well-known endpoints) |\n| `gsc` | `false` | Enrich crawled pages with Google Search Console clicks/impressions/CTR/position |\n| `gsc-property` | (auto-detect) | Override GSC property (URL-prefix or `sc-domain:example.com`) |\n| `gsc-days` | `90` | Days of GSC history to query |\n| `gsc-service-account-key` | (none) | Service-account JSON for GSC (use a repo secret); the SA email needs Restricted access on the property |\n| `ga4` | `false` | Enrich crawled pages with Google Analytics 4 metrics (sessions, pageviews, users, engagement rate) |\n| `ga4-property` | (auto-detect) | Override property auto-detection (e.g. `properties/123456789`) |\n| `ga4-days` | `90` | Days of GA4 data to query |\n| `ga4-service-account-key` | (none) | Service-account JSON for GA4 (use a repo secret); the SA email needs Viewer role on the property |\n| `user-agent` | (default) | Override the crawler's User-Agent |\n| `include-paths` | (none) | Comma-separated regex; only crawl matching URLs |\n| `exclude-paths` | (none) | Comma-separated regex; skip matching URLs |\n| `output-json` | `seo-report.json` | Where to write the JSON report |\n| `output-html` | (none) | Optional path for the HTML report |\n\n### Outputs\n\n| Output | Description |\n|---|---|\n| `high-issues` | Count of high-severity issues |\n| `medium-issues` | Count of medium-severity issues |\n| `low-issues` | Count of low-severity issues |\n| `pages-crawled` | Number of pages successfully crawled |\n| `report-path` | Path to the JSON report (use with `actions/upload-artifact`) |\n\n## Keyword Search\n\nSearch for specific keywords across a site:\n\n```bash\n# Search for keywords\nnpm run dev -- https://example.com --keyword \"seo audit\" --keyword \"site speed\"\n\n# Load keywords from a file (one per line, # comments supported)\nnpm run dev -- https://example.com --keyword-file keywords.txt\n\n# Combine keyword search with term extraction\nnpm run dev -- https://example.com --keyword-file keywords.txt --extract-terms --top-terms 30\n```\n\n## Offline Search\n\nSearch pre-downloaded HTML files for faster repeated searches:\n\n```bash\n# Download a site with httrack\nhttrack \"https://example.com\" -O \"./site_backup\"\n\n# Search the local copy (no network requests)\nnpm run dev -- --from-directory ./site_backup --keyword-file keywords.txt\nnpm run dev -- --from-directory ./site_backup --extract-terms\n```\n\n## Crawl Behavior\n\n- crawls up to `--max-pages` pages, or the full sitemap with `--full-sitemap`\n- fetches pages in parallel with `--concurrency` (run `--help` for the default)\n- retries slow or retryable requests with exponential backoff (also applies to `--render` on transient navigation failures)\n- deduplicates redirected pages by final URL\n- seeds the crawl queue from `sitemap.xml` automatically\n- walks one level of nested sitemap-index files (capped at 50 children)\n- supports `--include-path` and `--exclude-path` regex filters\n- samples representative sitemap URLs with `--sample-sitemap`\n- captures response headers per page for `X-Robots-Tag`, HSTS, Cache-Control, and Content-Type checks\n- optionally renders pages with headless Chromium via `--render` (Playwright) for SPAs and JS-challenge sites\n\n## Notes And Limits\n\n- `--sample-sitemap` is useful when you want representative sitemap coverage quickly, but link-graph findings are less complete because the crawl is intentionally sampled.\n- Sitemap reconciliation relies on the set of sitemap URLs the CLI was able to collect. The report marks sitemap coverage as partial when that set was truncated.\n- JSON-LD validation covers Google's rich-result requirements for Product, Article (BlogPosting/NewsArticle), FAQPage, BreadcrumbList, Organization, and LocalBusiness. Other schema types are still presence-only.\n- `--render` (Playwright/Chromium) handles SPAs and JS-challenge sites (Cloudflare turnstile, JS-rendered DOM) but `redirectChain` is not captured for rendered pages.\n\n## Privacy\n\nThis tool does not collect telemetry. No analytics, no phone-home, no install tracking. Crawl reports stay on your machine. The only outbound network traffic is:\n\n- HTTP fetches to the URLs you ask the tool to crawl\n- Optional: Google's CrUX API when you pass `--crux` (sends an origin string + your API key)\n- Optional: Chromium downloads from Microsoft's Playwright CDN on first install\n\nIf a future version ever adds opt-in telemetry, it will be exactly that — opt-in, with explicit disclosure.\n\n### Persistence\n\nEvery successful audit is auto-saved to `~/.config/seo-audit/crawls/<host>/<timestamp>.json`\n(mode `0700`, alongside the existing `gsc-key.json`). This enables:\n\n- `seo-audit diff <url>` — compare the two most recent crawls of a host without\n  having to remember file paths.\n- `seo-audit <url> --fail-on <severity>` — gate cron / CI runs against\n  regressions vs. the previous persisted crawl.\n\n**The persisted JSON contains everything the audit captured**, including page\nmetadata, GSC/GA4 traffic data when enriched (`--gsc` / `--ga4`), and the full\nbody text per page. Treat the directory as you would any other client-data\nartifact.\n\n**Opt out** with `--no-persist` per run, or `SEO_AUDIT_NO_PERSIST=1` (or\n`SEO_AUDIT_NO_PERSIST=true`) in your environment.\n\n**Override the location** with `SEO_AUDIT_CRAWLS_DIR=<path>` — useful when the\ndefault `$HOME/.config/` doesn't fit (sandboxed CI runners, separate volume, XDG\npreferences).\n\n**Retention:** none. Crawls accumulate forever. At ~5 MB per crawl × 250 weekly\ncrawls × 5 years per host, the footprint is around 1.3 GB per host long-term —\nmanageable. `rm -rf ~/.config/seo-audit/crawls/<host>/` if you ever want to\nreset.\n\n## CHANGELOG\n\n### v0.8.0 — 2026-04-27\n\n- **Added:** `seo-audit logs <path>` subcommand. Parses Apache/Nginx\n  Combined Log Format, generic JSON, Cloudflare Logpush, and Fastly\n  Real-Time logs. Verifies Googlebot/Bingbot/Applebot/DuckDuckBot/AI\n  bots via reverse-DNS with in-memory caching. Auto-falls-back to\n  unverified mode when DNS is unavailable.\n- **Added:** Joined-to-crawl findings — orphan pages\n  (`LOG_ORPHAN_PAGE`), stale priorities (`LOG_STALE_PRIORITY_PAGE`),\n  status mismatches (`LOG_STATUS_MISMATCH`). All low severity,\n  integrate with `--fail-on`.\n- **Added:** `analyzeLogs(input, options)` library API accepting a\n  path or `Readable` stream. `LogAnalysisReport` shape stable from v0.8.\n- **Added:** Structured `LogAnalysisProgressEvent` callback for\n  long-running log analyses.\n- **Privacy:** no raw IPs in any output, DNS cache in-memory only,\n  no telemetry.\n\n### v0.7.0 — 2026-04-26\n\n- **Added:** custom extraction rules. Define CSS-selector-based field\n  extractions with `--extract '<json>'` or `--extract-file <path>`. Selector\n  grammar: `selector` (text), `selector@attr` (attribute), `selector#html`\n  (inner HTML). Object form supports `all: true` (multi-match) and\n  `required: true` (low-severity `EXTRACTION_MISSING_REQUIRED` issue,\n  integrates with `--fail-on`).\n- **Added:** `AnalyzeOptions.extract` for library consumers, plus\n  `PageReport.extracted` and `SiteReport.extractionSummary` in the JSON\n  report.\n- **Added:** \"Custom extractions\" summary section in HTML/text/PDF reports.\n  Full per-page detail remains in JSON.\n\n### v0.6.0 — 2026-04-26\n\n- **Added:** interactive crawl progress for CLI runs. The indicator writes to\n  stderr only, is enabled only for TTY sessions, and is disabled in CI/non-TTY\n  contexts. Use `--no-progress` or `SEO_AUDIT_NO_PROGRESS=1` to opt out.\n- **Added:** `AnalyzeOptions.onProgress`, a structured progress callback for\n  library, job-runner, and agent integrations that need status updates without\n  parsing terminal output.\n\n### v0.5.0 — 2026-04-26\n\n- **Added:** near-duplicate content detection. Audits now include a\n  `report.contentDedup` summary with clusters of pages whose body text\n  exceeds 85% Jaccard similarity. Each cluster member gets a new\n  medium-severity `CONTENT_NEAR_DUPLICATE` issue. Skip via\n  `--no-content-dedup`.\n- **Added:** internal link-equity (PageRank) computation. Audits now\n  include a `report.linkGraph` summary with the top 10 pages by\n  PageRank and an \"underlinked important pages\" list (high content,\n  low rank — the \"your money page gets 1 internal link\" finding).\n  Each `PageReport` gains a `linkGraph.pageRank` field. Skip via\n  `--no-link-graph`.\n- **Changed:** the HTML report Pages table grows a conditional\n  `PageRank` column when any page has link-graph metrics, sortable\n  alongside the existing Title / Status / Issues columns.\n- **Note for `--fail-on` cron users:** the new\n  `CONTENT_NEAR_DUPLICATE` issue is medium severity and will surface\n  on most e-commerce / programmatic-SEO sites. **If you have an existing\n  persisted baseline from v0.4 (or earlier),** the first v0.5 run may\n  exit non-zero because the new check wasn't running when the baseline\n  was recorded — the new findings look like a regression to the diff\n  comparison. Recovery: run the audit once with `--no-content-dedup`\n  to refresh a clean baseline, then re-enable it; or raise\n  `--fail-on high`; or pass `--no-content-dedup` permanently to opt\n  out. Users on a fresh install (no prior crawl) are unaffected — the\n  first run skips the regression check entirely with a \"No prior crawl\n  found\" warning and exits 0.\n\n### v0.4.0 — 2026-04-26\n\n- **Added:** every successful audit is now persisted to\n  `~/.config/seo-audit/crawls/<host>/<timestamp>.json` (mode `0700`).\n  Opt out via `--no-persist` or `SEO_AUDIT_NO_PERSIST=1`.\n- **Added:** `seo-audit diff <url>` auto-picks the two most recent\n  persisted crawls of the host. The existing\n  `seo-audit diff <old.json> <new.json>` form still works for\n  explicit comparisons.\n- **Added:** `SEO_AUDIT_CRAWLS_DIR` env var overrides the default\n  crawls directory location.\n- **Added:** `evaluateFailOn` helper exported from `dist/diff.js`\n  for programmatic consumers.\n- **Changed:** `--fail-on <severity>` now applies in fresh-audit mode\n  too. It compares the new audit against the previous persisted crawl\n  and exits non-zero if issues at the named severity increased. On the\n  first run for a host, prints a \"skipping regression check\" warning\n  and exits `0`. (Diff-mode behavior is unchanged.)\n\n### v0.3.0 — 2026-04-25\n\n- **Added:** `--ga4` flag and supporting `--ga4-property`, `--ga4-days`,\n  `--ga4-service-account-key-file` options. Enriches each crawled page with\n  GA4 sessions, pageviews, users, and engagement rate; reuses the\n  service-account workflow set up by `--gsc-setup`.\n- **Added:** `ga4`, `ga4-property`, `ga4-days`, and `ga4-service-account-key`\n  inputs on the GitHub Action. The `ga4-service-account-key` input falls back\n  to `gsc-service-account-key` when blank, since the same SA can read both.\n- **Added:** `report.ga4` enrichment-result summary alongside `report.gsc`.\n- **Added:** `page.metrics.ga4` per-page GA4 metrics alongside\n  `page.metrics.gsc`.\n- **Changed:** `summary.priorityIssues[]` entries now carry `rankedBy`,\n  `rankValue`, and `metrics` fields; the GSC-specific `impressions`, `clicks`,\n  and `position` fields are kept populated for one minor cycle (deprecated).\n- **Changed:** `summary.priorityIssues` is now built whenever GSC **or** GA4\n  enrichment succeeds (previously only when GSC succeeded).\n- **Changed:** HTML report now renders `<h2>Analytics</h2>` (when GA4 is\n  enabled) and `<h2>Priority issues</h2>` as top-level sections; the priority\n  list previously rendered nested under `<h2>Search Console</h2>`, which hid\n  it on GA4-only audits.\n\n## License\n\n[MIT](LICENSE)\n","readmeFilename":"README.md"}