local-seo-lint 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Clicks Dynasty
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,233 @@
1
+ # local-seo-lint
2
+
3
+ **Catch local SEO and AI-search mistakes before they go live.**
4
+
5
+ `local-seo-lint` scans a static website folder and flags the problems that quietly hurt local rankings: a phone number that's different on one page, schema that disagrees with the page, two pages fighting for the same search, pages Google can't find, and more.
6
+
7
+ - Zero dependencies. Node 18+.
8
+ - Works on any static site: plain HTML, Hugo, Jekyll, Eleventy, Astro, Next.js static export.
9
+ - Runs locally, in CI, or as a GitHub Action with inline annotations on pull requests.
10
+ - Nothing is sent anywhere. It only reads your files.
11
+
12
+ ## Why this exists
13
+
14
+ General SEO linters check one page at a time: is there a title, is there a meta description. Local SEO problems usually live **between** pages:
15
+
16
+ - Your homepage schema says one phone number, your contact page shows another.
17
+ - The footer says "Suite 4" on most pages and leaves it out on one.
18
+ - Four pages all target "Bakersfield digital marketing agency", so Google splits the credit and none of them rank.
19
+ - A new landing page was never linked from anywhere, so Google never finds it.
20
+
21
+ We built this at [Clicks Dynasty](https://www.clicksdynasty.com), a local SEO agency, after finding exactly these problems on real sites, including our own. Running it against an older version of our site flagged the four pages that were all competing for "Bakersfield digital marketing agency", the same problem we had found by hand.
22
+
23
+ ## Quick start
24
+
25
+ ```bash
26
+ # Run straight from GitHub, no install
27
+ npx github:clicksdynastyincali/local-seo-lint ./my-site
28
+
29
+ # Or clone and run
30
+ git clone https://github.com/clicksdynastyincali/local-seo-lint.git
31
+ node local-seo-lint/bin/local-seo-lint.js ./my-site
32
+ ```
33
+
34
+ Point it at the folder that contains your HTML files (for generated sites, the build output such as `public/` or `dist/`).
35
+
36
+ ### Example output
37
+
38
+ ```text
39
+ local-seo-lint 5 pages scanned
40
+ Business: Oak Street Plumbing | 123 Example Ave, Suite 4 | +1-661-555-0123 (from schema on index.html)
41
+
42
+ error broken-link (1) Internal link points to a page that does not exist.
43
+ index.html Link to "pricing.html" goes nowhere.
44
+
45
+ error json-ld-syntax (1) JSON-LD blocks must be valid JSON, or search engines ignore them.
46
+ services.html JSON-LD block #1 is invalid: Expected double-quoted property name in JSON at position 77 (line 1 column 78)
47
+
48
+ error nap-schema-mismatch (1) Business schema on a page disagrees with your business name, address or phone.
49
+ contact.html Schema phone +1-661-555-0100 differs from your business phone +1-661-555-0123.
50
+
51
+ error sitemap-dead-url (1) sitemap.xml lists a URL with no matching page.
52
+ sitemap.xml Lists https://www.example.com/deleted-page.html, but no such page exists.
53
+
54
+ warn canonical-missing (1) Page has no canonical tag.
55
+ contact.html Add <link rel="canonical" href="https://www.example.com/contact.html">.
56
+
57
+ warn description-duplicate (2) Several pages share the same meta description.
58
+ drain-cleaning-bakersfield.html Same meta description as services.html.
59
+ services.html Same meta description as drain-cleaning-bakersfield.html.
60
+
61
+ warn keyword-overlap (1) Two pages target nearly the same search, so they compete with each other.
62
+ drain-cleaning-bakersfield.html "Drain Cleaning Services in Bakersfield" competes with services.html ("Drain Cleaning Services in Bakersfield"), 100% overlap. Pick one page for this search.
63
+
64
+ warn nap-address-mismatch (1) Your address is written differently on this page.
65
+ contact.html Address written as "123 Example Ave, Bakersfield anytime. Oak Str"; standard form is "123 Example Ave, Suite 4".
66
+
67
+ warn nap-phone-mismatch (1) A phone number on the page differs from your business phone.
68
+ services.html Shows (661) 555-0199; your business phone is +1-661-555-0123.
69
+
70
+ warn orphan-page (1) No other page links here, so crawlers struggle to find it.
71
+ drain-cleaning-bakersfield.html No other page links to this page.
72
+
73
+ warn schema-not-visible (1) Schema says something the visitor cannot see on the page.
74
+ contact.html Schema phone +1-661-555-0100 is not visible on the page.
75
+
76
+ warn sitemap-page-missing (1) Indexable page is not listed in sitemap.xml.
77
+ drain-cleaning-bakersfield.html Not listed in sitemap.xml.
78
+
79
+ warn title-duplicate (2) Several pages share the same title.
80
+ drain-cleaning-bakersfield.html Same title as services.html.
81
+ services.html Same title as drain-cleaning-bakersfield.html.
82
+
83
+ 4 errors, 11 warnings
84
+ ```
85
+
86
+ This is the output for the small demo site in [`test/fixtures/site`](test/fixtures/site).
87
+
88
+ ## What it checks
89
+
90
+ | Rule | Default | What it catches |
91
+ |---|---|---|
92
+ | `nap-schema-mismatch` | error | Business schema on a page has a different phone, street or ZIP than your business |
93
+ | `nap-phone-mismatch` | warn | A phone number shown on a page isn't your business number |
94
+ | `nap-address-mismatch` | warn | Your street address is written differently (knows that "Avenue #4" = "Ave, Suite 4") |
95
+ | `schema-not-visible` | warn | Schema claims a name or phone that visitors can't see on the page |
96
+ | `json-ld-syntax` | error | A JSON-LD block is invalid JSON, so search engines ignore it |
97
+ | `keyword-overlap` | warn | Two pages target nearly the same search (keyword cannibalization) |
98
+ | `orphan-page` | warn | No other page links to this page |
99
+ | `broken-link` | error | Internal link to a page or file that doesn't exist |
100
+ | `sitemap-page-missing` | warn | Indexable page not listed in `sitemap.xml` |
101
+ | `sitemap-dead-url` | error | `sitemap.xml` lists a page that doesn't exist |
102
+ | `sitemap-noindex` | error | `sitemap.xml` lists a page marked `noindex` |
103
+ | `sitemap-missing` | warn | No `sitemap.xml` at all |
104
+ | `canonical-missing` | warn | Page has no canonical tag |
105
+ | `canonical-multiple` | error | More than one canonical tag |
106
+ | `canonical-relative` | warn | Canonical URL isn't absolute |
107
+ | `canonical-other-page` | warn | Canonical points to a different page (so this one won't rank) |
108
+ | `canonical-broken` | error | Canonical points to a page that doesn't exist |
109
+ | `title-missing` / `title-duplicate` | error / warn | Missing or repeated `<title>` |
110
+ | `description-missing` / `description-duplicate` | warn | Missing or repeated meta description |
111
+ | `h1-missing` | warn | Page has no `<h1>` |
112
+ | `ai-crawler-blocked` | error | `robots.txt` blocks an AI search crawler (OAI-SearchBot, ChatGPT-User, PerplexityBot, Claude-SearchBot, Bingbot, Applebot), so the business can't appear in AI answers |
113
+ | `ai-entity-links-missing` | warn | Homepage business schema has no `sameAs` links (Google Business Profile, Facebook, LinkedIn, Yelp), so AI tools can't confirm which business it is |
114
+
115
+ Pages marked `noindex`, `404.html`, and pages that are 301-redirected in `.htaccess` or `_redirects` are skipped where it makes sense.
116
+
117
+ ### How business details are found
118
+
119
+ NAP checks compare every page against one "source of truth":
120
+
121
+ 1. The `business` block in `.localseorc.json`, if you have one, or
122
+ 2. The first LocalBusiness/Organization schema with an address or phone on your homepage.
123
+
124
+ ### How keyword overlap works
125
+
126
+ Each page's "target phrases" are its title segments (split on `|`, `-`, `:`) plus its `<h1>`, with your brand name removed. Phrases are compared using overlap weighted by how rare each word is on your site, so words used everywhere (your city, "services") count less than distinctive ones. A title suffix repeated on many pages (like "| Bakersfield & Central Valley, CA") is treated as boilerplate, not a target.
127
+
128
+ ## Configuration
129
+
130
+ Optional. Create `.localseorc.json` in your site folder:
131
+
132
+ ```json
133
+ {
134
+ "siteUrl": "https://www.example.com",
135
+ "business": {
136
+ "name": "Oak Street Plumbing",
137
+ "phone": "+1-661-555-0123",
138
+ "address": { "street": "123 Example Ave, Suite 4", "postalCode": "93301" }
139
+ },
140
+ "ignore": ["drafts/**", "backup/**"],
141
+ "overlap": { "threshold": 0.8 },
142
+ "rules": {
143
+ "description-duplicate": "off",
144
+ "orphan-page": "error"
145
+ }
146
+ }
147
+ ```
148
+
149
+ Every rule can be set to `"off"`, `"warn"` or `"error"`.
150
+
151
+ ## CLI options
152
+
153
+ ```text
154
+ Usage: local-seo-lint [folder] [options]
155
+
156
+ Checks a static website folder for local SEO problems.
157
+
158
+ Options:
159
+ --config <file> Config file (default: .localseorc.json in the folder)
160
+ --ignore <glob> Skip files/folders (repeatable), e.g. --ignore "drafts/**"
161
+ --format <pretty|json|github> Output format (default: pretty)
162
+ --limit <n> Max findings shown per rule in pretty output (0 = all, default 10)
163
+ --max-warnings <n> Fail if there are more than n warnings (default: no limit)
164
+ --no-color Disable colors
165
+ --no-footer Hide the "need help?" line under the results
166
+ -h, --help Show this help
167
+ -v, --version Show version
168
+
169
+ Exit code: 1 if any errors (or too many warnings), otherwise 0.
170
+ ```
171
+
172
+ ## GitHub Action
173
+
174
+ Add [`examples/local-seo-lint.yml`](examples/local-seo-lint.yml) to `.github/workflows/` in your website repository:
175
+
176
+ ```yaml
177
+ # Save as .github/workflows/local-seo-lint.yml in your website repository.
178
+ name: Local SEO Lint
179
+ on:
180
+ push:
181
+ pull_request:
182
+ jobs:
183
+ lint:
184
+ runs-on: ubuntu-latest
185
+ steps:
186
+ - uses: actions/checkout@v4
187
+ - uses: actions/setup-node@v4
188
+ with:
189
+ node-version: 20
190
+ - uses: clicksdynastyincali/local-seo-lint@main
191
+ with:
192
+ path: . # or your build output folder, e.g. public/ or dist/
193
+ ignore: "drafts/**"
194
+ ```
195
+
196
+ Problems appear as annotations on the changed files in pull requests, and errors fail the check.
197
+
198
+ ### AI search checks
199
+
200
+ More customers now ask ChatGPT and Perplexity to recommend local businesses. In our [California AI Visibility Index](https://www.clicksdynasty.com/california-ai-visibility-index.html), the two tools agreed on only about 1 in 4 of the businesses they recommended. Two things on your own site decide whether you can show up at all:
201
+
202
+ - **Crawler access.** If `robots.txt` blocks the crawlers that fetch pages for AI answers, those tools can't read your site. Blocking *training* crawlers such as `GPTBot` or `ClaudeBot` is your choice and is not flagged.
203
+ - **Entity links.** `sameAs` links in your business schema tie your website to your Google, Yelp and social profiles, which is how AI tools check they have the right business.
204
+
205
+ ## Limitations
206
+
207
+ - Built for static HTML. It does not run JavaScript, so content injected client-side isn't seen.
208
+ - Phone detection is tuned for US/Canada formats.
209
+ - It checks your own files, not third-party listings (Google Business Profile, Yelp and so on).
210
+ - Redirect detection understands simple `.htaccess` `Redirect`/`RewriteRule [R=301]` lines and Netlify-style `_redirects`; complex regex rules are ignored.
211
+
212
+ ## Roadmap
213
+
214
+ - Opening-hours consistency between schema and visible text
215
+ - Compare site details with a Google Business Profile export
216
+ - Near-duplicate location page detection (doorway pages)
217
+ - International phone formats
218
+
219
+ Issues and pull requests are welcome.
220
+
221
+ ## For agencies
222
+
223
+ Running this on client sites and finding more than your team has time to fix? [Clicks Dynasty](https://www.clicksdynasty.com) fixes local SEO and AI-visibility issues white-label, under your brand, and never contacts your clients. [Book a free call](https://www.clicksdynasty.com/book.html?service=strategy).
224
+
225
+ ## Development
226
+
227
+ ```bash
228
+ npm test
229
+ ```
230
+
231
+ ## License
232
+
233
+ MIT. Built by [Clicks Dynasty](https://www.clicksdynasty.com), a local SEO and digital marketing agency in Bakersfield, California.
package/action.yml ADDED
@@ -0,0 +1,40 @@
1
+ name: 'Local SEO Lint'
2
+ description: 'Catch local SEO and AI-search mistakes: blocked AI crawlers, NAP mismatches, schema, canonicals and sitemaps.'
3
+ author: 'Clicks Dynasty'
4
+ branding:
5
+ icon: 'map-pin'
6
+ color: 'green'
7
+ inputs:
8
+ path:
9
+ description: 'Folder containing your built HTML site'
10
+ required: false
11
+ default: '.'
12
+ config:
13
+ description: 'Path to a config file (default: .localseorc.json inside the folder)'
14
+ required: false
15
+ default: ''
16
+ ignore:
17
+ description: 'Comma-separated globs to skip, e.g. "drafts/**,backup/**"'
18
+ required: false
19
+ default: ''
20
+ max-warnings:
21
+ description: 'Fail the job if there are more warnings than this (leave empty for no limit)'
22
+ required: false
23
+ default: ''
24
+ runs:
25
+ using: 'composite'
26
+ steps:
27
+ - name: Run local-seo-lint
28
+ shell: bash
29
+ env:
30
+ LSL_PATH: ${{ inputs.path }}
31
+ LSL_CONFIG: ${{ inputs.config }}
32
+ LSL_IGNORE: ${{ inputs.ignore }}
33
+ LSL_MAXW: ${{ inputs.max-warnings }}
34
+ run: |
35
+ args=("$LSL_PATH" --format github --limit 0)
36
+ [ -n "$LSL_CONFIG" ] && args+=(--config "$LSL_CONFIG")
37
+ [ -n "$LSL_MAXW" ] && args+=(--max-warnings "$LSL_MAXW")
38
+ IFS=',' read -ra globs <<< "$LSL_IGNORE"
39
+ for g in "${globs[@]}"; do g="$(echo "$g" | xargs)"; [ -n "$g" ] && args+=(--ignore "$g"); done
40
+ node "${{ github.action_path }}/bin/local-seo-lint.js" "${args[@]}"
@@ -0,0 +1,53 @@
1
+ #!/usr/bin/env node
2
+ 'use strict';
3
+ const path = require('path');
4
+ const { lint } = require('../src');
5
+ const { pretty, github } = require('../src/report');
6
+
7
+ const HELP = `Usage: local-seo-lint [folder] [options]
8
+
9
+ Checks a static website folder for local SEO problems.
10
+
11
+ Options:
12
+ --config <file> Config file (default: .localseorc.json in the folder)
13
+ --ignore <glob> Skip files/folders (repeatable), e.g. --ignore "drafts/**"
14
+ --format <pretty|json|github> Output format (default: pretty)
15
+ --limit <n> Max findings shown per rule in pretty output (0 = all, default 10)
16
+ --max-warnings <n> Fail if there are more than n warnings (default: no limit)
17
+ --no-color Disable colors
18
+ --no-footer Hide the "need help?" line under the results
19
+ -h, --help Show this help
20
+ -v, --version Show version
21
+
22
+ Exit code: 1 if any errors (or too many warnings), otherwise 0.`;
23
+
24
+ function main(argv) {
25
+ const args = { ignore: [], format: 'pretty', limit: 10, color: process.stdout.isTTY, footer: true };
26
+ let dir = '.';
27
+ for (let i = 0; i < argv.length; i++) {
28
+ const a = argv[i];
29
+ const val = () => { if (i + 1 >= argv.length) throw new Error(`${a} needs a value`); return argv[++i]; };
30
+ if (a === '-h' || a === '--help') { console.log(HELP); return 0; }
31
+ else if (a === '-v' || a === '--version') { console.log(require('../package.json').version); return 0; }
32
+ else if (a === '--config') args.config = val();
33
+ else if (a === '--ignore') args.ignore.push(val());
34
+ else if (a === '--format') args.format = val();
35
+ else if (a === '--limit') args.limit = parseInt(val(), 10) || Infinity;
36
+ else if (a === '--max-warnings') args.maxWarnings = parseInt(val(), 10);
37
+ else if (a === '--no-color') args.color = false;
38
+ else if (a === '--no-footer') args.footer = false;
39
+ else if (a.startsWith('-')) throw new Error(`Unknown option ${a}`);
40
+ else dir = a;
41
+ }
42
+ const root = path.resolve(dir);
43
+ const res = lint(root, { config: args.config, ignore: args.ignore });
44
+ if (args.format === 'json') console.log(JSON.stringify(res, null, 2));
45
+ else if (args.format === 'github') { const g = github(res, path.relative(process.cwd(), root)); if (g) console.log(g); console.log(pretty(res, { color: false, limit: args.limit, footer: args.footer })); }
46
+ else console.log(pretty(res, { color: args.color, limit: args.limit, footer: args.footer }));
47
+ if (res.errors > 0) return 1;
48
+ if (Number.isFinite(args.maxWarnings) && res.warnings > args.maxWarnings) return 1;
49
+ return 0;
50
+ }
51
+
52
+ try { process.exitCode = main(process.argv.slice(2)); }
53
+ catch (e) { console.error('local-seo-lint: ' + e.message); process.exitCode = 2; }
package/package.json ADDED
@@ -0,0 +1,15 @@
1
+ {
2
+ "name": "local-seo-lint",
3
+ "version": "0.2.0",
4
+ "description": "Catch local SEO and AI-search mistakes before they go live: AI crawler access, NAP consistency, schema vs. page, canonicals, sitemap coverage, orphan pages and keyword overlap for static websites.",
5
+ "bin": { "local-seo-lint": "bin/local-seo-lint.js" },
6
+ "main": "src/index.js",
7
+ "files": ["bin", "src", "action.yml"],
8
+ "scripts": { "test": "node --test" },
9
+ "engines": { "node": ">=18" },
10
+ "keywords": ["seo", "local-seo", "lint", "schema", "json-ld", "nap", "static-site", "github-action", "sitemap", "canonical", "ai-search", "chatgpt", "perplexity", "robots-txt"],
11
+ "author": "Clicks Dynasty (https://www.clicksdynasty.com)",
12
+ "license": "MIT",
13
+ "repository": { "type": "git", "url": "https://github.com/clicksdynastyincali/local-seo-lint.git" },
14
+ "homepage": "https://github.com/clicksdynastyincali/local-seo-lint#readme"
15
+ }
package/src/checks.js ADDED
@@ -0,0 +1,257 @@
1
+ 'use strict';
2
+ const fs = require('fs');
3
+ const path = require('path');
4
+ const { phoneDigits, findPhones, normAddress, normName, businessEntities } = require('./nap');
5
+
6
+ const RULES = {
7
+ 'json-ld-syntax': { severity: 'error', about: 'JSON-LD blocks must be valid JSON, or search engines ignore them.' },
8
+ 'nap-schema-mismatch': { severity: 'error', about: 'Business schema on a page disagrees with your business name, address or phone.' },
9
+ 'nap-phone-mismatch': { severity: 'warn', about: 'A phone number on the page differs from your business phone.' },
10
+ 'nap-address-mismatch': { severity: 'warn', about: 'Your address is written differently on this page.' },
11
+ 'schema-not-visible': { severity: 'warn', about: 'Schema says something the visitor cannot see on the page.' },
12
+ 'canonical-missing': { severity: 'warn', about: 'Page has no canonical tag.' },
13
+ 'canonical-multiple': { severity: 'error', about: 'Page has more than one canonical tag.' },
14
+ 'canonical-relative': { severity: 'warn', about: 'Canonical URL should be absolute (https://...).' },
15
+ 'canonical-other-page': { severity: 'warn', about: 'Canonical points to a different page, so this page will not rank.' },
16
+ 'canonical-broken': { severity: 'error', about: 'Canonical points to a page that does not exist.' },
17
+ 'sitemap-missing': { severity: 'warn', about: 'No sitemap.xml found.' },
18
+ 'sitemap-page-missing': { severity: 'warn', about: 'Indexable page is not listed in sitemap.xml.' },
19
+ 'sitemap-dead-url': { severity: 'error', about: 'sitemap.xml lists a URL with no matching page.' },
20
+ 'sitemap-noindex': { severity: 'error', about: 'sitemap.xml lists a page marked noindex.' },
21
+ 'orphan-page': { severity: 'warn', about: 'No other page links here, so crawlers struggle to find it.' },
22
+ 'broken-link': { severity: 'error', about: 'Internal link points to a page that does not exist.' },
23
+ 'title-missing': { severity: 'error', about: 'Page has no <title>.' },
24
+ 'title-duplicate': { severity: 'warn', about: 'Several pages share the same title.' },
25
+ 'description-missing': { severity: 'warn', about: 'Page has no meta description.' },
26
+ 'description-duplicate': { severity: 'warn', about: 'Several pages share the same meta description.' },
27
+ 'h1-missing': { severity: 'warn', about: 'Page has no <h1>.' },
28
+ 'keyword-overlap': { severity: 'warn', about: 'Two pages target nearly the same search, so they compete with each other.' },
29
+ 'ai-crawler-blocked': { severity: 'error', about: 'robots.txt blocks an AI search crawler, so ChatGPT, Perplexity or Claude cannot read or recommend the site.' },
30
+ 'ai-entity-links-missing': { severity: 'warn', about: 'Business schema has no sameAs profile links, so AI tools cannot confirm which business this is.' },
31
+ };
32
+
33
+ // Crawlers that fetch pages for AI search answers (not model training). Blocking these hides a business from AI recommendations.
34
+ const AI_SEARCH_BOTS = ['OAI-SearchBot', 'ChatGPT-User', 'PerplexityBot', 'Perplexity-User', 'Claude-SearchBot', 'Claude-User', 'Bingbot', 'Applebot'];
35
+
36
+ // Minimal robots.txt reader: returns true if `agent` may not fetch "/".
37
+ function robotsBlocksRoot(txt, agent) {
38
+ const groups = []; let cur = null, lastWasAgent = false;
39
+ txt.split(/\r?\n/).forEach((raw) => {
40
+ const line = raw.replace(/#.*/, '').trim();
41
+ const m = line.match(/^([A-Za-z-]+)\s*:\s*(.*)$/);
42
+ if (!m) return;
43
+ const key = m[1].toLowerCase(), val = m[2].trim();
44
+ if (key === 'user-agent') { if (!lastWasAgent) { cur = { agents: [], rules: [] }; groups.push(cur); } cur.agents.push(val.toLowerCase()); lastWasAgent = true; return; }
45
+ lastWasAgent = false;
46
+ if (cur && (key === 'allow' || key === 'disallow')) cur.rules.push({ allow: key === 'allow', path: val });
47
+ });
48
+ const a = agent.toLowerCase();
49
+ let rules = groups.filter((g) => g.agents.includes(a)).flatMap((g) => g.rules);
50
+ if (!groups.some((g) => g.agents.includes(a))) rules = groups.filter((g) => g.agents.includes('*')).flatMap((g) => g.rules);
51
+ const hits = rules.filter((r) => r.path && '/'.startsWith(r.path.replace(/\*$/, '').replace(/\$$/, '')));
52
+ if (!hits.length) return false;
53
+ const best = hits.reduce((x, y) => (y.path.length > x.path.length || (y.path.length === x.path.length && y.allow) ? y : x));
54
+ return !best.allow;
55
+ }
56
+
57
+ const STOP = new Set('a an and or the of in on for to with by at from your our you we is are be as it this that best top free near me vs & | - – — : , ca usa us'.split(' '));
58
+
59
+ function resolver(site, siteOrigin) {
60
+ const hosts = new Set();
61
+ if (siteOrigin) { const h = new URL(siteOrigin).host.replace(/^www\./, ''); hosts.add(h); hosts.add('www.' + h); }
62
+ const lookup = (p) => {
63
+ let dp; try { dp = decodeURIComponent(p); } catch (e) { dp = p; }
64
+ return site.byPath.get(dp) || site.byPath.get(dp + (dp.endsWith('/') ? 'index.html' : '/index.html')) || site.byPath.get(dp + '.html') || site.byPath.get(dp.replace(/\/$/, '')) || null;
65
+ };
66
+ const resolve = (fromPath, href) => {
67
+ if (!href) return { skip: true };
68
+ const h = href.trim();
69
+ if (/^(mailto:|tel:|javascript:|data:|sms:|#)/i.test(h)) return { skip: true };
70
+ let u;
71
+ try { u = new URL(h, 'https://local.invalid' + fromPath); } catch (e) { return { skip: true }; }
72
+ const local = u.host === 'local.invalid' || hosts.has(u.host);
73
+ if (!local) return { external: true };
74
+ const page = lookup(u.pathname);
75
+ let asset = false;
76
+ if (!page) { let dp; try { dp = decodeURIComponent(u.pathname); } catch (e) { dp = u.pathname; } const f = path.join(site.root, dp); asset = dp !== '/' && fs.existsSync(f) && fs.statSync(f).isFile(); }
77
+ return { path: u.pathname, page, asset, absolute: u.host !== 'local.invalid' };
78
+ };
79
+ return { resolve, lookup };
80
+ }
81
+
82
+ function detectBusiness(site, cfg) {
83
+ if (cfg.business && (cfg.business.name || cfg.business.phone || cfg.business.address)) {
84
+ const b = cfg.business;
85
+ return { source: 'config', name: b.name || null, phone: b.phone || null, street: (b.address && b.address.street) || b.street || null, postalCode: (b.address && b.address.postalCode) || b.postalCode || null };
86
+ }
87
+ const home = site.byPath.get('/') || site.pages[0];
88
+ if (!home) return null;
89
+ const ents = businessEntities(home.jsonldParsed).filter((e) => e.street || e.telephone);
90
+ if (!ents.length) return null;
91
+ const e = ents[0];
92
+ return { source: 'schema on ' + home.file, name: e.name, phone: e.telephone, street: e.street, postalCode: e.postalCode };
93
+ }
94
+
95
+ const GENERIC_SEG = new Set(['about us', 'about', 'contact us', 'contact', 'home', 'homepage', 'blog', 'faq', 'faqs', 'services', 'privacy policy', 'terms of service', 'careers', 'press', 'login', 'client login']);
96
+
97
+ // Each title segment (split on | – — :) plus the H1 is a "target phrase".
98
+ function targetPhrases(p, brand) {
99
+ const brandWords = new Set(brand ? normName(brand).split(' ') : []);
100
+ const segs = (p.title || '').split(/\s[|–—:-]\s|\s\|\s?|\s[–—]\s?/).map((x) => x.trim()).filter(Boolean);
101
+ if (p.h1[0]) segs.push(p.h1[0]);
102
+ return segs
103
+ .map((seg, i) => ({ seg, first: i === 0, h1: i === segs.length - 1 && !!p.h1[0] }))
104
+ .filter(({ seg }) => !(brand && normName(seg) === normName(brand)) && !GENERIC_SEG.has(seg.toLowerCase()))
105
+ .map(({ seg, first, h1 }) => ({ seg, first, h1, key: normName(seg), t: new Set(seg.toLowerCase().replace(/[^a-z0-9 ]+/g, ' ').split(/\s+/).filter((w) => w && !STOP.has(w) && w.length > 1 && !brandWords.has(w))) }))
106
+ .filter((x) => x.t.size >= 2);
107
+ }
108
+
109
+ function run(site, cfg = {}) {
110
+ const findings = [];
111
+ const add = (rule, file, message) => findings.push({ rule, severity: RULES[rule].severity, file, message });
112
+ const pages = site.pages;
113
+ const home = site.byPath.get('/');
114
+ const siteOrigin = cfg.siteUrl || (home && home.canonicals[0] && /^https?:\/\//.test(home.canonicals[0]) ? new URL(home.canonicals[0]).origin : null);
115
+ const { resolve } = resolver(site, siteOrigin);
116
+ const business = detectBusiness(site, cfg);
117
+ const bizPhone = business && business.phone ? phoneDigits(business.phone) : null;
118
+ const bizStreet = business && business.street ? normAddress(business.street) : null;
119
+ const bizZip = business && business.postalCode ? String(business.postalCode).trim() : null;
120
+ const bizStreetNum = bizStreet ? (bizStreet.match(/^\d+/) || [null])[0] : null;
121
+ const indexable = pages.filter((p) => !p.noindex && !p.redirected && !/(^|\/)404\.html?$/i.test(p.file));
122
+
123
+ // Links, orphans
124
+ const inbound = new Map(pages.map((p) => [p.file, 0]));
125
+ pages.forEach((p) => {
126
+ const seen = new Set();
127
+ p.hrefs.forEach((href) => {
128
+ const r = resolve(p.urlPath, href);
129
+ if (r.skip || r.external) return;
130
+ if (!r.page && r.asset) return;
131
+ if (!r.page) { if (!seen.has('x' + r.path)) add('broken-link', p.file, `Link to "${href}" goes nowhere.`); seen.add('x' + r.path); return; }
132
+ if (r.page !== p && !seen.has(r.page.file)) { inbound.set(r.page.file, inbound.get(r.page.file) + 1); seen.add(r.page.file); }
133
+ });
134
+ });
135
+ indexable.forEach((p) => { if (p.urlPath !== '/' && inbound.get(p.file) === 0) add('orphan-page', p.file, 'No other page links to this page.'); });
136
+
137
+ // Canonicals
138
+ pages.forEach((p) => {
139
+ if (p.noindex || p.redirected) return;
140
+ if (!p.canonicals.length) return add('canonical-missing', p.file, 'Add <link rel="canonical" href="' + (siteOrigin || 'https://your-site') + p.urlPath + '">.');
141
+ if (p.canonicals.length > 1) add('canonical-multiple', p.file, `Found ${p.canonicals.length} canonical tags; keep one.`);
142
+ const c = p.canonicals[0];
143
+ if (!/^https?:\/\//i.test(c)) add('canonical-relative', p.file, `Canonical "${c}" is relative; use the full https:// URL.`);
144
+ const r = resolve(p.urlPath, c);
145
+ if (r.external) return;
146
+ if (!r.skip && !r.page) add('canonical-broken', p.file, `Canonical "${c}" points to a page that does not exist.`);
147
+ else if (r.page && r.page !== p) add('canonical-other-page', p.file, `Canonical points to ${r.page.file}, so this page is hidden from search. Intended?`);
148
+ });
149
+
150
+ // Sitemap
151
+ if (!site.sitemap) add('sitemap-missing', 'sitemap.xml', 'Create a sitemap.xml listing every page you want indexed.');
152
+ else {
153
+ const listed = new Set();
154
+ site.sitemap.urls.forEach((u) => {
155
+ const r = resolve('/', u);
156
+ if (r.external) return;
157
+ if (!r.page) add('sitemap-dead-url', site.sitemap.file, `Lists ${u}, but no such page exists.`);
158
+ else { listed.add(r.page.file); if (r.page.noindex) add('sitemap-noindex', site.sitemap.file, `Lists ${u}, but that page is marked noindex.`); }
159
+ });
160
+ indexable.forEach((p) => {
161
+ const canonicalElsewhere = p.canonicals[0] && (() => { const r = resolve(p.urlPath, p.canonicals[0]); return r.page && r.page !== p; })();
162
+ if (!listed.has(p.file) && !canonicalElsewhere) add('sitemap-page-missing', p.file, 'Not listed in sitemap.xml.');
163
+ });
164
+ }
165
+
166
+ // Titles, descriptions, h1
167
+ const dupes = (key, rule, label) => {
168
+ const m = new Map();
169
+ indexable.forEach((p) => { const v = (p[key] || '').trim().toLowerCase(); if (v) m.set(v, (m.get(v) || []).concat(p)); });
170
+ m.forEach((list) => { if (list.length > 1) list.forEach((p) => add(rule, p.file, `Same ${label} as ${list.filter((x) => x !== p).map((x) => x.file).slice(0, 3).join(', ')}.`)); });
171
+ };
172
+ indexable.forEach((p) => {
173
+ if (!p.title) add('title-missing', p.file, 'Add a <title> that says what the page offers and where.');
174
+ if (!p.description) add('description-missing', p.file, 'Add a meta description (about 150 characters).');
175
+ if (!p.h1.length) add('h1-missing', p.file, 'Add one <h1> heading.');
176
+ });
177
+ dupes('title', 'title-duplicate', 'title');
178
+ dupes('description', 'description-duplicate', 'meta description');
179
+
180
+ // Keyword overlap (cannibalization): compare target phrases with IDF-weighted overlap,
181
+ // so words used everywhere (city names, "services") count less than distinctive ones.
182
+ const brand = business && business.name;
183
+ const phr = indexable.map((p) => ({ p, list: targetPhrases(p, brand) })).filter((x) => x.list.length);
184
+ const df = new Map();
185
+ phr.forEach(({ list }) => { const u = new Set(); list.forEach((x) => x.t.forEach((w) => u.add(w))); u.forEach((w) => df.set(w, (df.get(w) || 0) + 1)); });
186
+ const N = phr.length;
187
+ // Boilerplate title parts (the same suffix on 3+ pages, never used as the lead phrase) are not targets.
188
+ const segPages = new Map(), leadSegs = new Set();
189
+ phr.forEach(({ list }) => list.forEach((x) => { if (x.h1) return; segPages.set(x.key, (segPages.get(x.key) || 0) + 1); if (x.first) leadSegs.add(x.key); }));
190
+ phr.forEach((x) => { x.list = x.list.filter((A) => A.h1 || !(segPages.get(A.key) >= 3 && !leadSegs.has(A.key))); });
191
+ const wt = (w) => Math.log((N + 1) / ((df.get(w) || 0) + 1)) + 0.15;
192
+ const threshold = cfg.overlap && cfg.overlap.threshold ? cfg.overlap.threshold : 0.8;
193
+ for (let i = 0; i < phr.length; i++) for (let j = i + 1; j < phr.length; j++) {
194
+ let best = null;
195
+ phr[i].list.forEach((A) => phr[j].list.forEach((B) => {
196
+ let inter = 0, uni = 0, shared = [];
197
+ new Set([...A.t, ...B.t]).forEach((w) => { const x = wt(w); uni += x; if (A.t.has(w) && B.t.has(w)) { inter += x; shared.push(w); } });
198
+ const score = uni ? inter / uni : 0;
199
+ if (shared.length >= 3 && (!best || score > best.score)) best = { score, a: A.seg, b: B.seg };
200
+ }));
201
+ if (best && best.score >= threshold) add('keyword-overlap', phr[i].p.file, `"${best.a}" competes with ${phr[j].p.file} ("${best.b}"), ${Math.round(best.score * 100)}% overlap. Pick one page for this search.`);
202
+ }
203
+
204
+ // JSON-LD + NAP
205
+ pages.forEach((p) => {
206
+ p.jsonldErrors.forEach((e) => add('json-ld-syntax', p.file, `JSON-LD block #${e.index + 1} is invalid: ${e.message}`));
207
+ const digitsText = p.text.replace(/\D/g, '');
208
+ const ents = businessEntities(p.jsonldParsed);
209
+ ents.forEach((e) => {
210
+ const same = business && business.name && normName(e.name) === normName(business.name);
211
+ if (same) {
212
+ if (bizPhone && e.telephone && phoneDigits(e.telephone) !== bizPhone) add('nap-schema-mismatch', p.file, `Schema phone ${e.telephone} differs from your business phone ${business.phone}.`);
213
+ if (bizStreet && e.street && normAddress(e.street) !== bizStreet) add('nap-schema-mismatch', p.file, `Schema street "${e.street}" differs from "${business.street}".`);
214
+ if (bizZip && e.postalCode && String(e.postalCode).trim() !== bizZip) add('nap-schema-mismatch', p.file, `Schema ZIP ${e.postalCode} differs from ${bizZip}.`);
215
+ }
216
+ if (e.name && !normName(p.text).includes(normName(e.name))) add('schema-not-visible', p.file, `Schema names "${e.name}", but that name is not visible on the page.`);
217
+ if (e.telephone && phoneDigits(e.telephone).length >= 10 && !digitsText.includes(phoneDigits(e.telephone))) add('schema-not-visible', p.file, `Schema phone ${e.telephone} is not visible on the page.`);
218
+ });
219
+ if (bizPhone) {
220
+ const seen = new Set();
221
+ findPhones(p.text).forEach((ph) => { if (ph.digits !== bizPhone && !seen.has(ph.digits)) { seen.add(ph.digits); add('nap-phone-mismatch', p.file, `Shows ${ph.raw}; your business phone is ${business.phone}.`); } });
222
+ }
223
+ if (bizStreet && bizStreetNum) {
224
+ const t = p.text;
225
+ const re = new RegExp('\\b' + bizStreetNum + '\\s+[A-Za-z][^\\n]{0,40}', 'g');
226
+ let m; const seen = new Set();
227
+ while ((m = re.exec(t))) {
228
+ const cand = normAddress(m[0]);
229
+ const firstWord = bizStreet.split(' ')[1];
230
+ if (!firstWord || !cand.split(' ').slice(1, 3).includes(firstWord)) continue; // not our street
231
+ if (!cand.startsWith(bizStreet) && !seen.has(cand)) { seen.add(cand); add('nap-address-mismatch', p.file, `Address written as "${m[0].trim().slice(0, 45)}"; standard form is "${business.street}".`); }
232
+ }
233
+ }
234
+ });
235
+
236
+ // AI search readiness
237
+ const robotsFile = require('path').join(site.root, 'robots.txt');
238
+ if (fs.existsSync(robotsFile)) {
239
+ const txt = fs.readFileSync(robotsFile, 'utf8');
240
+ const blocked = AI_SEARCH_BOTS.filter((b) => robotsBlocksRoot(txt, b));
241
+ if (blocked.length) add('ai-crawler-blocked', 'robots.txt', `Blocks ${blocked.join(', ')}. Add "User-agent: ${blocked[0]}" with "Allow: /" if you want to appear in AI answers.`);
242
+ }
243
+ if (home && business && business.source !== 'config') {
244
+ const ents = businessEntities(home.jsonldParsed).filter((e) => normName(e.name) === normName(business.name));
245
+ if (ents.length && !ents.some((e) => e.sameAs.length)) add('ai-entity-links-missing', home.file, 'Add "sameAs" links to your Google Business Profile, Facebook, LinkedIn and Yelp pages in the business schema.');
246
+ }
247
+
248
+ const off = cfg.rules || {};
249
+ return {
250
+ business, siteOrigin,
251
+ findings: findings
252
+ .filter((f) => off[f.rule] !== 'off')
253
+ .map((f) => (off[f.rule] === 'warn' || off[f.rule] === 'error' ? Object.assign(f, { severity: off[f.rule] }) : f)),
254
+ };
255
+ }
256
+
257
+ module.exports = { run, RULES, robotsBlocksRoot, AI_SEARCH_BOTS };
package/src/html.js ADDED
@@ -0,0 +1,85 @@
1
+ 'use strict';
2
+ // Minimal, dependency-free HTML helpers. Good enough for linting static sites;
3
+ // not a full HTML5 parser.
4
+
5
+ const ENT = { amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", nbsp: ' ', ndash: '-', mdash: '-', middot: '·', rarr: '→', larr: '←', copy: '©' };
6
+
7
+ function decode(s) {
8
+ return String(s)
9
+ .replace(/&#x([0-9a-f]+);/gi, (_, h) => String.fromCodePoint(parseInt(h, 16)))
10
+ .replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(+d))
11
+ .replace(/&([a-z]+);/gi, (m, n) => (ENT[n.toLowerCase()] !== undefined ? ENT[n.toLowerCase()] : m));
12
+ }
13
+
14
+ function attrs(tag) {
15
+ const out = {};
16
+ const re = /([a-zA-Z_:][-a-zA-Z0-9_:.]*)\s*(?:=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+)))?/g;
17
+ const body = tag.replace(/^<\s*[a-zA-Z0-9-]+/, '').replace(/\/?>$/, '');
18
+ let m;
19
+ while ((m = re.exec(body))) {
20
+ const v = m[2] !== undefined ? m[2] : m[3] !== undefined ? m[3] : m[4] !== undefined ? m[4] : '';
21
+ out[m[1].toLowerCase()] = decode(v);
22
+ }
23
+ return out;
24
+ }
25
+
26
+ function tags(html, name) {
27
+ const re = new RegExp('<' + name + '\\b[^>]*>', 'gi');
28
+ return (html.match(re) || []).map(attrs);
29
+ }
30
+
31
+ function stripComments(html) {
32
+ return html.replace(/<!--[\s\S]*?-->/g, '');
33
+ }
34
+
35
+ function inner(html, name) {
36
+ const re = new RegExp('<' + name + '\\b[^>]*>([\\s\\S]*?)</' + name + '>', 'gi');
37
+ const out = [];
38
+ let m;
39
+ while ((m = re.exec(html))) out.push(m[1]);
40
+ return out;
41
+ }
42
+
43
+ function text(fragment) {
44
+ return decode(String(fragment).replace(/<[^>]+>/g, ' ')).replace(/\s+/g, ' ').trim();
45
+ }
46
+
47
+ function visibleText(html) {
48
+ const body = (html.match(/<body\b[^>]*>([\s\S]*)<\/body>/i) || [null, html])[1];
49
+ return text(
50
+ stripComments(body)
51
+ .replace(/<(script|style|noscript|template|svg)\b[\s\S]*?<\/\1>/gi, ' ')
52
+ );
53
+ }
54
+
55
+ function parse(html) {
56
+ html = stripComments(html);
57
+ const metas = tags(html, 'meta');
58
+ const meta = (key) => {
59
+ const t = metas.find((a) => (a.name || a.property || '').toLowerCase() === key);
60
+ return t ? (t.content || '').trim() : null;
61
+ };
62
+ const links = tags(html, 'link');
63
+ const jsonld = [];
64
+ const scriptRe = /<script\b([^>]*)>([\s\S]*?)<\/script>/gi;
65
+ let m;
66
+ while ((m = scriptRe.exec(html))) {
67
+ const a = attrs('<script ' + m[1] + '>');
68
+ if ((a.type || '').toLowerCase() === 'application/ld+json') jsonld.push(m[2].trim());
69
+ }
70
+ const titles = inner(html, 'title').map(text);
71
+ const h1 = inner(html, 'h1').map(text);
72
+ return {
73
+ title: titles[0] || null,
74
+ titleCount: titles.length,
75
+ description: meta('description'),
76
+ robots: (meta('robots') || '').toLowerCase(),
77
+ canonicals: links.filter((l) => (l.rel || '').toLowerCase().split(/\s+/).includes('canonical')).map((l) => l.href || ''),
78
+ hrefs: tags(html, 'a').map((a) => a.href).filter((h) => h !== undefined),
79
+ h1,
80
+ jsonld,
81
+ text: visibleText(html),
82
+ };
83
+ }
84
+
85
+ module.exports = { parse, decode, text };
package/src/index.js ADDED
@@ -0,0 +1,23 @@
1
+ 'use strict';
2
+ const fs = require('fs');
3
+ const path = require('path');
4
+ const { loadSite } = require('./site');
5
+ const { run, RULES } = require('./checks');
6
+
7
+ function loadConfig(root, file) {
8
+ const candidates = file ? [file] : ['.localseorc.json', 'localseo.config.json'].map((f) => path.join(root, f));
9
+ for (const f of candidates) if (fs.existsSync(f)) return Object.assign({ _file: f }, JSON.parse(fs.readFileSync(f, 'utf8')));
10
+ return {};
11
+ }
12
+
13
+ function lint(root, options = {}) {
14
+ const cfg = Object.assign({}, loadConfig(root, options.config), options.overrides || {});
15
+ const ignore = [].concat(cfg.ignore || [], options.ignore || []);
16
+ const site = loadSite(root, { ignore, sitemap: cfg.sitemap });
17
+ const res = run(site, cfg);
18
+ const errors = res.findings.filter((f) => f.severity === 'error').length;
19
+ const warnings = res.findings.filter((f) => f.severity === 'warn').length;
20
+ return Object.assign(res, { pages: site.pages.length, errors, warnings, config: cfg._file || null });
21
+ }
22
+
23
+ module.exports = { lint, RULES };
package/src/nap.js ADDED
@@ -0,0 +1,66 @@
1
+ 'use strict';
2
+ // Name / Address / Phone helpers.
3
+
4
+ function phoneDigits(s) {
5
+ let d = String(s || '').replace(/\D/g, '');
6
+ if (d.length === 11 && d[0] === '1') d = d.slice(1);
7
+ return d;
8
+ }
9
+
10
+ // US/CA-style numbers: (661) 555-0123, 661-555-0123, +1 661 555 0123, 661.555.0123
11
+ const PHONE_RE = /(?:\+?1[\s.-]?)?\(?\b[2-9]\d{2}\)?[\s.-]?\d{3}[\s.-]\d{4}\b/g;
12
+
13
+ function findPhones(text) {
14
+ return (String(text).match(PHONE_RE) || []).map((raw) => ({ raw: raw.trim(), digits: phoneDigits(raw) }));
15
+ }
16
+
17
+ const ABBR = [
18
+ [/\bstreet\b/g, 'st'], [/\bavenue\b/g, 'ave'], [/\bav\b/g, 'ave'], [/\blane\b/g, 'ln'], [/\broad\b/g, 'rd'],
19
+ [/\bboulevard\b/g, 'blvd'], [/\bdrive\b/g, 'dr'], [/\bhighway\b/g, 'hwy'], [/\bparkway\b/g, 'pkwy'],
20
+ [/\bcourt\b/g, 'ct'], [/\bplace\b/g, 'pl'], [/\bcircle\b/g, 'cir'], [/\bsquare\b/g, 'sq'],
21
+ [/\bnorth\b/g, 'n'], [/\bsouth\b/g, 's'], [/\beast\b/g, 'e'], [/\bwest\b/g, 'w'],
22
+ [/\bsuite\b/g, 'ste'], [/\bunit\b/g, 'ste'], [/\bapartment\b/g, 'ste'], [/\bapt\b/g, 'ste'], [/#\s*/g, 'ste '], [/\bno\.?\s*(?=\d)/g, 'ste ']
23
+ ];
24
+
25
+ function normAddress(s) {
26
+ let t = String(s || '').toLowerCase();
27
+ ABBR.forEach(([re, rep]) => { t = t.replace(re, rep); });
28
+ return t.replace(/[^a-z0-9 ]+/g, ' ').replace(/\s+/g, ' ').trim();
29
+ }
30
+
31
+ function normName(s) {
32
+ return String(s || '').toLowerCase().replace(/&/g, 'and').replace(/[^a-z0-9]+/g, ' ').replace(/\b(llc|inc|co|ltd|corp|the)\b/g, '').replace(/\s+/g, ' ').trim();
33
+ }
34
+
35
+ // Pull business-like entities out of parsed JSON-LD (handles arrays and @graph).
36
+ function businessEntities(jsonldValues) {
37
+ const out = [];
38
+ const walk = (node) => {
39
+ if (!node || typeof node !== 'object') return;
40
+ if (Array.isArray(node)) return node.forEach(walk);
41
+ if (node['@graph']) walk(node['@graph']);
42
+ const addr = node.address;
43
+ const hasAddr = addr && (typeof addr === 'string' || typeof addr === 'object');
44
+ if (node.name && (hasAddr || node.telephone)) {
45
+ const a = Array.isArray(addr) ? addr[0] : addr;
46
+ out.push({
47
+ type: [].concat(node['@type'] || []).join(','),
48
+ name: node.name,
49
+ telephone: node.telephone || null,
50
+ street: a && typeof a === 'object' ? a.streetAddress || null : typeof a === 'string' ? a : null,
51
+ locality: a && typeof a === 'object' ? a.addressLocality || null : null,
52
+ region: a && typeof a === 'object' ? a.addressRegion || null : null,
53
+ postalCode: a && typeof a === 'object' ? a.postalCode || null : null,
54
+ openingHours: node.openingHoursSpecification || node.openingHours || null,
55
+ sameAs: [].concat(node.sameAs || []).filter((u) => typeof u === 'string' && /^https?:\/\//i.test(u)),
56
+ });
57
+ }
58
+ Object.keys(node).forEach((k) => { if (k !== '@graph' && typeof node[k] === 'object') walk(node[k]); });
59
+ };
60
+ jsonldValues.forEach(walk);
61
+ // de-duplicate identical entities found through nesting
62
+ const seen = new Set();
63
+ return out.filter((e) => { const k = JSON.stringify(e); if (seen.has(k)) return false; seen.add(k); return true; });
64
+ }
65
+
66
+ module.exports = { phoneDigits, findPhones, normAddress, normName, businessEntities };
package/src/report.js ADDED
@@ -0,0 +1,44 @@
1
+ 'use strict';
2
+ const { RULES } = require('./checks');
3
+
4
+ function color(enabled) {
5
+ const c = (n) => (s) => (enabled ? `\x1b[${n}m${s}\x1b[0m` : s);
6
+ return { red: c(31), yellow: c(33), green: c(32), dim: c(2), bold: c(1), cyan: c(36) };
7
+ }
8
+
9
+ function pretty(res, opts = {}) {
10
+ const k = color(opts.color !== false);
11
+ const out = [];
12
+ out.push(k.bold('local-seo-lint') + k.dim(` ${res.pages} pages scanned`));
13
+ if (res.business) out.push(k.dim(`Business: ${res.business.name || '?'} | ${res.business.street || 'no street'} | ${res.business.phone || 'no phone'} (from ${res.business.source})`));
14
+ else out.push(k.yellow('No business details found. Add LocalBusiness schema to your homepage or a "business" block to .localseorc.json to enable NAP checks.'));
15
+ out.push('');
16
+ const byRule = new Map();
17
+ res.findings.forEach((f) => byRule.set(f.rule, (byRule.get(f.rule) || []).concat(f)));
18
+ const order = [...byRule.keys()].sort((a, b) => (RULES[a].severity === RULES[b].severity ? a.localeCompare(b) : RULES[a].severity === 'error' ? -1 : 1));
19
+ const limit = opts.limit || 10;
20
+ order.forEach((rule) => {
21
+ const list = byRule.get(rule);
22
+ const sev = list[0].severity === 'error' ? k.red('error') : k.yellow('warn ');
23
+ out.push(`${sev} ${k.bold(rule)} ${k.dim('(' + list.length + ')')} ${k.dim(RULES[rule].about)}`);
24
+ list.slice(0, limit).forEach((f) => out.push(` ${k.cyan(f.file)} ${f.message}`));
25
+ if (list.length > limit) out.push(k.dim(` ...and ${list.length - limit} more (use --limit 0 to show all)`));
26
+ out.push('');
27
+ });
28
+ const summary = `${res.errors} error${res.errors === 1 ? '' : 's'}, ${res.warnings} warning${res.warnings === 1 ? '' : 's'}`;
29
+ out.push(res.errors ? k.red(summary) : res.warnings ? k.yellow(summary) : k.green('No problems found. ' + summary));
30
+ if (opts.footer !== false && (res.errors || res.warnings)) {
31
+ out.push('');
32
+ out.push(k.dim('Need these fixed? Clicks Dynasty fixes local SEO and AI-visibility issues, including white-label for agencies:'));
33
+ out.push(k.dim('https://www.clicksdynasty.com/book.html?service=ai-check (hide this line with --no-footer)'));
34
+ }
35
+ return out.join('\n');
36
+ }
37
+
38
+ function github(res, prefix = '') {
39
+ const pre = prefix && prefix !== '.' ? prefix.replace(/\\/g, '/').replace(/\/$/, '') + '/' : '';
40
+ const esc = (s) => String(s).replace(/%/g, '%25').replace(/\r/g, '%0D').replace(/\n/g, '%0A');
41
+ return res.findings.map((f) => `::${f.severity === 'error' ? 'error' : 'warning'} file=${esc(pre + f.file)},title=${esc(f.rule)}::${esc(f.message)}`).join('\n');
42
+ }
43
+
44
+ module.exports = { pretty, github };
package/src/site.js ADDED
@@ -0,0 +1,80 @@
1
+ 'use strict';
2
+ const fs = require('fs');
3
+ const path = require('path');
4
+ const { parse } = require('./html');
5
+
6
+ const DEFAULT_IGNORE = ['node_modules/**', '.git/**', '**/node_modules/**'];
7
+
8
+ function globToRe(g) {
9
+ g = String(g).replace(/^\.\//, '').replace(/\/$/, '');
10
+ let s = '';
11
+ for (let i = 0; i < g.length; i++) {
12
+ const c = g[i];
13
+ if (c === '*' && g[i + 1] === '*') { if (g[i + 2] === '/') { s += '(?:.*/)?'; i += 2; } else { s += '.*'; i += 1; } }
14
+ else if (c === '*') s += '[^/]*';
15
+ else if (c === '?') s += '[^/]';
16
+ else s += c.replace(/[.+^${}()|[\]\\]/g, '\\$&');
17
+ }
18
+ return new RegExp('^' + s + '(?:/.*)?$');
19
+ }
20
+
21
+ function walk(dir, root, ignoreRes, out) {
22
+ for (const ent of fs.readdirSync(dir, { withFileTypes: true })) {
23
+ const abs = path.join(dir, ent.name);
24
+ const rel = path.relative(root, abs).split(path.sep).join('/');
25
+ if (ignoreRes.some((re) => re.test(rel))) continue;
26
+ if (ent.isDirectory()) walk(abs, root, ignoreRes, out);
27
+ else if (/\.html?$/i.test(ent.name)) out.push(rel);
28
+ }
29
+ return out;
30
+ }
31
+
32
+ function urlPathFor(rel) {
33
+ if (/(^|\/)index\.html?$/i.test(rel)) return '/' + rel.replace(/index\.html?$/i, '');
34
+ return '/' + rel;
35
+ }
36
+
37
+ function loadSite(root, opts = {}) {
38
+ const ignore = DEFAULT_IGNORE.concat(opts.ignore || []);
39
+ const files = walk(root, root, ignore.map(globToRe), []).sort();
40
+ const pages = files.map((rel) => {
41
+ const html = fs.readFileSync(path.join(root, rel), 'utf8');
42
+ const p = parse(html);
43
+ p.file = rel;
44
+ p.urlPath = urlPathFor(rel);
45
+ p.noindex = /\bnoindex\b/.test(p.robots);
46
+ p.jsonldParsed = [];
47
+ p.jsonldErrors = [];
48
+ p.jsonld.forEach((raw, i) => {
49
+ try { p.jsonldParsed.push(JSON.parse(raw)); } catch (e) { p.jsonldErrors.push({ index: i, message: e.message }); }
50
+ });
51
+ return p;
52
+ });
53
+ const byPath = new Map();
54
+ pages.forEach((p) => {
55
+ byPath.set(p.urlPath, p);
56
+ if (p.urlPath.endsWith('/') && p.urlPath !== '/') byPath.set(p.urlPath.slice(0, -1), p);
57
+ });
58
+ // Pages that are 301-redirected elsewhere (Apache .htaccess or Netlify/Cloudflare _redirects) are skipped.
59
+ const redirected = new Set();
60
+ const addRedirect = (from) => { const f = String(from).replace(/^\^?\/?/, '/').replace(/\$$/, '').replace(/\\\./g, '.'); if (!/[*()[\]|+?]/.test(f)) redirected.add(f.toLowerCase()); };
61
+ const ht = path.join(root, '.htaccess');
62
+ if (fs.existsSync(ht)) fs.readFileSync(ht, 'utf8').split(/\r?\n/).forEach((line) => {
63
+ const l = line.trim(); let m;
64
+ if ((m = l.match(/^Redirect(?:Permanent|\s+(?:301|permanent))?\s+(\S+)\s+\S+/i))) addRedirect(m[1]);
65
+ else if ((m = l.match(/^RewriteRule\s+(\S+)\s+\S+\s+\[[^\]]*R=30[18][^\]]*\]/i))) addRedirect(m[1]);
66
+ });
67
+ const rd = path.join(root, '_redirects');
68
+ if (fs.existsSync(rd)) fs.readFileSync(rd, 'utf8').split(/\r?\n/).forEach((line) => { const m = line.trim().match(/^(\/\S+)\s+\S+\s*(30[18])?/); if (m && !line.trim().startsWith('#')) addRedirect(m[1]); });
69
+ pages.forEach((p) => { p.redirected = redirected.has(p.urlPath.toLowerCase()); });
70
+
71
+ let sitemap = null;
72
+ const smPath = path.join(root, opts.sitemap || 'sitemap.xml');
73
+ if (fs.existsSync(smPath)) {
74
+ const xml = fs.readFileSync(smPath, 'utf8');
75
+ sitemap = { file: path.relative(root, smPath).split(path.sep).join('/'), urls: (xml.match(/<loc>\s*([^<]+?)\s*<\/loc>/gi) || []).map((l) => l.replace(/<\/?loc>/gi, '').trim()) };
76
+ }
77
+ return { root, pages, byPath, sitemap, redirected };
78
+ }
79
+
80
+ module.exports = { loadSite, globToRe };