サイトの全ページのtitle・description・見出し・構造化データ・canonicalを、ビルド後のHTMLから一括で点検するスクリプト。重複、noindexの混入、内部リンク切れまで、このブログで流した結果つき
ビルド後の
このdistの
この
公開中のサイトを巡回せず、ビルド後のHTMLを読む
点検の
公開中の
ただし、
点検する項目と、エラーと注意の分け方
エラーは
| 項目 | 見つけ方 | 区分 | 根拠 |
|---|---|---|---|
| titleが |
<title>を |
エラー | Googleは、 |
| titleが |
文字数が |
注意 | 長さの |
| descriptionが |
titleと |
エラー | 同じ |
| descriptionが |
50字未満・120字超 | 注意 | 上限は |
| h1が無い | <h1>が0個 |
エラー | ページの |
| h1が複数 | <h1>が2個以上 |
注意 | 自社の |
| JSON-LDが |
JSON.parseで@typeや@contextが無い |
エラー | 読めない |
| canonicalが |
<link rel="canonical">を |
エラー | Googleは |
| canonicalが |
canonicalの |
注意 | 意図して |
| canonicalの |
指した |
エラー | 存在しない・載せない |
| noindexなのに |
metaの_headersの |
エラー | サイトマップには |
| 内部リンク切れ・画像が |
hrefとsrcをdistの |
エラー | 読者にも |
| ページ内リンクの |
#見出しの |
エラー | 目次や |
| リンク先が |
末尾の/が無い、_redirectsに当たる |
注意 | 届くが、 |
noindexは、_headersファイルで_headersを
構造化データの
点検のスクリプト
スクリプトは、node seo-audit.mjs <distのフォルダ> <サイトのURL>で
// seo-audit.mjs — ビルド後の dist のHTMLを全部読んで、SEOの基本項目を一括で点検する
// 使い方: node seo-audit.mjs <distのフォルダ> <サイトのURL>
// 例: node seo-audit.mjs ./dist https://www.example.com
// 依存: cheerio(npm i -D cheerio)
import fs from 'node:fs';
import path from 'node:path';
import * as cheerio from 'cheerio';
const DIST = path.resolve(process.argv[2] ?? 'dist');
const SITE = new URL(process.argv[3] ?? 'https://www.example.com');
// 同じサイトとして扱うホスト(www あり・なし)
const SAME_HOSTS = new Set([SITE.host, SITE.host.replace(/^www\./, ''), `www.${SITE.host.replace(/^www\./, '')}`]);
// しきい値は自社の目安(Googleが決めた上限ではない)
const RULES = { titleMax: 45, descMin: 50, descMax: 120 };
// 点検しないファイル(404ページなど)
const SKIP = new Set(['404.html']);
const issues = []; // { level: 'error' | 'warn', kind, page, detail }
const add = (level, kind, page, detail = '') => issues.push({ level, kind, page, detail });
const len = (s) => [...s].length; // 日本語も1文字ずつ数える
// ---- 1. dist のHTMLを集め、ファイルの場所からURLのパスを決める ----
function walk(dir) {
return fs.readdirSync(dir, { withFileTypes: true }).flatMap((e) => {
const p = path.join(dir, e.name);
return e.isDirectory() ? walk(p) : p.endsWith('.html') ? [p] : [];
});
}
const fileToPath = (file) => {
const rel = path.relative(DIST, file).split(path.sep).join('/');
return '/' + rel.replace(/(^|\/)index\.html$/, '$1').replace(/\.html$/, '');
};
// ---- 2. _headers の X-Robots-Tag と _redirects を読む(Cloudflare Pages / Netlify の書式) ----
const toMatcher = (pattern) => {
const re = pattern.split('*').map((s) => s.replace(/[.+?^${}()|[\]\\]/g, '\\$&')).join('.*');
return new RegExp(`^${re}$`);
};
const headerRules = [];
if (fs.existsSync(path.join(DIST, '_headers'))) {
let current = null;
for (const line of fs.readFileSync(path.join(DIST, '_headers'), 'utf8').split('\n')) {
if (!line.trim() || line.trim().startsWith('#')) continue;
if (!/^\s/.test(line)) current = toMatcher(line.trim());
else if (current && /^\s*x-robots-tag\s*:/i.test(line)) headerRules.push({ re: current, value: line.split(':').slice(1).join(':').trim() });
}
}
const redirectRules = [];
if (fs.existsSync(path.join(DIST, '_redirects'))) {
for (const line of fs.readFileSync(path.join(DIST, '_redirects'), 'utf8').split('\n')) {
const [from, to, status] = line.trim().split(/\s+/);
if (!from || from.startsWith('#') || !to) continue;
redirectRules.push({ re: toMatcher(from.replace(/:\w+/g, '*')), to, status: status ?? '301' });
}
}
// ---- 3. サイトマップのURLを読む ----
const sitemapPaths = new Set();
for (const f of fs.readdirSync(DIST).filter((f) => /^sitemap.*\.xml$/.test(f))) {
const xml = fs.readFileSync(path.join(DIST, f), 'utf8');
if (xml.includes('<sitemapindex')) continue; // 索引ファイルは中身のサイトマップを直接読む
for (const m of xml.matchAll(/<loc>([^<]+)<\/loc>/g)) sitemapPaths.add(decodeURI(new URL(m[1]).pathname));
}
// ---- 4. ページごとに読む ----
const skipped = [];
const pages = new Map(); // パス → { title, desc, canonical, noindex, ids, links }
for (const file of walk(DIST)) {
const rel = path.relative(DIST, file);
if (SKIP.has(rel)) continue;
const urlPath = fileToPath(file);
const html = fs.readFileSync(file, 'utf8');
// 拡張子が .html でも中身がHTMLでないもの(RSSを index.html で書き出した場合など)は飛ばす
if (!/<html[\s>]/i.test(html.slice(0, 2000))) { skipped.push(urlPath); continue; }
const $ = cheerio.load(html);
const titles = $('head title');
const title = titles.first().text().trim();
const descs = $('head meta[name="description"]');
const desc = (descs.attr('content') ?? '').trim();
const canons = $('head link[rel="canonical"]');
const canonical = canons.attr('href') ?? '';
const metaRobots = $('meta[name="robots"], meta[name="googlebot"]').map((_, el) => $(el).attr('content')).get().join(',');
const headerRobots = headerRules.filter((r) => r.re.test(urlPath)).map((r) => r.value).join(',');
const noindex = /noindex/i.test(`${metaRobots},${headerRobots}`);
// title
if (titles.length === 0 || !title) add('error', 'titleが無い', urlPath);
if (titles.length > 1) add('error', 'titleが複数', urlPath, `${titles.length}個`);
if (len(title) > RULES.titleMax) add('warn', 'titleが長い', urlPath, `${len(title)}字: ${title}`);
// description
if (descs.length === 0 || !desc) add(noindex ? 'warn' : 'error', 'descriptionが無い', urlPath);
if (descs.length > 1) add('error', 'descriptionが複数', urlPath, `${descs.length}個`);
if (desc && len(desc) > RULES.descMax) add('warn', 'descriptionが長い', urlPath, `${len(desc)}字`);
if (desc && len(desc) < RULES.descMin) add('warn', 'descriptionが短い', urlPath, `${len(desc)}字: ${desc}`);
// h1
const h1 = $('h1').length;
if (h1 === 0) add('error', 'h1が無い', urlPath);
if (h1 > 1) add('warn', 'h1が複数', urlPath, `${h1}個`);
// canonical
if (canons.length === 0) add(noindex ? 'warn' : 'error', 'canonicalが無い', urlPath);
if (canons.length > 1) add('error', 'canonicalが複数', urlPath, `${canons.length}個`);
if (canonical) {
let c;
try { c = new URL(canonical); } catch { add('error', 'canonicalが絶対URLでない', urlPath, canonical); }
if (c) {
if (c.host !== SITE.host) add('error', 'canonicalのホストが違う', urlPath, canonical);
else if (decodeURI(c.pathname) !== urlPath) add('warn', 'canonicalが自分以外を指す', urlPath, canonical);
}
const og = $('meta[property="og:url"]').attr('content');
if (og && og !== canonical) add('warn', 'og:urlとcanonicalが違う', urlPath, `${og} / ${canonical}`);
}
// noindex とサイトマップの食い違い
if (noindex && sitemapPaths.has(urlPath)) add('error', 'noindexなのにサイトマップに載っている', urlPath, metaRobots || headerRobots);
if (!noindex && sitemapPaths.size && !sitemapPaths.has(urlPath)) add('warn', 'インデックス可なのにサイトマップに無い', urlPath);
// JSON-LD
$('script[type="application/ld+json"]').each((i, el) => {
let data;
try { data = JSON.parse($(el).text()); } catch (e) { return add('error', 'JSON-LDが読めない', urlPath, `${i + 1}個目: ${e.message}`); }
for (const node of [data].flat().flatMap((d) => d?.['@graph'] ?? [d])) {
if (!node?.['@type']) add('error', 'JSON-LDに@typeが無い', urlPath, `${i + 1}個目`);
if (!data['@context'] && !node['@context']) add('error', 'JSON-LDに@contextが無い', urlPath, `${i + 1}個目`);
// 記事・ページの型は、示すURLが canonical と同じか
if (/Article|BlogPosting|WebPage/.test([node['@type']].flat().join()) && canonical) {
const main = typeof node.mainEntityOfPage === 'string' ? node.mainEntityOfPage : node.mainEntityOfPage?.['@id'];
const u = main ?? node.url;
if (u && u !== canonical) add('warn', 'JSON-LDのURLとcanonicalが違う', urlPath, `${node['@type']}: ${u}`);
}
}
});
// 内部リンク・画像(あとでまとめて確かめる)
const links = [];
$('a[href]').each((_, el) => links.push({ kind: 'a', href: $(el).attr('href') }));
$('img[src]').each((_, el) => links.push({ kind: 'img', href: $(el).attr('src') }));
const ids = new Set($('[id]').map((_, el) => $(el).attr('id')).get());
$('a[name]').each((_, el) => ids.add($(el).attr('name')));
pages.set(urlPath, { title, desc, canonical, noindex, ids, links, file });
}
// ---- 5. 重複(インデックス可のページどうし) ----
const dupes = (key, kind) => {
const groups = new Map();
for (const [p, v] of pages) if (!v.noindex && v[key]) groups.set(v[key], [...(groups.get(v[key]) ?? []), p]);
for (const [value, ps] of groups) if (ps.length > 1) add('error', kind, ps.join(' , '), value);
};
dupes('title', 'titleが重複');
dupes('desc', 'descriptionが重複');
// canonical の先がインデックス可のページか
for (const [p, v] of pages) {
if (!v.canonical) continue;
const target = decodeURI(new URL(v.canonical, SITE).pathname);
if (target === p) continue;
const t = pages.get(target);
if (!t) add('error', 'canonicalの先が存在しない', p, v.canonical);
else if (t.noindex) add('error', 'canonicalの先がnoindex', p, v.canonical);
}
// ---- 6. 内部リンク切れ ----
const fileExists = (p) => fs.existsSync(path.join(DIST, p)) && fs.statSync(path.join(DIST, p)).isFile();
function resolvePath(p) {
if (pages.has(p)) return { page: p };
if (fileExists(p)) return { file: p };
if (!p.endsWith('/') && pages.has(p + '/')) return { page: p + '/', note: '末尾の/が無い(リダイレクトを1回挟む)' };
const r = redirectRules.find((r) => r.re.test(p));
if (r) return { redirect: r.to, note: `_redirectsで${r.status}転送` };
return null;
}
const seen = new Set();
let checkedLinks = 0;
for (const [p, v] of pages) {
for (const { kind, href } of v.links) {
if (!href || /^(mailto|tel|javascript|data):/i.test(href)) continue;
let u;
try { u = new URL(href, new URL(p, SITE)); } catch { add('error', 'リンクのURLが壊れている', p, href); continue; }
if (!/^https?:$/.test(u.protocol) || !SAME_HOSTS.has(u.host)) continue; // 外部リンクは見ない
if (u.host !== SITE.host) add('warn', '正式でないホストへのリンク', p, href);
let target;
try { target = decodeURI(u.pathname); } catch { target = u.pathname; }
const key = `${p}|${kind}|${target}|${u.hash}`;
if (seen.has(key)) continue;
seen.add(key);
checkedLinks++;
const res = resolvePath(target);
if (!res) { add('error', kind === 'img' ? '画像が見つからない' : '内部リンク切れ', p, href); continue; }
if (res.note) add('warn', 'リンク先が転送される', p, `${href}(${res.note})`);
// #見出し へのリンクは、相手のページにその id があるか
if (kind === 'a' && u.hash.length > 1 && res.page) {
const id = decodeURIComponent(u.hash.slice(1));
if (!pages.get(res.page).ids.has(id)) add('error', 'ページ内リンクの行き先が無い', p, href);
}
}
}
// サイトマップにあるのに、書き出されていないURL
for (const sp of sitemapPaths) if (!pages.has(sp)) add('error', 'サイトマップのURLのページが無い', sp);
// ---- 7. 結果を出す ----
const byKind = new Map();
for (const i of issues) byKind.set(`${i.level}\t${i.kind}`, [...(byKind.get(`${i.level}\t${i.kind}`) ?? []), i]);
console.log(`点検したページ: ${pages.size}(noindex: ${[...pages.values()].filter((v) => v.noindex).length})`);
console.log(`サイトマップのURL: ${sitemapPaths.size} / 確かめた内部リンク・画像: ${checkedLinks}`);
if (skipped.length) console.log(`HTMLでないので飛ばした: ${skipped.join(' , ')}`);
console.log(`エラー: ${issues.filter((i) => i.level === 'error').length} / 注意: ${issues.filter((i) => i.level === 'warn').length}\n`);
for (const [k, list] of [...byKind].sort()) {
const [level, kind] = k.split('\t');
console.log(`[${level === 'error' ? 'エラー' : '注意'}] ${kind}: ${list.length}件`);
for (const i of list.slice(0, 10)) console.log(` - ${i.page}${i.detail ? ` ${i.detail}` : ''}`);
if (list.length > 10) console.log(` …ほか${list.length - 10}件`);
}
if (process.env.REPORT_JSON) fs.writeFileSync(process.env.REPORT_JSON, JSON.stringify(issues, null, 2));
process.exitCode = issues.some((i) => i.level === 'error') ? 1 : 0;作りの
- ファイルの
場所から URLを 決める。 about/index.htmlは/about/、というように、 書き出しの 形 (Astroの build.format: 'directory')に合わせて パスを 決め、 canonicalやリンクの 先と 比べます。 - 重複は、
インデックス可の noindexのページどうしだけで 数える。 ページは 検索結果に 出ないので、 titleが 同じでも 問題にしません。 - 内部リンクは、
同じリンクを ヘッダーや1回だけ 確かめる。 フッターの リンクは 全ページに 出るので、 ページ・種類・ 行き先・ #の組み合わせで 重複を 除いています。 _redirectsの規則を 転送の読む。 元に なっている URLへの リンクは、 切れではなく 「転送される」注意と して 出します。 :splatのような書き方は *と同じに 扱う、 簡単な 読み方です。
このブログで流した結果:エラー0件、注意74件
2026年10月4日21時55分にdistを
$ node seo-audit.mjs ./dist https://www.yamayamabloglink.com
点検したページ: 265(noindex: 11)
サイトマップのURL: 254 / 確かめた内部リンク・画像: 11134
HTMLでないので飛ばした: /feed/
エラー: 0 / 注意: 74
[注意] descriptionが短い: 69件
- /about/ 45字: やまやまブログの運営者、津嘉山 洸(株式会社bundlyze CTO)のプロフィールです。
- /category/bridal/同棲/ 18字: 同棲に関する記事の一覧です(7件)。
…(中略)
[注意] descriptionが長い: 1件
- /line-reminder-cloudflare-workers-cron/ 132字
[注意] titleが長い: 4件
- /bridal-fair-form-mobile/ 56字
- /one-click-copy/ 57字
- /personal-gym-line-trial-booking/ 56字
- /wordpress-astro-cloudflare-migration/ 78字エラーに
見つかった
| 見つかった |
件数 | 扱い |
|---|---|---|
/feed/が |
4件 |
スクリプトをfeed/index.htmlと |
| 説明文が |
69件 | 残している。 内訳は、 |
| titleが |
4件 | 直した。 3件は |
| 説明文が |
1件 |
直した。 言い回しを |
最初の
説明文が
わざと壊したコピーで、見落とさないかを確かめた
エラーがdistの
// 点検の網にかかるかを確かめるため、コピーした dist をわざと壊す
import fs from 'node:fs';
const D = 'dist-broken';
const edit = (p, fn) => { const f = `${D}${p}index.html`; fs.writeFileSync(f, fn(fs.readFileSync(f, 'utf8'))); };
const titleOf = (p) => fs.readFileSync(`${D}${p}index.html`, 'utf8').match(/<title>([^<]*)<\/title>/)[1];
// 1. titleの重複
edit('/xampp-setup/', (h) => h.replace(/<title>[^<]*<\/title>/, `<title>${titleOf('/php-array-loop/')}</title>`));
// 2. JSON-LDを壊す(最初のJSON-LDの最後の } を消す)
edit('/indexnow-cloudflare-pages-deploy/', (h) => h.replace(/(<script type="application\/ld\+json">[\s\S]*?)\}(<\/script>)/, '$1$2'));
// 3. h1を2つにする
edit('/supabase-rls-multi-tenant-reservations/', (h) => h.replace('<main', '<h1>重複した見出し</h1><main'));
// 4. canonicalを別のページに向ける
edit('/llms-txt-astro-build-auto-generate/', (h) => h.replace(/(<link rel="canonical" href=")[^"]+/, '$1https://www.yamayamabloglink.com/indexnow-cloudflare-pages-deploy/'));
// 5. サイトマップに載っているページにnoindexを入れる
edit('/about/', (h) => h.replace('<head>', '<head><meta name="robots" content="noindex">'));
// 6. 内部リンク切れと、ページ内リンクの行き先切れ
edit('/structured-data-jsonld-implementation-check/', (h) => h.replace('</main>', '<a href="/no-such-page/">x</a><a href="/indexnow-cloudflare-pages-deploy/#no-such-heading">y</a></main>'));
console.log('壊したコピーを作りました');$ node break.mjs && node seo-audit.mjs ./dist-broken https://www.yamayamabloglink.com
壊したコピーを作りました
点検したページ: 265(noindex: 12)
サイトマップのURL: 254 / 確かめた内部リンク・画像: 11136
HTMLでないので飛ばした: /feed/
エラー: 5 / 注意: 78
[エラー] JSON-LDが読めない: 1件
- /indexnow-cloudflare-pages-deploy/ 1個目: Expected ',' or '}' after property value in JSON at position 1604 (line 1 column 1605)
[エラー] noindexなのにサイトマップに載っている: 1件
- /about/ noindex
[エラー] titleが重複: 1件
- /php-array-loop/ , /xampp-setup/ PHPの配列ループ|foreach・forの書き方と使い分け | やまやまブログ
[エラー] ページ内リンクの行き先が無い: 1件
- /structured-data-jsonld-implementation-check/ /indexnow-cloudflare-pages-deploy/#no-such-heading
[エラー] 内部リンク切れ: 1件
- /structured-data-jsonld-implementation-check/ /no-such-page/
[注意] JSON-LDのURLとcanonicalが違う: 1件
- /llms-txt-astro-build-auto-generate/ BlogPosting: https://www.yamayamabloglink.com/llms-txt-astro-build-auto-generate/
[注意] canonicalが自分以外を指す: 1件
- /llms-txt-astro-build-auto-generate/ https://www.yamayamabloglink.com/indexnow-cloudflare-pages-deploy/
[注意] h1が複数: 1件
- /supabase-rls-multi-tenant-reservations/ 2個
[注意] og:urlとcanonicalが違う: 1件
- /llms-txt-astro-build-auto-generate/ https://www.yamayamabloglink.com/llms-txt-astro-build-auto-generate/ / https://www.yamayamabloglink.com/indexnow-cloudflare-pages-deploy/
(長さの注意は、壊す前と同じなので省略)入れた
ビルドのたびに流すときは、終了コードで止める
ビルドのpackage.jsonの
{
"scripts": {
"build": "astro build && node scripts/seo-audit.mjs dist https://www.example.com"
}
}この
点検で
このスクリプトでは分からないこと
ファイルを
- サーバーや
CDNが 付ける ヘッダー。 _headersは読んでいますが、 関数 (この ブログでは functions/_middleware.js)が付ける ヘッダーや 転送は 見ていません。 公開後に curl -Iで確かめます。 - robots.txtでの
ブロック。 Googleの説明では、 noindexは robots.txtで ブロックされていない ページでしか 働きません。 robots.txtと noindexの 組み合わせは、 この スクリプトでは 突き合わせていません。 - JavaScriptで
書き換える 表示の部分。 あとに titleやリンクを 書き換える サイトでは、 書き出した HTMLと 読まれる 中身が 違います。 - 外部サイトへの
リンク。 相手のサイトに 問い合わせる ことに なるので、 対象から 外しています。 - 見た目の
長さ。 長さは文字数で 数えていて、 検索結果で 切られる 幅 (ピクセル)ではありません。
検索と
動作確認した環境
確認日は
| 項目 | バージョン・内容 |
|---|---|
| OS | macOS 26 |
| Node.js | 22.23.2 |
| cheerio | 1.2.0 |
| 点検した |
このdistのfeed/index.htmlを |
| 確かめた |
実際のdistでの |
ビルドのpage.htmlのように
あとからfileExistsがscripts/site-audit.mjsでは、
参照した公式ドキュメント(確認日:2026年10月4日)
- Google 検索セントラル
「Google 検索結果の :titleにタイトルリンクに 影響を 与える」 長さの 上限は なく 端末の 幅で 切られる こと (この 記述は 英語版に あり、 日本語版には 「不必要に 長い ものは 避ける」とだけ ある)、 ページごとに 違う titleを 付ける こと、 1か所だけが 違う 長い 定型の 題を 避ける こと - Google 検索セントラル
「検索結果の :meta descriptionにスニペットを 管理する」 長さの 上限は ない こと、 同じか 似た 説明文が 並ぶのは 役に 立たない こと - Google 検索セントラル
「重複した :canonicalはURL を 統合する」 絶対URLで 書く こと、 サイトマップと canonicalで 別々の URLを 示さない こと、 正規の URLを サイトマップに 載せる こと、 noindexを 正規化に 使わない こと - Google 検索セントラル
「noindex を :metaタグと使用して 検索インデックス登録を ブロックする」 X-Robots-Tagヘッダーの 2つの 指定の しかた、 robots.txtで ブロックしていると noindexが 働かない こと
よくある質問
titleやdescriptionの重複は、公開中のサイトを巡回するツールでなくても見つけられますか?
静的に
titleとdescriptionは何文字までにすればよいですか?
Googleの
h1が2つあると、検索で不利になりますか?
この
noindexのページがサイトマップに載っていると、何がいけないのですか?
サイトマップは