Skip to content

Commit 0d7743e

Browse files
authored
feat(docs): wire AI-crawler baseline (preset 3.4.0 + validator + llms.txt) (#239)
1 parent 5f6bdbf commit 0d7743e

3 files changed

Lines changed: 207 additions & 1 deletion

File tree

docs/package.json

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,6 +6,8 @@
66
"docusaurus": "docusaurus",
77
"start": "docusaurus start",
88
"build": "docusaurus build",
9+
"postbuild": "node scripts/validate-ai-baseline.mjs",
10+
"validate:ai-baseline": "node scripts/validate-ai-baseline.mjs",
911
"swizzle": "docusaurus swizzle",
1012
"deploy": "docusaurus deploy",
1113
"clear": "docusaurus clear",
@@ -15,7 +17,7 @@
1517
"ci": "npm ci --legacy-peer-deps && npm run build"
1618
},
1719
"dependencies": {
18-
"@conduction/docusaurus-preset": "^2.6.1",
20+
"@conduction/docusaurus-preset": "^3.4.0",
1921
"@docusaurus/core": "^3.7.0",
2022
"@docusaurus/preset-classic": "^3.7.0",
2123
"@docusaurus/theme-mermaid": "^3.7.0",
Lines changed: 177 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,177 @@
1+
#!/usr/bin/env node
2+
/**
3+
* scripts/validate-ai-baseline.mjs
4+
*
5+
* Generic AI-crawler baseline validator. Runs as a postbuild step on
6+
* every Conduction Docusaurus site that consumes
7+
* @conduction/docusaurus-preset >= 3.4.0. Asserts the SSG output
8+
* carries the contract AI crawlers (GPTBot, ClaudeBot, PerplexityBot,
9+
* OAI-SearchBot, Claude-SearchBot, Google AI Overviews) expect.
10+
*
11+
* Universal checks only - no site-specific routes. Sites that want
12+
* additional gates (per-app SoftwareApplication, FAQPage on specific
13+
* pages, etc.) extend this script in place. See conduction-website's
14+
* version for an example of additional checks.
15+
*
16+
* Exit codes:
17+
* 0 all checks passed
18+
* 1 one or more checks failed (CI should block)
19+
* 2 build directory not found (script invoked before build)
20+
*/
21+
22+
import {readFileSync, existsSync, statSync} from 'node:fs';
23+
import {join, resolve} from 'node:path';
24+
25+
const buildDir = resolve(process.argv[2] || 'build');
26+
27+
if (!existsSync(buildDir)) {
28+
console.error(`✗ build directory not found: ${buildDir}`);
29+
console.error(` Run \`npx docusaurus build\` first.`);
30+
process.exit(2);
31+
}
32+
33+
const results = [];
34+
35+
function check(name, fn) {
36+
try {
37+
const r = fn();
38+
results.push({name, ok: r.ok, msg: r.msg});
39+
} catch (e) {
40+
results.push({name, ok: false, msg: `threw: ${e.message}`});
41+
}
42+
}
43+
44+
function readBuild(p) {
45+
return readFileSync(join(buildDir, p), 'utf8');
46+
}
47+
48+
/* robots.txt - shipped by the preset's ai-crawling plugin (or the
49+
site's own static/robots.txt). Either way, the file must exist
50+
and name at least one AI search bot so a `grep` audit can confirm
51+
the posture at a glance. */
52+
check('robots.txt exists and is non-empty', () => {
53+
const path = join(buildDir, 'robots.txt');
54+
if (!existsSync(path)) return {ok: false, msg: 'missing'};
55+
const size = statSync(path).size;
56+
if (size < 50) return {ok: false, msg: `too small (${size} bytes)`};
57+
return {ok: true, msg: `${size} bytes`};
58+
});
59+
60+
check('robots.txt names at least one AI search bot', () => {
61+
const body = readBuild('robots.txt');
62+
const candidates = ['OAI-SearchBot', 'Claude-SearchBot', 'PerplexityBot', 'ChatGPT-User', 'Claude-User'];
63+
const found = candidates.filter(ua => body.includes(`User-agent: ${ua}`));
64+
if (found.length === 0) {
65+
return {ok: false, msg: `none of [${candidates.join(', ')}] referenced`};
66+
}
67+
return {ok: true, msg: `${found.length} bot(s): ${found.join(', ')}`};
68+
});
69+
70+
check('robots.txt has a Sitemap line', () => {
71+
const body = readBuild('robots.txt');
72+
const matches = body.match(/^Sitemap:\s+https?:\/\//gm) || [];
73+
if (matches.length === 0) return {ok: false, msg: 'no Sitemap: line'};
74+
return {ok: true, msg: `${matches.length} sitemap line(s)`};
75+
});
76+
77+
/* sitemap.xml - emitted by @docusaurus/plugin-sitemap (loaded via
78+
the classic preset). Locale-specific sitemaps (e.g. /nl/sitemap.xml)
79+
are present for i18n builds; we only check the canonical one
80+
because some sites are single-locale. */
81+
check('sitemap.xml exists and has at least 1 URL', () => {
82+
const path = join(buildDir, 'sitemap.xml');
83+
if (!existsSync(path)) return {ok: false, msg: 'missing'};
84+
const body = readBuild('sitemap.xml');
85+
const n = (body.match(/<loc>/g) || []).length;
86+
if (n < 1) return {ok: false, msg: 'no <loc> entries'};
87+
return {ok: true, msg: `${n} URLs`};
88+
});
89+
90+
/* Helper for the JSON-LD checks below. Docusaurus emits ld+json
91+
tags via two paths with different attribute ordering: top-level
92+
headTags renders <script type="..."> first, while Helmet (used
93+
by <Head> from inside React components like <DetailHero>, <FAQ>)
94+
prefixes data-rh="true". The regex matches either ordering. */
95+
function extractJsonLdBlocks(html) {
96+
const out = [];
97+
const re = /<script\b[^>]*\btype="application\/ld\+json"[^>]*>([\s\S]*?)<\/script>/g;
98+
let m;
99+
while ((m = re.exec(html)) !== null) {
100+
out.push(m[1]);
101+
}
102+
return out;
103+
}
104+
105+
check('homepage emits >= 2 JSON-LD blocks, all valid JSON', () => {
106+
if (!existsSync(join(buildDir, 'index.html'))) return {ok: false, msg: 'no index.html'};
107+
const html = readBuild('index.html');
108+
const blocks = extractJsonLdBlocks(html);
109+
if (blocks.length < 2) return {ok: false, msg: `only ${blocks.length} block(s)`};
110+
for (const [i, b] of blocks.entries()) {
111+
try {JSON.parse(b);} catch (e) {
112+
return {ok: false, msg: `block ${i} invalid JSON: ${e.message}`};
113+
}
114+
}
115+
return {ok: true, msg: `${blocks.length} blocks, all valid`};
116+
});
117+
118+
check('homepage JSON-LD includes Organization and WebSite', () => {
119+
const html = readBuild('index.html');
120+
const types = extractJsonLdBlocks(html).map(b => {
121+
try {return JSON.parse(b)['@type'];} catch {return null;}
122+
});
123+
const want = ['Organization', 'WebSite'];
124+
const missing = want.filter(t => !types.includes(t));
125+
if (missing.length) return {ok: false, msg: `missing @type: ${missing.join(', ')}`};
126+
return {ok: true, msg: types.filter(Boolean).join(' + ')};
127+
});
128+
129+
/* Social-card meta. og:image is the one that breaks LinkedIn /
130+
Slack / AI previews when it 404s, so we also resolve the URL to
131+
a local file in the build output. */
132+
function metaTag(html, key) {
133+
const re = new RegExp(`<meta[^>]+(?:name|property)="${key}"[^>]+content="([^"]+)"`, 'i');
134+
const m = html.match(re);
135+
return m ? m[1] : null;
136+
}
137+
138+
check('homepage has og:image, og:type, twitter:site, twitter:card', () => {
139+
const html = readBuild('index.html');
140+
const checks = {
141+
'og:image': metaTag(html, 'og:image'),
142+
'og:type': metaTag(html, 'og:type'),
143+
'twitter:site': metaTag(html, 'twitter:site'),
144+
'twitter:card': metaTag(html, 'twitter:card'),
145+
};
146+
const missing = Object.entries(checks).filter(([, v]) => !v).map(([k]) => k);
147+
if (missing.length) return {ok: false, msg: `missing: ${missing.join(', ')}`};
148+
return {ok: true, msg: 'all four present'};
149+
});
150+
151+
check('og:image URL resolves to a file in the build', () => {
152+
const html = readBuild('index.html');
153+
const url = metaTag(html, 'og:image');
154+
if (!url) return {ok: false, msg: 'no og:image meta'};
155+
const path = url.replace(/^https?:\/\/[^/]+\//, '');
156+
const local = join(buildDir, path);
157+
if (!existsSync(local)) return {ok: false, msg: `og:image refers to ${url}, not found at ${local}`};
158+
const size = statSync(local).size;
159+
if (size < 1024) return {ok: false, msg: `og:image file suspiciously small (${size} bytes)`};
160+
return {ok: true, msg: `${path} (${size} bytes)`};
161+
});
162+
163+
/* Report */
164+
let failed = 0;
165+
for (const {name, ok, msg} of results) {
166+
const icon = ok ? '✓' : '✗';
167+
console.log(`${icon} ${name} - ${msg}`);
168+
if (!ok) failed++;
169+
}
170+
console.log('');
171+
if (failed) {
172+
console.error(`${failed} of ${results.length} checks failed.`);
173+
console.error('AI-crawler baseline regressed. Fix the failures above before merging.');
174+
process.exit(1);
175+
} else {
176+
console.log(`All ${results.length} AI-baseline checks passed.`);
177+
}

docs/static/llms.txt

Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,27 @@
1+
# SoftwareCatalog
2+
3+
> SoftwareCatalog is an open-source IT-asset-management app for the Nextcloud workspace.
4+
5+
SoftwareCatalog is an open-source IT-asset-management app for the Nextcloud workspace. It keeps a central register of every application, licence, contract, and dependency the organisation runs, with renewal alerts before the auto-renewal date. Dashboard widgets surface upcoming renewals, an inventory snapshot by category, and discovery deltas showing newly added or removed apps. Built for the IT department that needs to know what runs where, without a separate asset database or second login. Released under EUPL-1.2 and maintained by Conduction since 2019.
6+
7+
## Docs
8+
9+
- [Tutorials](https://softwarecatalog.conduction.nl/tutorials): step-by-step walkthroughs.
10+
- [Features](https://softwarecatalog.conduction.nl/Features): what the app does, organised by capability.
11+
- [Use cases](https://softwarecatalog.conduction.nl/UseCases): real customer scenarios.
12+
- [Integrations](https://softwarecatalog.conduction.nl/Integrations): which external systems the app talks to.
13+
- [Technical](https://softwarecatalog.conduction.nl/Technical): architecture, data model, security posture.
14+
- [User guide](https://softwarecatalog.conduction.nl/user-guide): end-user documentation.
15+
- [API](https://softwarecatalog.conduction.nl/api): OpenAPI reference.
16+
17+
## Optional
18+
19+
- [Install](https://softwarecatalog.conduction.nl/installation): install from the Nextcloud app store.
20+
- [Source](https://softwarecatalog.conduction.nl): repository and issue tracker.
21+
- [App page](https://www.conduction.nl/apps/softwarecatalog): product positioning on conduction.nl.
22+
23+
## Contact
24+
25+
- Email: info@conduction.nl
26+
- GitHub: https://github.com/ConductionNL/softwarecatalog
27+
- Conduction B.V. · KvK 76741850 · Lauriergracht 14h, Amsterdam, Netherlands

0 commit comments

Comments
 (0)