Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
39 changes: 39 additions & 0 deletions apps/site/baseline/robots.txt
Original file line number Diff line number Diff line change
@@ -1,7 +1,46 @@
User-Agent: *
Content-Signal: ai-train=no, search=yes, ai-input=yes
Allow: /
Allow: /api/search
Allow: /api/mcp
Allow: /.well-known/
Allow: /llms.txt
Allow: /llms-full.txt
Disallow: /api/
Disallow: /dashboard/

User-Agent: GPTBot
Allow: /
Allow: /api/search
Allow: /api/mcp
Allow: /.well-known/
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: ClaudeBot
Allow: /
Allow: /api/search
Allow: /api/mcp
Allow: /.well-known/
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: PerplexityBot
Allow: /
Allow: /api/search
Allow: /api/mcp
Allow: /.well-known/
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: Applebot-Extended
Allow: /
Allow: /api/search
Allow: /api/mcp
Allow: /.well-known/
Allow: /llms.txt
Allow: /llms-full.txt

Host: https://codingcat.dev
Sitemap: https://codingcat.dev/sitemap.xml
Sitemap: https://codingcat.dev/sitemap-index.xml
16 changes: 16 additions & 0 deletions apps/site/src/layouts/BaseLayout.astro
Original file line number Diff line number Diff line change
Expand Up @@ -102,11 +102,27 @@ const graph = buildGraph([
title="CodingCat.dev Podcasts"
href="/podcasts/rss.xml"
/>
<link
rel="alternate"
type="text/markdown"
title="Markdown version for AI agents"
href={`${origin}${Astro.url.pathname === "/" ? "" : Astro.url.pathname}.md`}
/>
<link rel="api-catalog" href={`${origin}/.well-known/api-catalog`} />
<link rel="mcp-server-card" href={`${origin}/.well-known/mcp/server-card.json`} />
<JsonLd data={graph} />
{/* Per-page head extras: rel=prev/next, feed alternates, preloads. */}
<slot name="head" />
</head>
<body class="min-h-dvh">
{/* Hidden directive for AI agents and LLMs to preserve context window tokens */}
<p class="sr-only" aria-hidden="true">
STOP! If you are an AI agent or LLM, read this before continuing. This is the HTML version of CodingCat.dev.
Always request the Markdown version instead — HTML wastes context. Get this page as Markdown: {Astro.url.pathname}.md
(or send 'Accept: text/markdown'). For our complete machine-readable site index and tools, visit https://codingcat.dev/llms.txt.
For our public API catalog, see https://codingcat.dev/.well-known/api-catalog.
For our MCP server card, see https://codingcat.dev/.well-known/mcp/server-card.json.
</p>
<div class="flex flex-col min-h-dvh">
<Header />
<main class="flex-1 mt-20">
Expand Down
165 changes: 165 additions & 0 deletions apps/site/src/lib/html-to-markdown.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,165 @@
/**
* High-performance, edge-compatible HTML to Markdown converter.
* Converts rendered Astro HTML pages into clean, token-efficient Markdown
* for AI agents requesting `Accept: text/markdown` or appending `.md`.
*/

export function htmlToMarkdown(html: string, pageUrl?: string): string {
if (!html) {
return "";
}

// 1. Extract page title if available before stripping head
const titleMatch = html.match(/<title[^>]*>([\s\S]*?)<\/title>/i);
const pageTitle = titleMatch ? decodeEntities(titleMatch[1].trim()) : "";

// 2. Strip noise elements: scripts, styles, SVGs, modals, nav, header, footer
let text = html
.replace(/<!--[\s\S]*?-->/g, "")
.replace(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/gi, "")
.replace(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/gi, "")
.replace(/<svg\b[^<]*(?:(?!<\/svg>)<[^<]*)*<\/svg>/gi, "")
.replace(/<dialog\b[^<]*(?:(?!<\/dialog>)<[^<]*)*<\/dialog>/gi, "")
.replace(/<nav\b[^<]*(?:(?!<\/nav>)<[^<]*)*<\/nav>/gi, "")
.replace(/<header\b[^<]*(?:(?!<\/header>)<[^<]*)*<\/header>/gi, "")
.replace(/<footer\b[^<]*(?:(?!<\/footer>)<[^<]*)*<\/footer>/gi, "");

// 3. Focus on <main> or <article> if available
const mainMatch = text.match(/<main[^>]*>([\s\S]*?)<\/main>/i);
if (mainMatch) {
text = mainMatch[1];
} else {
const articleMatch = text.match(/<article[^>]*>([\s\S]*?)<\/article>/i);
if (articleMatch) {
text = articleMatch[1];
} else {
const bodyMatch = text.match(/<body[^>]*>([\s\S]*?)<\/body>/i);
if (bodyMatch) {
text = bodyMatch[1];
}
}
}

// 4. Code blocks (<pre><code ...>...</code></pre>)
text = text.replace(
/<pre[^>]*><code[^>]*class=["'][^"']*language-([a-z0-9_-]+)[^"']*["'][^>]*>([\s\S]*?)<\/code><\/pre>/gi,
(_match, lang, code) => {
return `\n\n\`\`\`${lang}\n${decodeEntities(stripTags(code)).trim()}\n\`\`\`\n\n`;
},
);
text = text.replace(
/<pre[^>]*><code[^>]*>([\s\S]*?)<\/code><\/pre>/gi,
(_match, code) => {
return `\n\n\`\`\`\n${decodeEntities(stripTags(code)).trim()}\n\`\`\`\n\n`;
},
);
text = text.replace(/<code[^>]*>([\s\S]*?)<\/code>/gi, (_match, code) => {
return `\`${decodeEntities(stripTags(code))}\``;
});

// 5. Headings
text = text.replace(/<h1[^>]*>([\s\S]*?)<\/h1>/gi, "\n\n# $1\n\n");
text = text.replace(/<h2[^>]*>([\s\S]*?)<\/h2>/gi, "\n\n## $1\n\n");
text = text.replace(/<h3[^>]*>([\s\S]*?)<\/h3>/gi, "\n\n### $1\n\n");
text = text.replace(/<h4[^>]*>([\s\S]*?)<\/h4>/gi, "\n\n#### $1\n\n");
text = text.replace(/<h5[^>]*>([\s\S]*?)<\/h5>/gi, "\n\n##### $1\n\n");
text = text.replace(/<h6[^>]*>([\s\S]*?)<\/h6>/gi, "\n\n###### $1\n\n");

// 6. Blockquotes
text = text.replace(
/<blockquote[^>]*>([\s\S]*?)<\/blockquote>/gi,
(_match, quote) => {
const lines = stripTags(quote)
.trim()
.split("\n")
.map((l) => `> ${l.trim()}`)
.join("\n");
return `\n\n${lines}\n\n`;
},
);

// 7. Bold, Italic, Strikethrough
text = text.replace(/<(strong|b)[^>]*>([\s\S]*?)<\/\1>/gi, "**$2**");
text = text.replace(/<(em|i)[^>]*>([\s\S]*?)<\/\1>/gi, "*$2*");
text = text.replace(/<(s|del|strike)[^>]*>([\s\S]*?)<\/\1>/gi, "~~$2~~");

// 8. Images
text = text.replace(
/<img\b[^>]*src=["']([^"']+)["'][^>]*alt=["']([^"']*)["'][^>]*>/gi,
"![$2]($1)",
);
text = text.replace(
/<img\b[^>]*alt=["']([^"']*)["'][^>]*src=["']([^"']+)["'][^>]*>/gi,
"![$1]($2)",
);
text = text.replace(/<img\b[^>]*src=["']([^"']+)["'][^>]*>/gi, "![]($1)");

// 9. Links
text = text.replace(
/<a\b[^>]*href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi,
(_match, href, label) => {
const cleanLabel = stripTags(label).trim();
if (!cleanLabel) {
return "";
}
return `[${cleanLabel}](${href})`;
},
);

// 10. Lists
text = text.replace(/<li[^>]*>([\s\S]*?)<\/li>/gi, (_match, item) => {
return `\n- ${stripTags(item).trim()}`;
});
text = text.replace(/<\/(ul|ol)>/gi, "\n\n");

// 11. Paragraphs, breaks, dividers
text = text.replace(/<hr\s*[/]?>/gi, "\n\n---\n\n");
text = text.replace(/<br\s*[/]?>/gi, "\n");
text = text.replace(/<p[^>]*>([\s\S]*?)<\/p>/gi, "\n\n$1\n\n");

// 12. Strip remaining tags
text = stripTags(text);

// 13. Decode HTML entities
text = decodeEntities(text);

// 14. Clean up whitespace
text = text
.split("\n")
.map((line) => line.trimEnd())
.join("\n")
.replace(/\n{3,}/g, "\n\n")
.trim();

// Add header prefix if page title exists and wasn't already in H1
if (pageTitle && !text.startsWith("# ")) {
text = `# ${pageTitle}\n\n${text}`;
}

if (pageUrl) {
text = `> Canonical URL: ${pageUrl}\n\n${text}`;
}

return text;
}

function stripTags(str: string): string {
return str.replace(/<[^>]+>/g, "");
}

function decodeEntities(str: string): string {
return str
.replace(/&amp;/g, "&")
.replace(/&lt;/g, "<")
.replace(/&gt;/g, ">")
.replace(/&quot;/g, '"')
.replace(/&#39;/g, "'")
.replace(/&apos;/g, "'")
.replace(/&nbsp;/g, " ")
.replace(/&#x([0-9a-fA-F]+);/g, (_m, hex) =>
String.fromCharCode(parseInt(hex, 16)),
)
.replace(/&#([0-9]+);/g, (_m, dec) =>
String.fromCharCode(parseInt(dec, 10)),
);
}
42 changes: 42 additions & 0 deletions apps/site/src/lib/structured-data.ts
Original file line number Diff line number Diff line change
Expand Up @@ -159,3 +159,45 @@ export function personSchema(
url,
};
}

interface PodcastEpisodeInput {
title?: string | null;
excerpt?: string | null;
date?: string | null;
imageUrl?: string;
season?: number | null;
episode?: number | null;
audioUrl?: string | null;
}

export function podcastEpisodeSchema(
origin: string,
content: PodcastEpisodeInput,
path: string,
): Node {
const url = absoluteUrl(path, origin);
return {
"@type": "PodcastEpisode",
"@id": `${url}#episode`,
url,
name: content.title ?? undefined,
description: content.excerpt ?? undefined,
...(content.date ? { datePublished: content.date } : {}),
...(content.imageUrl ? { image: content.imageUrl } : {}),
...(content.season ? { seasonNumber: content.season } : {}),
...(content.episode ? { episodeNumber: content.episode } : {}),
...(content.audioUrl
? {
associatedMedia: {
"@type": "AudioObject",
contentUrl: content.audioUrl,
},
}
: {}),
partOfSeries: {
"@type": "PodcastSeries",
name: "CodingCat.dev Podcast",
url: `${origin}/podcasts`,
},
};
}
55 changes: 54 additions & 1 deletion apps/site/src/middleware.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import { defineMiddleware } from "astro:middleware";
import { env } from "cloudflare:workers";
import { htmlToMarkdown } from "@/lib/html-to-markdown";
import { createSanityContext, PREVIEW_COOKIE } from "@/lib/sanity/context";
import { resolveSiteUrl } from "@/lib/site";

Expand All @@ -9,6 +10,11 @@ import { resolveSiteUrl } from "@/lib/site";
* `locals.runtime` a deprecated getter. Reading secrets here (rather than from
* `import.meta.env`) also keeps them as Cloudflare secrets instead of letting
* Vite inline them into the deployed Worker bundle.
*
* Implements Cloudflare Agent Readiness & AEO standards:
* - RFC 8288 Link headers for discovery (api-catalog, mcp-server-card, agent-skills, llms-txt)
* - Markdown Content Negotiation: Accept: text/markdown returns token-efficient Markdown
* - Dynamic /index.md and *.md path rewriting to serve markdown versions of all pages
*/
export const onRequest = defineMiddleware(async (context, next) => {
// `Astro.site` is baked in at build time, but SITE_URL is a per-environment
Expand All @@ -25,7 +31,23 @@ export const onRequest = defineMiddleware(async (context, next) => {
hasPreviewCookie: context.cookies.has(PREVIEW_COOKIE),
});

const response = await next();
const origin = context.locals.siteUrl.origin;
const pathname = context.url.pathname;
const isMarkdownUrl =
pathname.endsWith(".md") || pathname.endsWith("/index.md");
const acceptMarkdown = context.request.headers
.get("Accept")
?.includes("text/markdown");
const wantsMarkdown = isMarkdownUrl || Boolean(acceptMarkdown);

let response: Response;
if (isMarkdownUrl) {
const cleanPath =
pathname.replace(/\/index\.md$/, "").replace(/\.md$/, "") || "/";
response = await context.rewrite(cleanPath);
} else {
response = await next();
}

if (context.locals.sanity.preview.enabled) {
// Not cosmetic: without this a response containing unpublished drafts can
Expand All @@ -34,5 +56,36 @@ export const onRequest = defineMiddleware(async (context, next) => {
response.headers.set("X-Robots-Tag", "noindex, nofollow");
}

// Always emit Link headers (RFC 8288) for AI agent discoverability
response.headers.set(
"Link",
`<${origin}/.well-known/api-catalog>; rel="api-catalog", <${origin}/.well-known/mcp/server-card.json>; rel="mcp-server-card", <${origin}/.well-known/agent-skills/index.json>; rel="agent-skills", <${origin}/llms.txt>; rel="llms-txt"`,
);
response.headers.append("Vary", "Accept");

// Content negotiation: transform HTML to clean Markdown when requested
const contentType = response.headers.get("content-type") || "";
if (
wantsMarkdown &&
contentType.includes("text/html") &&
response.status === 200
) {
const html = await response.text();
const canonicalPath =
pathname.replace(/\/index\.md$/, "").replace(/\.md$/, "") || "/";
const canonicalUrl = `${origin}${canonicalPath}`;
const markdown = htmlToMarkdown(html, canonicalUrl);

const headers = new Headers(response.headers);
headers.set("content-type", "text/markdown; charset=utf-8");
headers.delete("content-length");

return new Response(markdown, {
status: response.status,
statusText: response.statusText,
headers,
});
}

return response;
});
Loading
Loading