From 27cff0dd1095b2133f770eb4b040fa0ee056222d Mon Sep 17 00:00:00 2001 From: Alex Patterson Date: Wed, 7 Oct 2026 20:19:17 -0400 Subject: [PATCH] feat(aeo): implement Cloudflare Agent Readiness standards and enhanced AEO - Add Content-Signal and bot access control rules to robots.txt - Add /llms.txt and /llms-full.txt following llmstxt.org specification - Add RFC 9727 /.well-known/api-catalog endpoint - Add MCP server cards at /.well-known/mcp/server-card.json and /.well-known/mcp.json - Add Cloudflare Agent Skills RFC catalog at /.well-known/agent-skills/index.json - Add Streamable HTTP MCP JSON-RPC 2.0 endpoint at /api/mcp for search_content - Add edge HTML-to-Markdown transformer with Accept: text/markdown content negotiation and *.md route rewrites - Add RFC 8288 Link discovery headers in middleware and link tags in BaseLayout - Add PodcastEpisode Schema.org structured data and hidden agent token-saving directives --- apps/site/baseline/robots.txt | 39 +++ apps/site/src/layouts/BaseLayout.astro | 16 ++ apps/site/src/lib/html-to-markdown.ts | 165 ++++++++++++ apps/site/src/lib/structured-data.ts | 42 +++ apps/site/src/middleware.ts | 55 +++- .../.well-known/agent-skills/index.json.ts | 52 ++++ .../site/src/pages/.well-known/api-catalog.ts | 94 +++++++ apps/site/src/pages/.well-known/mcp.json.ts | 25 ++ .../pages/.well-known/mcp/server-card.json.ts | 56 ++++ apps/site/src/pages/api/mcp.ts | 255 ++++++++++++++++++ apps/site/src/pages/llms-full.txt.ts | 76 ++++++ apps/site/src/pages/llms.txt.ts | 46 ++++ apps/site/src/pages/podcast/[slug].astro | 39 ++- apps/site/src/pages/robots.txt.ts | 45 ++++ 14 files changed, 995 insertions(+), 10 deletions(-) create mode 100644 apps/site/src/lib/html-to-markdown.ts create mode 100644 apps/site/src/pages/.well-known/agent-skills/index.json.ts create mode 100644 apps/site/src/pages/.well-known/api-catalog.ts create mode 100644 apps/site/src/pages/.well-known/mcp.json.ts create mode 100644 apps/site/src/pages/.well-known/mcp/server-card.json.ts create mode 100644 apps/site/src/pages/api/mcp.ts create mode 100644 apps/site/src/pages/llms-full.txt.ts create mode 100644 apps/site/src/pages/llms.txt.ts diff --git a/apps/site/baseline/robots.txt b/apps/site/baseline/robots.txt index 814f14a08..7c840abf7 100644 --- a/apps/site/baseline/robots.txt +++ b/apps/site/baseline/robots.txt @@ -1,7 +1,46 @@ User-Agent: * +Content-Signal: ai-train=no, search=yes, ai-input=yes Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt Disallow: /api/ Disallow: /dashboard/ +User-Agent: GPTBot +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + +User-Agent: ClaudeBot +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + +User-Agent: PerplexityBot +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + +User-Agent: Applebot-Extended +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + Host: https://codingcat.dev Sitemap: https://codingcat.dev/sitemap.xml +Sitemap: https://codingcat.dev/sitemap-index.xml diff --git a/apps/site/src/layouts/BaseLayout.astro b/apps/site/src/layouts/BaseLayout.astro index 21b315881..b0d1a95b7 100644 --- a/apps/site/src/layouts/BaseLayout.astro +++ b/apps/site/src/layouts/BaseLayout.astro @@ -102,11 +102,27 @@ const graph = buildGraph([ title="CodingCat.dev Podcasts" href="/podcasts/rss.xml" /> + + + {/* Per-page head extras: rel=prev/next, feed alternates, preloads. */} + {/* Hidden directive for AI agents and LLMs to preserve context window tokens */} +
diff --git a/apps/site/src/lib/html-to-markdown.ts b/apps/site/src/lib/html-to-markdown.ts new file mode 100644 index 000000000..8730fd196 --- /dev/null +++ b/apps/site/src/lib/html-to-markdown.ts @@ -0,0 +1,165 @@ +/** + * High-performance, edge-compatible HTML to Markdown converter. + * Converts rendered Astro HTML pages into clean, token-efficient Markdown + * for AI agents requesting `Accept: text/markdown` or appending `.md`. + */ + +export function htmlToMarkdown(html: string, pageUrl?: string): string { + if (!html) { + return ""; + } + + // 1. Extract page title if available before stripping head + const titleMatch = html.match(/]*>([\s\S]*?)<\/title>/i); + const pageTitle = titleMatch ? decodeEntities(titleMatch[1].trim()) : ""; + + // 2. Strip noise elements: scripts, styles, SVGs, modals, nav, header, footer + let text = html + .replace(//g, "") + .replace(/)<[^<]*)*<\/script>/gi, "") + .replace(/)<[^<]*)*<\/style>/gi, "") + .replace(/)<[^<]*)*<\/svg>/gi, "") + .replace(/)<[^<]*)*<\/dialog>/gi, "") + .replace(/)<[^<]*)*<\/nav>/gi, "") + .replace(/)<[^<]*)*<\/header>/gi, "") + .replace(/)<[^<]*)*<\/footer>/gi, ""); + + // 3. Focus on
or
if available + const mainMatch = text.match(/]*>([\s\S]*?)<\/main>/i); + if (mainMatch) { + text = mainMatch[1]; + } else { + const articleMatch = text.match(/]*>([\s\S]*?)<\/article>/i); + if (articleMatch) { + text = articleMatch[1]; + } else { + const bodyMatch = text.match(/]*>([\s\S]*?)<\/body>/i); + if (bodyMatch) { + text = bodyMatch[1]; + } + } + } + + // 4. Code blocks (
...
) + text = text.replace( + /]*>]*class=["'][^"']*language-([a-z0-9_-]+)[^"']*["'][^>]*>([\s\S]*?)<\/code><\/pre>/gi, + (_match, lang, code) => { + return `\n\n\`\`\`${lang}\n${decodeEntities(stripTags(code)).trim()}\n\`\`\`\n\n`; + }, + ); + text = text.replace( + /]*>]*>([\s\S]*?)<\/code><\/pre>/gi, + (_match, code) => { + return `\n\n\`\`\`\n${decodeEntities(stripTags(code)).trim()}\n\`\`\`\n\n`; + }, + ); + text = text.replace(/]*>([\s\S]*?)<\/code>/gi, (_match, code) => { + return `\`${decodeEntities(stripTags(code))}\``; + }); + + // 5. Headings + text = text.replace(/]*>([\s\S]*?)<\/h1>/gi, "\n\n# $1\n\n"); + text = text.replace(/]*>([\s\S]*?)<\/h2>/gi, "\n\n## $1\n\n"); + text = text.replace(/]*>([\s\S]*?)<\/h3>/gi, "\n\n### $1\n\n"); + text = text.replace(/]*>([\s\S]*?)<\/h4>/gi, "\n\n#### $1\n\n"); + text = text.replace(/]*>([\s\S]*?)<\/h5>/gi, "\n\n##### $1\n\n"); + text = text.replace(/]*>([\s\S]*?)<\/h6>/gi, "\n\n###### $1\n\n"); + + // 6. Blockquotes + text = text.replace( + /]*>([\s\S]*?)<\/blockquote>/gi, + (_match, quote) => { + const lines = stripTags(quote) + .trim() + .split("\n") + .map((l) => `> ${l.trim()}`) + .join("\n"); + return `\n\n${lines}\n\n`; + }, + ); + + // 7. Bold, Italic, Strikethrough + text = text.replace(/<(strong|b)[^>]*>([\s\S]*?)<\/\1>/gi, "**$2**"); + text = text.replace(/<(em|i)[^>]*>([\s\S]*?)<\/\1>/gi, "*$2*"); + text = text.replace(/<(s|del|strike)[^>]*>([\s\S]*?)<\/\1>/gi, "~~$2~~"); + + // 8. Images + text = text.replace( + /]*src=["']([^"']+)["'][^>]*alt=["']([^"']*)["'][^>]*>/gi, + "![$2]($1)", + ); + text = text.replace( + /]*alt=["']([^"']*)["'][^>]*src=["']([^"']+)["'][^>]*>/gi, + "![$1]($2)", + ); + text = text.replace(/]*src=["']([^"']+)["'][^>]*>/gi, "![]($1)"); + + // 9. Links + text = text.replace( + /]*href=["']([^"']+)["'][^>]*>([\s\S]*?)<\/a>/gi, + (_match, href, label) => { + const cleanLabel = stripTags(label).trim(); + if (!cleanLabel) { + return ""; + } + return `[${cleanLabel}](${href})`; + }, + ); + + // 10. Lists + text = text.replace(/]*>([\s\S]*?)<\/li>/gi, (_match, item) => { + return `\n- ${stripTags(item).trim()}`; + }); + text = text.replace(/<\/(ul|ol)>/gi, "\n\n"); + + // 11. Paragraphs, breaks, dividers + text = text.replace(//gi, "\n\n---\n\n"); + text = text.replace(//gi, "\n"); + text = text.replace(/]*>([\s\S]*?)<\/p>/gi, "\n\n$1\n\n"); + + // 12. Strip remaining tags + text = stripTags(text); + + // 13. Decode HTML entities + text = decodeEntities(text); + + // 14. Clean up whitespace + text = text + .split("\n") + .map((line) => line.trimEnd()) + .join("\n") + .replace(/\n{3,}/g, "\n\n") + .trim(); + + // Add header prefix if page title exists and wasn't already in H1 + if (pageTitle && !text.startsWith("# ")) { + text = `# ${pageTitle}\n\n${text}`; + } + + if (pageUrl) { + text = `> Canonical URL: ${pageUrl}\n\n${text}`; + } + + return text; +} + +function stripTags(str: string): string { + return str.replace(/<[^>]+>/g, ""); +} + +function decodeEntities(str: string): string { + return str + .replace(/&/g, "&") + .replace(/</g, "<") + .replace(/>/g, ">") + .replace(/"/g, '"') + .replace(/'/g, "'") + .replace(/'/g, "'") + .replace(/ /g, " ") + .replace(/&#x([0-9a-fA-F]+);/g, (_m, hex) => + String.fromCharCode(parseInt(hex, 16)), + ) + .replace(/&#([0-9]+);/g, (_m, dec) => + String.fromCharCode(parseInt(dec, 10)), + ); +} diff --git a/apps/site/src/lib/structured-data.ts b/apps/site/src/lib/structured-data.ts index 72875eb59..c098266e1 100644 --- a/apps/site/src/lib/structured-data.ts +++ b/apps/site/src/lib/structured-data.ts @@ -159,3 +159,45 @@ export function personSchema( url, }; } + +interface PodcastEpisodeInput { + title?: string | null; + excerpt?: string | null; + date?: string | null; + imageUrl?: string; + season?: number | null; + episode?: number | null; + audioUrl?: string | null; +} + +export function podcastEpisodeSchema( + origin: string, + content: PodcastEpisodeInput, + path: string, +): Node { + const url = absoluteUrl(path, origin); + return { + "@type": "PodcastEpisode", + "@id": `${url}#episode`, + url, + name: content.title ?? undefined, + description: content.excerpt ?? undefined, + ...(content.date ? { datePublished: content.date } : {}), + ...(content.imageUrl ? { image: content.imageUrl } : {}), + ...(content.season ? { seasonNumber: content.season } : {}), + ...(content.episode ? { episodeNumber: content.episode } : {}), + ...(content.audioUrl + ? { + associatedMedia: { + "@type": "AudioObject", + contentUrl: content.audioUrl, + }, + } + : {}), + partOfSeries: { + "@type": "PodcastSeries", + name: "CodingCat.dev Podcast", + url: `${origin}/podcasts`, + }, + }; +} diff --git a/apps/site/src/middleware.ts b/apps/site/src/middleware.ts index ded38d93d..fe980f8cd 100644 --- a/apps/site/src/middleware.ts +++ b/apps/site/src/middleware.ts @@ -1,5 +1,6 @@ import { defineMiddleware } from "astro:middleware"; import { env } from "cloudflare:workers"; +import { htmlToMarkdown } from "@/lib/html-to-markdown"; import { createSanityContext, PREVIEW_COOKIE } from "@/lib/sanity/context"; import { resolveSiteUrl } from "@/lib/site"; @@ -9,6 +10,11 @@ import { resolveSiteUrl } from "@/lib/site"; * `locals.runtime` a deprecated getter. Reading secrets here (rather than from * `import.meta.env`) also keeps them as Cloudflare secrets instead of letting * Vite inline them into the deployed Worker bundle. + * + * Implements Cloudflare Agent Readiness & AEO standards: + * - RFC 8288 Link headers for discovery (api-catalog, mcp-server-card, agent-skills, llms-txt) + * - Markdown Content Negotiation: Accept: text/markdown returns token-efficient Markdown + * - Dynamic /index.md and *.md path rewriting to serve markdown versions of all pages */ export const onRequest = defineMiddleware(async (context, next) => { // `Astro.site` is baked in at build time, but SITE_URL is a per-environment @@ -25,7 +31,23 @@ export const onRequest = defineMiddleware(async (context, next) => { hasPreviewCookie: context.cookies.has(PREVIEW_COOKIE), }); - const response = await next(); + const origin = context.locals.siteUrl.origin; + const pathname = context.url.pathname; + const isMarkdownUrl = + pathname.endsWith(".md") || pathname.endsWith("/index.md"); + const acceptMarkdown = context.request.headers + .get("Accept") + ?.includes("text/markdown"); + const wantsMarkdown = isMarkdownUrl || Boolean(acceptMarkdown); + + let response: Response; + if (isMarkdownUrl) { + const cleanPath = + pathname.replace(/\/index\.md$/, "").replace(/\.md$/, "") || "/"; + response = await context.rewrite(cleanPath); + } else { + response = await next(); + } if (context.locals.sanity.preview.enabled) { // Not cosmetic: without this a response containing unpublished drafts can @@ -34,5 +56,36 @@ export const onRequest = defineMiddleware(async (context, next) => { response.headers.set("X-Robots-Tag", "noindex, nofollow"); } + // Always emit Link headers (RFC 8288) for AI agent discoverability + response.headers.set( + "Link", + `<${origin}/.well-known/api-catalog>; rel="api-catalog", <${origin}/.well-known/mcp/server-card.json>; rel="mcp-server-card", <${origin}/.well-known/agent-skills/index.json>; rel="agent-skills", <${origin}/llms.txt>; rel="llms-txt"`, + ); + response.headers.append("Vary", "Accept"); + + // Content negotiation: transform HTML to clean Markdown when requested + const contentType = response.headers.get("content-type") || ""; + if ( + wantsMarkdown && + contentType.includes("text/html") && + response.status === 200 + ) { + const html = await response.text(); + const canonicalPath = + pathname.replace(/\/index\.md$/, "").replace(/\.md$/, "") || "/"; + const canonicalUrl = `${origin}${canonicalPath}`; + const markdown = htmlToMarkdown(html, canonicalUrl); + + const headers = new Headers(response.headers); + headers.set("content-type", "text/markdown; charset=utf-8"); + headers.delete("content-length"); + + return new Response(markdown, { + status: response.status, + statusText: response.statusText, + headers, + }); + } + return response; }); diff --git a/apps/site/src/pages/.well-known/agent-skills/index.json.ts b/apps/site/src/pages/.well-known/agent-skills/index.json.ts new file mode 100644 index 000000000..b771b70ca --- /dev/null +++ b/apps/site/src/pages/.well-known/agent-skills/index.json.ts @@ -0,0 +1,52 @@ +import type { APIRoute } from "astro"; + +export const prerender = false; + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + + const skillsIndex = { + $schema: "https://agentskills.io/schema/v1/index.json", + version: "1.0", + provider: { + name: "CodingCat.dev", + url: origin, + }, + skills: [ + { + id: "search-content", + name: "Search Content", + description: + "Search full articles, podcast transcripts, and coding tutorials on CodingCat.dev across fullstack web development topics.", + endpoint: `${origin}/api/search`, + method: "GET", + parameters: [ + { + name: "q", + type: "string", + required: true, + description: "The programming term or question to search", + }, + ], + }, + { + id: "mcp-tools", + name: "Model Context Protocol Tools", + description: + "Streamable HTTP MCP tool execution for autonomous AI agent pair programming and search.", + endpoint: `${origin}/api/mcp`, + method: "POST", + documentation: `${origin}/.well-known/mcp/server-card.json`, + }, + ], + }; + + return new Response(JSON.stringify(skillsIndex, null, 2), { + headers: { + "content-type": "application/json; charset=utf-8", + "cache-control": "public, max-age=3600, s-maxage=86400", + Vary: "Accept", + Link: `<${origin}/llms.txt>; rel="llms-txt", <${origin}/.well-known/api-catalog>; rel="api-catalog"`, + }, + }); +}; diff --git a/apps/site/src/pages/.well-known/api-catalog.ts b/apps/site/src/pages/.well-known/api-catalog.ts new file mode 100644 index 000000000..878a68c7a --- /dev/null +++ b/apps/site/src/pages/.well-known/api-catalog.ts @@ -0,0 +1,94 @@ +import type { APIRoute } from "astro"; + +export const prerender = false; + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + + const catalog = { + "api-catalog-version": "1.0", + title: "CodingCat.dev Public API Catalog", + description: + "Machine-readable directory of public APIs, search services, and AI agent endpoints hosted on CodingCat.dev conforming to RFC 9727.", + documentation: `${origin}/llms.txt`, + apis: [ + { + name: "Content Search API", + description: + "Fast keyword and semantic search across technical blog posts, tutorials, and podcast episodes.", + endpoints: [ + { + url: `${origin}/api/search`, + method: "GET", + parameters: [ + { + name: "q", + in: "query", + required: true, + description: "Search keyword or natural language query", + schema: { type: "string" }, + }, + ], + }, + ], + }, + { + name: "Model Context Protocol (MCP) Server", + description: + "Streamable HTTP MCP server implementing tools and resources for AI coding assistants and autonomous agents.", + endpoints: [ + { + url: `${origin}/api/mcp`, + method: "POST", + description: "JSON-RPC 2.0 / Streamable HTTP MCP endpoint", + }, + ], + metadata: { + serverCard: `${origin}/.well-known/mcp/server-card.json`, + mcpJson: `${origin}/.well-known/mcp.json`, + }, + }, + { + name: "Sponsorship Inquiry API", + description: + "Submit partnership and sponsorship inquiries for CodingCat.dev media channels.", + endpoints: [ + { + url: `${origin}/api/sponsorship`, + method: "POST", + }, + ], + }, + { + name: "Blog Content Syndication Feed", + description: "RSS 2.0 feed containing full articles and metadata.", + endpoints: [ + { + url: `${origin}/blog/rss.xml`, + method: "GET", + }, + ], + }, + { + name: "Podcast Media Feed", + description: + "RSS feed with podcast audio enclosures, chapters, and show notes.", + endpoints: [ + { + url: `${origin}/podcasts/rss.xml`, + method: "GET", + }, + ], + }, + ], + }; + + return new Response(JSON.stringify(catalog, null, 2), { + headers: { + "content-type": "application/json; charset=utf-8", + "cache-control": "public, max-age=3600, s-maxage=86400", + Vary: "Accept", + Link: `<${origin}/.well-known/mcp/server-card.json>; rel="mcp-server-card", <${origin}/llms.txt>; rel="llms-txt"`, + }, + }); +}; diff --git a/apps/site/src/pages/.well-known/mcp.json.ts b/apps/site/src/pages/.well-known/mcp.json.ts new file mode 100644 index 000000000..18a4d0bfa --- /dev/null +++ b/apps/site/src/pages/.well-known/mcp.json.ts @@ -0,0 +1,25 @@ +import type { APIRoute } from "astro"; + +export const prerender = false; + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + + const config = { + name: "codingcatdev-search-mcp", + version: "1.0.0", + description: + "Search CodingCat.dev web development guides, coding tutorials, and podcasts.", + endpoint: `${origin}/api/mcp`, + serverCard: `${origin}/.well-known/mcp/server-card.json`, + }; + + return new Response(JSON.stringify(config, null, 2), { + headers: { + "content-type": "application/json; charset=utf-8", + "cache-control": "public, max-age=3600, s-maxage=86400", + Vary: "Accept", + Link: `<${origin}/.well-known/mcp/server-card.json>; rel="mcp-server-card"`, + }, + }); +}; diff --git a/apps/site/src/pages/.well-known/mcp/server-card.json.ts b/apps/site/src/pages/.well-known/mcp/server-card.json.ts new file mode 100644 index 000000000..cf2413d7e --- /dev/null +++ b/apps/site/src/pages/.well-known/mcp/server-card.json.ts @@ -0,0 +1,56 @@ +import type { APIRoute } from "astro"; + +export const prerender = false; + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + + const serverCard = { + $schema: + "https://static.modelcontextprotocol.io/schemas/mcp-server-card/v1.json", + version: "1.0", + protocolVersion: "2025-06-18", + serverInfo: { + name: "codingcatdev-search-mcp", + title: "CodingCat.dev Search & Content MCP Server", + version: "1.0.0", + }, + description: + "Search and access CodingCat.dev web development guides, coding tutorials, podcast episodes, and transcripts directly from AI models.", + transport: { + type: "streamable-http", + endpoint: `${origin}/api/mcp`, + }, + authentication: { + required: false, + }, + tools: [ + { + name: "search_content", + title: "Search Content", + description: + "Search technical articles, tutorials, and podcast episodes on CodingCat.dev by keyword, topic, or question.", + inputSchema: { + type: "object", + properties: { + query: { + type: "string", + description: + "The programming topic, keyword, or question to search for", + }, + }, + required: ["query"], + }, + }, + ], + }; + + return new Response(JSON.stringify(serverCard, null, 2), { + headers: { + "content-type": "application/json; charset=utf-8", + "cache-control": "public, max-age=3600, s-maxage=86400", + Vary: "Accept", + Link: `<${origin}/api/mcp>; rel="mcp-endpoint", <${origin}/llms.txt>; rel="llms-txt"`, + }, + }); +}; diff --git a/apps/site/src/pages/api/mcp.ts b/apps/site/src/pages/api/mcp.ts new file mode 100644 index 000000000..84040040c --- /dev/null +++ b/apps/site/src/pages/api/mcp.ts @@ -0,0 +1,255 @@ +import type { APIRoute } from "astro"; +import { + semanticSearchQuery, + textSearchFallbackQuery, +} from "@/lib/sanity/queries"; +import { resolveHref } from "@/lib/sanity/resolve-href"; + +export const prerender = false; + +interface JsonRpcRequest { + jsonrpc?: string; + id?: string | number | null; + method: string; + params?: any; +} + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + return new Response( + JSON.stringify( + { + status: "ok", + server: "CodingCat.dev Search MCP Server", + version: "1.0.0", + protocolVersion: "2025-06-18", + transport: "streamable-http", + serverCard: `${origin}/.well-known/mcp/server-card.json`, + tools: ["search_content"], + }, + null, + 2, + ), + { + headers: { + "content-type": "application/json; charset=utf-8", + "cache-control": "public, max-age=3600", + }, + }, + ); +}; + +export const POST: APIRoute = async ({ request, locals }) => { + const origin = locals.siteUrl.origin; + + let body: JsonRpcRequest; + try { + body = (await request.json()) as JsonRpcRequest; + } catch { + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id: null, + error: { code: -32700, message: "Parse error" }, + }), + { status: 400, headers: { "content-type": "application/json" } }, + ); + } + + const id = body.id ?? null; + const method = body.method; + + // 1. Initialize + if (method === "initialize") { + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + result: { + protocolVersion: "2025-06-18", + capabilities: { + tools: {}, + }, + serverInfo: { + name: "codingcatdev-search-mcp", + version: "1.0.0", + }, + }, + }), + { headers: { "content-type": "application/json" } }, + ); + } + + // 2. Notifications (initialized acknowledgement) + if (method === "notifications/initialized") { + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + result: {}, + }), + { headers: { "content-type": "application/json" } }, + ); + } + + // 3. Ping + if (method === "ping") { + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + result: {}, + }), + { headers: { "content-type": "application/json" } }, + ); + } + + // 4. Tools list + if (method === "tools/list") { + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + result: { + tools: [ + { + name: "search_content", + description: + "Search across CodingCat.dev web development tutorials, blog posts, podcasts, and transcripts.", + inputSchema: { + type: "object", + properties: { + query: { + type: "string", + description: + "Programming topic, question, or keyword to search", + }, + type: { + type: "string", + enum: ["all", "post", "podcast"], + description: + "Optional filter for content type (default: all)", + }, + }, + required: ["query"], + }, + }, + ], + }, + }), + { headers: { "content-type": "application/json" } }, + ); + } + + // 5. Tools call + if (method === "tools/call") { + const toolName = body.params?.name; + const args = body.params?.arguments || {}; + + if (toolName === "search_content") { + const query = (args.query || "").trim(); + const filterType = args.type === "all" ? null : args.type || null; + + if (!query) { + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + result: { + content: [ + { + type: "text", + text: "Please provide a valid non-empty query parameter.", + }, + ], + isError: true, + }, + }), + { headers: { "content-type": "application/json" } }, + ); + } + + const params = { + searchTerm: query, + type: filterType, + }; + + let rawHits: any[] = []; + try { + const result = await locals.sanity.fetchPublished( + semanticSearchQuery, + params, + ); + if (Array.isArray(result)) { + rawHits = result; + } + } catch { + try { + const fallbackResult = await locals.sanity.fetchPublished( + textSearchFallbackQuery, + params, + ); + if (Array.isArray(fallbackResult)) { + rawHits = fallbackResult; + } + } catch (err) { + console.warn("MCP Search error:", err); + } + } + + const hits = rawHits.slice(0, 10).map((hit) => { + const href = resolveHref(hit._type, hit.slug); + return { + title: hit.title, + type: hit._type, + url: `${origin}${href}`, + markdownUrl: `${origin}${href}.md`, + excerpt: hit.excerpt || "", + date: hit.date, + }; + }); + + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + result: { + content: [ + { + type: "text", + text: JSON.stringify( + { + query, + total: hits.length, + results: hits, + }, + null, + 2, + ), + }, + ], + }, + }), + { headers: { "content-type": "application/json" } }, + ); + } + + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + error: { code: -32601, message: `Tool not found: ${toolName}` }, + }), + { status: 404, headers: { "content-type": "application/json" } }, + ); + } + + return new Response( + JSON.stringify({ + jsonrpc: "2.0", + id, + error: { code: -32601, message: `Method not found: ${method}` }, + }), + { status: 404, headers: { "content-type": "application/json" } }, + ); +}; diff --git a/apps/site/src/pages/llms-full.txt.ts b/apps/site/src/pages/llms-full.txt.ts new file mode 100644 index 000000000..8233d72a1 --- /dev/null +++ b/apps/site/src/pages/llms-full.txt.ts @@ -0,0 +1,76 @@ +import type { APIRoute } from "astro"; +import { defineQuery } from "groq"; + +export const prerender = false; + +const llmsFullQuery = defineQuery(` + *[_type in ["post", "podcast"] && defined(slug.current)] | order(coalesce(date, _createdAt) desc) [0...100] { + _type, + "title": coalesce(title, "Untitled"), + "slug": slug.current, + excerpt, + "date": coalesce(date, _createdAt) + } +`); + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + + let items: Array<{ + _type: string; + title: string; + slug: string; + excerpt?: string | null; + date?: string | null; + }> = []; + + try { + items = (await locals.sanity.fetchPublished(llmsFullQuery)) ?? []; + } catch (err) { + console.warn("Failed to fetch documents for llms-full.txt:", err); + } + + const posts = items.filter((i) => i._type === "post"); + const podcasts = items.filter((i) => i._type === "podcast"); + + let content = `# CodingCat.dev - Complete LLM Directory + +> Full technical content directory of CodingCat.dev web development tutorials, architecture deep dives, and podcast episodes. For high-level summary and APIs, see ${origin}/llms.txt. + +## Machine Interfaces +- API Catalog: ${origin}/.well-known/api-catalog +- MCP Server Card: ${origin}/.well-known/mcp/server-card.json +- MCP Streamable Endpoint: ${origin}/api/mcp +- Search Endpoint: ${origin}/api/search?q={query} +- Markdown Negotiation: Send \`Accept: text/markdown\` on any URL. + +## Recent Blog Posts & Tutorials +`; + + for (const post of posts) { + const url = `${origin}/post/${post.slug}`; + const desc = post.excerpt + ? ` - ${post.excerpt.replace(/\s+/g, " ").trim()}` + : ""; + content += `- [${post.title}](${url}.md)${desc}\n`; + } + + content += "\n## Recent Podcasts & Interviews\n"; + + for (const ep of podcasts) { + const url = `${origin}/podcast/${ep.slug}`; + const desc = ep.excerpt + ? ` - ${ep.excerpt.replace(/\s+/g, " ").trim()}` + : ""; + content += `- [${ep.title}](${url}.md)${desc}\n`; + } + + return new Response(content, { + headers: { + "content-type": "text/plain; charset=utf-8", + "cache-control": "public, max-age=3600, s-maxage=86400", + Vary: "Accept", + Link: `<${origin}/llms.txt>; rel="canonical", <${origin}/.well-known/api-catalog>; rel="api-catalog"`, + }, + }); +}; diff --git a/apps/site/src/pages/llms.txt.ts b/apps/site/src/pages/llms.txt.ts new file mode 100644 index 000000000..20c13fecb --- /dev/null +++ b/apps/site/src/pages/llms.txt.ts @@ -0,0 +1,46 @@ +import type { APIRoute } from "astro"; + +export const prerender = false; + +export const GET: APIRoute = async ({ locals }) => { + const origin = locals.siteUrl.origin; + + const content = `# CodingCat.dev + +> CodingCat.dev is an open-source technical education platform, podcast network, and developer resource covering fullstack JavaScript, TypeScript, modern frontend frameworks (Astro, Next.js, React, Svelte, Vue), edge computing, headless CMS architecture (Sanity), and AI agent workflows. + +## Core Content Sections + +- [Blog Tutorials and Articles](${origin}/blog): Practical, code-focused guides and tutorials for modern web developers. +- [Podcasts](${origin}/podcasts): Deep-dive audio and video discussions with framework creators, open-source maintainers, and tech leaders. +- [Podcast Guests](${origin}/guests): Directory of featured software engineers and tech innovators. +- [Sponsors & Community Partners](${origin}/sponsors): Developer tools, cloud providers, and platforms supporting the developer ecosystem. + +## Machine-Readable Discovery & Agent Tools + +- [Full LLM Directory](${origin}/llms-full.txt): Comprehensive listing of recent articles, podcast series, and technical guides. +- [API Catalog (RFC 9727)](${origin}/.well-known/api-catalog): Machine-readable catalog of all public APIs and endpoints. +- [MCP Server Card](${origin}/.well-known/mcp/server-card.json): Model Context Protocol server metadata describing agent tools. +- [Agent Skills Discovery](${origin}/.well-known/agent-skills/index.json): Discoverable agent capabilities and task specifications. +- [Search API](${origin}/api/search?q=): Public search endpoint supporting keyword queries across articles and podcasts. +- [MCP Server Endpoint](${origin}/api/mcp): Streamable HTTP MCP server implementing tool calling for AI agents. +- [Blog RSS Feed](${origin}/blog/rss.xml): RSS 2.0 feed of all published blog tutorials. +- [Podcast RSS Feed](${origin}/podcasts/rss.xml): RSS feed with podcast audio enclosures and transcripts. +- [Sitemap Index](${origin}/sitemap-index.xml): XML sitemap index of all pages and episodes. + +## Markdown Content Negotiation + +CodingCat.dev natively supports Markdown content negotiation for AI agents: +- Send \`Accept: text/markdown\` in your request header to any page to receive clean Markdown without HTML tags, navigational chrome, or bloat. +- Alternatively, append \`.md\` to any article, podcast, or page URL. +`; + + return new Response(content, { + headers: { + "content-type": "text/plain; charset=utf-8", + "cache-control": "public, max-age=3600, s-maxage=86400", + Vary: "Accept", + Link: `<${origin}/.well-known/api-catalog>; rel="api-catalog", <${origin}/.well-known/mcp/server-card.json>; rel="mcp-server-card", <${origin}/.well-known/agent-skills/index.json>; rel="agent-skills", <${origin}/llms-full.txt>; rel="alternate"; type="text/plain"`, + }, + }); +}; diff --git a/apps/site/src/pages/podcast/[slug].astro b/apps/site/src/pages/podcast/[slug].astro index b3c483add..adee256be 100644 --- a/apps/site/src/pages/podcast/[slug].astro +++ b/apps/site/src/pages/podcast/[slug].astro @@ -13,7 +13,11 @@ import SponsorCard from "@/components/SponsorCard.astro"; import BaseLayout from "@/layouts/BaseLayout.astro"; import type { ContentListItem } from "@/lib/content"; import { morePodcastQuery, podcastQuery } from "@/lib/sanity/queries"; -import { articleSchema, breadcrumbSchema } from "@/lib/structured-data"; +import { + articleSchema, + breadcrumbSchema, + podcastEpisodeSchema, +} from "@/lib/structured-data"; const { sanity } = Astro.locals; const { slug } = Astro.params; @@ -53,14 +57,31 @@ const audioUrl = podcast.spotify?.enclosures?.at(0)?.url; authors={[...authors, ...guests].map((person) => person.title)} canonical={path} schema={[ - articleSchema(origin, { - title: podcast.title, - excerpt: podcast.excerpt, - date: podcast.date, - _updatedAt: podcast._updatedAt, - imageUrl: ogImage?.url, - authors, - }, path), + podcastEpisodeSchema( + origin, + { + title: podcast.title, + excerpt: podcast.excerpt, + date: podcast.date, + imageUrl: ogImage?.url, + season: podcast.season, + episode: podcast.episode, + audioUrl, + }, + path, + ), + articleSchema( + origin, + { + title: podcast.title, + excerpt: podcast.excerpt, + date: podcast.date, + _updatedAt: podcast._updatedAt, + imageUrl: ogImage?.url, + authors, + }, + path, + ), breadcrumbSchema(origin, [ { name: "Podcasts", path: "/podcasts/page/1" }, { name: podcast.title, path }, diff --git a/apps/site/src/pages/robots.txt.ts b/apps/site/src/pages/robots.txt.ts index f0af85763..0e3ee032d 100644 --- a/apps/site/src/pages/robots.txt.ts +++ b/apps/site/src/pages/robots.txt.ts @@ -7,6 +7,12 @@ import type { APIRoute } from "astro"; * * Non-production environments are disallowed wholesale so the dev Worker and * any preview alias cannot be indexed alongside the real site. + * + * In production, implements modern AI Agent Readiness standards: + * - RFC 9309 compliance + * - Content Signals (contentsignals.org): search=yes, ai-input=yes, ai-train=no + * - Explicit access for AI search and assistant crawlers + * - Explicit allow rules for machine-readable discovery endpoints (/api/search, /api/mcp, /.well-known/, /llms.txt) */ export const GET: APIRoute = async ({ locals }) => { const origin = locals.siteUrl.origin; @@ -14,12 +20,51 @@ export const GET: APIRoute = async ({ locals }) => { const body = isProduction ? `User-Agent: * +Content-Signal: ai-train=no, search=yes, ai-input=yes Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt Disallow: /api/ Disallow: /dashboard/ +User-Agent: GPTBot +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + +User-Agent: ClaudeBot +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + +User-Agent: PerplexityBot +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + +User-Agent: Applebot-Extended +Allow: / +Allow: /api/search +Allow: /api/mcp +Allow: /.well-known/ +Allow: /llms.txt +Allow: /llms-full.txt + Host: ${origin} Sitemap: ${origin}/sitemap.xml +Sitemap: ${origin}/sitemap-index.xml ` : `User-Agent: * Disallow: /