Charge AI training crawlers for access (x402 gateway) (#141)

* Charge AI training crawlers for access (@profullstack/x402-gateway)

Training crawlers (GPTBot, ClaudeBot, CCBot, meta-externalagent, Bytespider,
Applebot-Extended) get 402 Payment Required with an x402 offer, or the sales
page at /crawl, and a paid pass opens the site for a day. People, search
engines and retrieval crawlers pass through untouched. robots.txt is now
generated from the same lists.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01YafYxayh7Gqe5MWNNQMev2

* Type the middleware as returning Response | NextResponse

The crawl gateway answers with a plain Fetch Response.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01YafYxayh7Gqe5MWNNQMev2

* Contract test awaits the now-async proxy

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01YafYxayh7Gqe5MWNNQMev2

---------

Co-authored-by: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
Anthony Ettinger 2026-09-05 14:49:39 -07:00 committed by GitHub
parent 80a36269bb
commit 93af4770ac
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 65 additions and 35 deletions

View file

@ -60,21 +60,21 @@ function jsonResponse(body: unknown, status = 200): Response {
}
describe("www canonical redirect (proxy.ts)", () => {
it("301s www to the apex host over https, preserving path + query", () => {
it("301s www to the apex host over https, preserving path + query", async () => {
const request = new NextRequest("https://www.logicsrc.com/openspec?ref=email", {
headers: { host: "www.logicsrc.com" }
});
const response = proxy(request);
const response = await proxy(request);
expect(response.status).toBe(301);
expect(response.headers.get("location")).toBe("https://logicsrc.com/openspec?ref=email");
});
it("passes through apex requests untouched", () => {
it("passes through apex requests untouched", async () => {
const request = new NextRequest("https://logicsrc.com/hire-us", {
headers: { host: "logicsrc.com" }
});
const response = proxy(request);
const response = await proxy(request);
// NextResponse.next() yields a non-redirect response.
expect(response.status).toBe(200);

View file

@ -16,6 +16,7 @@
"@logicsrc/openprd": "file:../../packages/openprd",
"@profullstack/autoblog": "github:profullstack/autoblog#75e54af",
"@profullstack/stack": "^0.1.3",
"@profullstack/x402-gateway": "^0.1.0",
"@supabase/supabase-js": "^2.105.4",
"marked": "^18.0.5",
"next": "16.2.6",

View file

@ -1,30 +0,0 @@
import type { MetadataRoute } from "next";
const SITE_URL = (process.env.PUBLIC_URL ?? "https://logicsrc.com").replace(/\/$/, "");
// Welcome mainstream + AI crawlers; keep them out of the API surface.
export default function robots(): MetadataRoute.Robots {
return {
rules: [
{
userAgent: [
"*",
"GPTBot",
"OAI-SearchBot",
"ChatGPT-User",
"ClaudeBot",
"Claude-Web",
"anthropic-ai",
"PerplexityBot",
"Google-Extended",
"Applebot-Extended",
"CCBot",
],
allow: "/",
disallow: ["/api/", "/health"],
},
],
sitemap: `${SITE_URL}/sitemap.xml`,
host: SITE_URL,
};
}

View file

@ -0,0 +1,9 @@
import { robotsRoute } from "@profullstack/x402-gateway/next";
import { gateway } from "@/lib/crawl-gateway";
// Generated from the same crawler lists the gateway enforces: training
// crawlers are refused everywhere but /crawl (where they can buy a pass),
// retrieval crawlers are named as welcome, everyone else gets the rules below.
export const GET = robotsRoute(gateway, {
disallow: ["/api/", "/health"],
});

View file

@ -0,0 +1,27 @@
import { createGateway } from "@profullstack/x402-gateway";
import { x402Proxy } from "@profullstack/x402-gateway/next";
/**
* Sells crawl access to AI training crawlers (GPTBot, ClaudeBot, CCBot,
* meta-externalagent, Bytespider, Applebot-Extended, ...) by the day over
* x402, settled by CoinPay in USDC. People, Googlebot and the retrieval
* crawlers behind AI search pass through untouched.
*
* Runs inside the middleware, so nothing here may import Node-only modules.
* The env is read through a non-literal key on purpose: Next inlines
* `process.env.NAME` at build time, and these are runtime secrets. Without
* COINPAY_X402_KEY and CRAWL_PAY_TO the gateway still answers training
* crawlers with 402, just with an empty offer.
*/
const env = (name: string) => process.env[name];
export const gateway = createGateway({
siteUrl: env("SITE_URL") || env("NEXT_PUBLIC_SITE_URL") || "https://logicsrc.com",
siteName: "LogicSRC",
coinpay: { apiKey: env("COINPAY_X402_KEY") },
payTo: env("CRAWL_PAY_TO"),
contact: "mailto:support@logicsrc.com",
});
/** Resolves to a Response for a refused crawler, or undefined to carry on. */
export const gate = x402Proxy(gateway);

View file

@ -1,3 +1,4 @@
import { gate } from "@/lib/crawl-gateway";
import { NextResponse } from "next/server";
import type { NextRequest } from "next/server";
@ -6,7 +7,13 @@ import type { NextRequest } from "next/server";
// This is the Next 16 "proxy" (formerly middleware) entrypoint.
const ALLOWED_APEX = process.env.PUBLIC_DOMAIN || "logicsrc.com";
export function proxy(request: NextRequest): NextResponse {
export async function proxy(request: NextRequest): Promise<Response | NextResponse> {
// Crawl gateway first: AI training crawlers get 402 Payment Required (or the
// sales page at /crawl) unless they present a paid pass. People, Googlebot
// and retrieval crawlers fall through to everything below.
const answer = await gate(request);
if (answer) return answer;
const host = request.headers.get("host") ?? "";
if (host === `www.${ALLOWED_APEX}`) {
const { pathname, search } = request.nextUrl;