From 1243e52585552d50f8312eb7e029b69f3e23efd4 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 25 Sep 2026 16:11:12 +0000 Subject: [PATCH] Refuse GoogleOther with a bare 403 robots.txt refuses it since #5, but Google caches robots.txt for up to a day, and meanwhile it walks every filter combination of /servers and /api/v1/search at ~120/min. Answer it in the gate with a few bytes and no data. robots.txt stays readable so it can learn the refusal; Googlebot is untouched. Co-Authored-By: Claude Opus 5.5 (1M context) --- apps/web/src/lib/gate.js | 8 ++++++++ test/api.test.js | 11 +++++++++++ 2 files changed, 19 insertions(+) diff --git a/apps/web/src/lib/gate.js b/apps/web/src/lib/gate.js index 2ccf90e..140b0b2 100644 --- a/apps/web/src/lib/gate.js +++ b/apps/web/src/lib/gate.js @@ -77,6 +77,14 @@ export const throttle = createThrottle({ /** The gate as one Hono middleware: crawlers, then the site-wide allowance. */ export async function gate(c, next) { + // GoogleOther (Google's non-Search crawler) walked every filter combination + // of /servers and /api/v1/search at ~120/min, past its allowance and into + // the 402 page. robots.txt refuses it, but Google caches robots.txt for up to + // a day, so answer it here too: a few bytes, no data. robots.txt stays open, + // or it could never learn it was refused. Googlebot itself is untouched. + if (/GoogleOther/i.test(c.req.header('user-agent') ?? '') && c.req.path !== '/robots.txt') { + return c.text('GoogleOther is refused here; see /robots.txt\n', 403); + } const answer = await gateway.handle(c.req.raw); if (answer) return answer; const over = await throttle.handle(c.req.raw); diff --git a/test/api.test.js b/test/api.test.js index f65ee35..b1dfdea 100644 --- a/test/api.test.js +++ b/test/api.test.js @@ -308,6 +308,17 @@ if (!url) { expect(txt).toContain('User-agent: GoogleOther\nDisallow: /\n'); expect(txt).toContain('User-agent: *\nAllow: /'); }); + test('GoogleOther gets a bare 403 but can still read robots.txt; Googlebot is served', async () => { + const ua = (bot) => ({ + headers: { 'user-agent': `Mozilla/5.0 (compatible; ${bot})`, 'x-real-ip': '203.0.113.20' }, + }); + const refused = await app.request('/api/v1/search?kind=vps', ua('GoogleOther')); + expect(refused.status).toBe(403); + expect((await refused.text()).length).toBeLessThan(100); + expect((await app.request('/servers', ua('GoogleOther'))).status).toBe(403); + expect((await app.request('/robots.txt', ua('GoogleOther'))).status).toBe(200); + expect((await app.request('/servers', ua('Googlebot/2.1'))).status).toBe(200); + }); test('training crawlers get 402, readers do not', async () => { expect( (