Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions apps/web/src/lib/gate.js
Original file line number Diff line number Diff line change
Expand Up @@ -77,6 +77,14 @@ export const throttle = createThrottle({

/** The gate as one Hono middleware: crawlers, then the site-wide allowance. */
export async function gate(c, next) {
// GoogleOther (Google's non-Search crawler) walked every filter combination
// of /servers and /api/v1/search at ~120/min, past its allowance and into
// the 402 page. robots.txt refuses it, but Google caches robots.txt for up to
// a day, so answer it here too: a few bytes, no data. robots.txt stays open,
// or it could never learn it was refused. Googlebot itself is untouched.
if (/GoogleOther/i.test(c.req.header('user-agent') ?? '') && c.req.path !== '/robots.txt') {
return c.text('GoogleOther is refused here; see /robots.txt\n', 403);
}
const answer = await gateway.handle(c.req.raw);
if (answer) return answer;
const over = await throttle.handle(c.req.raw);
Expand Down
11 changes: 11 additions & 0 deletions test/api.test.js
Original file line number Diff line number Diff line change
Expand Up @@ -308,6 +308,17 @@ if (!url) {
expect(txt).toContain('User-agent: GoogleOther\nDisallow: /\n');
expect(txt).toContain('User-agent: *\nAllow: /');
});
test('GoogleOther gets a bare 403 but can still read robots.txt; Googlebot is served', async () => {
const ua = (bot) => ({
headers: { 'user-agent': `Mozilla/5.0 (compatible; ${bot})`, 'x-real-ip': '203.0.113.20' },
});
const refused = await app.request('/api/v1/search?kind=vps', ua('GoogleOther'));
expect(refused.status).toBe(403);
expect((await refused.text()).length).toBeLessThan(100);
expect((await app.request('/servers', ua('GoogleOther'))).status).toBe(403);
expect((await app.request('/robots.txt', ua('GoogleOther'))).status).toBe(200);
expect((await app.request('/servers', ua('Googlebot/2.1'))).status).toBe(200);
});
test('training crawlers get 402, readers do not', async () => {
expect(
(
Expand Down
Loading