feat(blocklist): add url-domain host patterns (#3164)

This commit is contained in:
Hampus
2026-10-03 14:46:08 +02:00
committed by GitHub
parent e9167d96ec
commit 7eebfca20b
28 changed files with 1074 additions and 134 deletions
+1
View File
@@ -79,6 +79,7 @@
"sharp": "catalog:",
"stripe": "catalog:",
"tempy": "catalog:",
"tldts": "catalog:",
"transliteration": "catalog:",
"tsx": "catalog:",
"uint8array-extras": "catalog:",
@@ -99,7 +99,8 @@ const BLOCKLIST_CATALOG = [
},
{
list_type: 'url-domain' as const,
description: 'Domains blocked from being linked, optionally covering every subdomain rooted at the domain.',
description:
'Domains blocked from being linked, optionally covering every subdomain rooted at the domain. A value whose leftmost label contains * is a pattern that matches that one label under a registrable domain.',
value_field: 'domain',
fields: ['match_subdomains', 'category', 'severity', 'source_url', 'notes'],
scoped: false,
@@ -488,7 +489,7 @@ export function BanAdminController(app: HonoApp) {
security: ['adminApiKey'],
tags: ['Admin'],
description:
'Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a URL can match a banned domain. The profile-substring blocklist requires a scope.',
'Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a url-domain value can be a hostname or an http(s) URL that a stored domain or pattern covers. The profile-substring blocklist requires a scope.',
}),
async (ctx) => {
const adminService = ctx.get('adminService');
@@ -24,6 +24,7 @@ import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache';
import {profileSubstringBlocklistCache} from '@app/api/middleware/ProfileSubstringBlocklistCache';
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
import {canonicalizeStoredPhrase} from '@app/api/utils/PhraseBlocklistNormalization';
import {parseUrlDomainEntry} from '@app/api/utils/UrlHostRules';
import {canonicalizeUrl} from '@app/api/utils/UrlNormalizer';
import {APIErrorCodes} from '@fluxer/constants/src/ApiErrorCodes';
import {ValidationErrorCodes} from '@fluxer/constants/src/ValidationErrorCodes';
@@ -108,6 +109,16 @@ function normalizeAvatarHashes(hashes: Array<string>): Array<string> {
return Array.from(new Set(hashes.map((hash) => stripAvatarAnimationPrefix(hash.toLowerCase()))));
}
function hostFromUrlOrHostname(value: string): string | null {
const trimmed = value.trim();
if (!/^https?:\/\//i.test(trimmed)) return trimmed;
try {
return new URL(trimmed).hostname;
} catch {
return null;
}
}
function withReasonMetadata(entries: Array<[string, string]>, reason: string | undefined): Map<string, string> {
if (!reason) {
return new Map(entries);
@@ -366,7 +377,9 @@ export class AdminBanManagementService {
) {
const {adminRepository} = this.deps;
const {cache: cacheService} = this.deps.apiContext.services;
const d = data.domain.toLowerCase();
const entry = parseUrlDomainEntry(data.domain);
if (!entry.ok) throw InputValidationError.create('domain', entry.message);
const d = entry.value;
const matchSubs = data.match_subdomains ?? true;
await adminRepository.banUrlDomain({
domain: d,
@@ -378,7 +391,7 @@ export class AdminBanManagementService {
added_by: adminUserId,
notes: data.notes ?? null,
});
urlBlocklistCache.addDomain(d);
urlBlocklistCache.addDomain(d, matchSubs);
await cacheService.publish(BANNED_URL_DOMAINS_REFRESH_CHANNEL, 'refresh');
await this.createBlocklistAuditLog({
adminUserId,
@@ -388,6 +401,7 @@ export class AdminBanManagementService {
metadata: new Map([
['domain', d],
['match_subdomains', String(matchSubs)],
['pattern', String(entry.pattern)],
]),
});
}
@@ -401,7 +415,8 @@ export class AdminBanManagementService {
) {
const {adminRepository} = this.deps;
const {cache: cacheService} = this.deps.apiContext.services;
const d = data.domain.toLowerCase();
const entry = parseUrlDomainEntry(data.domain);
const d = entry.ok ? entry.value : data.domain.trim().toLowerCase();
await adminRepository.unbanUrlDomain(d);
urlBlocklistCache.removeDomain(d);
await cacheService.publish(BANNED_URL_DOMAINS_REFRESH_CHANNEL, 'refresh');
@@ -417,7 +432,8 @@ export class AdminBanManagementService {
async checkUrlDomainBan(data: {domain: string}): Promise<{
banned: boolean;
}> {
return {banned: urlBlocklistCache.isHostnameBanned(data.domain)};
const host = hostFromUrlOrHostname(data.domain);
return {banned: host != null && urlBlocklistCache.isHostnameBanned(host)};
}
async banFileSha(
@@ -0,0 +1,170 @@
// SPDX-License-Identifier: AGPL-3.0-or-later
import {createTestAccount, setUserACLs, type TestAccount} from '@app/api/auth/tests/AuthTestUtils';
import {createChannel, createGuild} from '@app/api/channel/tests/ChannelTestUtils';
import {ensureSessionStarted} from '@app/api/message/tests/MessageTestUtils';
import {getAdminRepository} from '@app/api/middleware/ServiceSingletons';
import {type ApiTestHarness, createApiTestHarness} from '@app/api/test/ApiTestHarness';
import {createBuilder} from '@app/api/test/TestRequestBuilder';
import {APIErrorCodes} from '@fluxer/constants/src/ApiErrorCodes';
import {afterAll, beforeAll, beforeEach, describe, expect, it} from 'vitest';
interface ValidationErrorResponse {
code: string;
errors?: Array<{path: string; message: string}>;
}
interface EntryPage {
items: Array<{value: string; match_subdomains: boolean | null}>;
}
describe('Admin url-domain blocklist patterns', () => {
let harness: ApiTestHarness;
let admin: TestAccount;
beforeAll(async () => {
harness = await createApiTestHarness();
});
beforeEach(async () => {
await harness.reset();
admin = await setUserACLs(harness, await createTestAccount(harness), [
'admin:authenticate',
'ban:url_domain:add',
'ban:url_domain:check',
'ban:url_domain:remove',
]);
});
afterAll(async () => {
await harness?.shutdown();
});
async function add(domain: string, matchSubdomains?: boolean): Promise<void> {
await createBuilder(harness, admin.token)
.post('/admin/blocklists/url-domain/entries')
.body(matchSubdomains === undefined ? {domain} : {domain, match_subdomains: matchSubdomains})
.expect(204)
.execute();
}
async function check(value: string): Promise<boolean> {
const json = await createBuilder<{banned: boolean}>(harness, admin.token)
.get(`/admin/blocklists/url-domain/entries/${encodeURIComponent(value)}`)
.expect(200)
.execute();
return json.banned;
}
async function list(): Promise<EntryPage['items']> {
const json = await createBuilder<EntryPage>(harness, admin.token)
.get('/admin/blocklists/url-domain/entries?limit=200')
.expect(200)
.execute();
return json.items;
}
it('stores a canonical pattern and reports the hosts it covers', async () => {
await add('**Shop**.OnRender.com.');
expect(await list()).toMatchObject([{value: '*shop*.onrender.com', match_subdomains: true}]);
expect(await check('shop-2.onrender.com')).toBe(true);
expect(await check('https://www.myshop.onrender.com/checkout')).toBe(true);
expect(await check('onrender.com')).toBe(false);
expect(await check('docs.onrender.com')).toBe(false);
});
it('records whether the entry is a pattern in the audit log', async () => {
await add('*shop*.onrender.com', false);
await add('shop.example.com');
const logs = (await getAdminRepository().listAllAuditLogsPaginated(1000)).filter(
(log) => log.action === 'ban_url_domain',
);
const metadata = logs.map((log) => Object.fromEntries(log.metadata));
expect(metadata).toEqual(
expect.arrayContaining([
{domain: '*shop*.onrender.com', match_subdomains: 'false', pattern: 'true'},
{domain: 'shop.example.com', match_subdomains: 'true', pattern: 'false'},
]),
);
});
it('rejects patterns that are too broad or malformed', async () => {
for (const domain of ['*', '*.com', '*shop*.co.uk', '*.onrender.com', '*ab*.onrender.com', 'shop.*.example.com']) {
const json = await createBuilder<ValidationErrorResponse>(harness, admin.token)
.post('/admin/blocklists/url-domain/entries')
.body({domain})
.expect(400, 'INVALID_FORM_BODY')
.execute();
expect(json.errors?.[0]?.path, domain).toBe('domain');
}
expect(await list()).toEqual([]);
});
it('validates the value on update', async () => {
const json = await createBuilder<ValidationErrorResponse>(harness, admin.token)
.patch(`/admin/blocklists/url-domain/entries/${encodeURIComponent('*.com')}`)
.body({})
.expect(400, 'INVALID_FORM_BODY')
.execute();
expect(json.errors?.[0]?.path).toBe('domain');
});
it('stores internationalized domains in ASCII form', async () => {
await add('Bücher.Example.');
expect((await list()).map((entry) => entry.value)).toEqual(['xn--bcher-kva.example']);
expect(await check('www.bücher.example')).toBe(true);
});
it('accepts an add for a domain that is already blocked', async () => {
await add('shop.example.com');
await add('shop.example.com', false);
expect(await list()).toMatchObject([{value: 'shop.example.com', match_subdomains: false}]);
});
it('removes a pattern through any spelling that canonicalizes to it', async () => {
await add('*shop*.onrender.com');
await createBuilder(harness, admin.token)
.delete(`/admin/blocklists/url-domain/entries/${encodeURIComponent('*SHOP**.onrender.com')}`)
.expect(204)
.execute();
expect(await list()).toEqual([]);
expect(await check('shop.onrender.com')).toBe(false);
});
it('blocks messages whose masked links or autolinks point at a covered host', async () => {
await add('*shop*.onrender.com');
const member = await createTestAccount(harness);
const guild = await createGuild(harness, member.token, 'Links');
const channel = await createChannel(harness, member.token, guild.id, 'general');
await ensureSessionStarted(harness, member.token);
for (const content of [
'[open the store](https://shop-2.onrender.com)',
'<https://[email protected]:8443/x>',
'https://SHOP.onrender.com./',
]) {
await createBuilder(harness, member.token)
.post(`/channels/${channel.id}/messages`)
.body({content})
.expect(403, APIErrorCodes.CONTENT_BLOCKED)
.execute();
}
await createBuilder(harness, member.token)
.post(`/channels/${channel.id}/messages`)
.body({content: '[docs](https://docs.onrender.com) and https://onrender.com'})
.expect(200)
.execute();
});
it('blocks rich embeds that link to a covered host', async () => {
await add('*shop*.onrender.com');
const member = await createTestAccount(harness);
const guild = await createGuild(harness, member.token, 'Embeds');
const channel = await createChannel(harness, member.token, guild.id, 'general');
await ensureSessionStarted(harness, member.token);
await createBuilder(harness, member.token)
.post(`/channels/${channel.id}/messages`)
.body({embeds: [{title: 'Store', url: 'https://shop.onrender.com/'}]})
.expect(403, APIErrorCodes.CONTENT_BLOCKED)
.execute();
});
});
@@ -33,7 +33,7 @@ function attachment(id: bigint, hash: string | null): MessageAttachment {
};
}
function message(attachments: Array<MessageAttachment>): Message {
function message(attachments: Array<MessageAttachment>, content = ''): Message {
return new Message({
channel_id: createChannelID(10n),
bucket: 0,
@@ -43,7 +43,7 @@ function message(attachments: Array<MessageAttachment>): Message {
webhook_id: null,
webhook_name: null,
webhook_avatar_hash: null,
content: '',
content,
edited_timestamp: null,
pinned_timestamp: null,
flags: 0,
@@ -75,10 +75,10 @@ describe('message activity', () => {
resetActivityEventsForTests();
});
function params(attachments: Array<MessageAttachment>) {
function params(attachments: Array<MessageAttachment>, content = '') {
return {
user: {id: createUserID(3n), isBot: false} as unknown as User,
message: message(attachments),
message: message(attachments, content),
channel: {id: createChannelID(10n), type: ChannelTypes.DM} as unknown as Channel,
guildId: null,
guildOwnerId: null,
@@ -117,4 +117,18 @@ describe('message activity', () => {
expect(updated.data).toMatchObject({message_id: '100', attachments: [{hash: HASH.toLowerCase()}]});
expect(updated.id).not.toBe(created.id);
});
it('records the link domain of a masked markdown link without its brackets', async () => {
const publisher = new CapturingPublisher();
await startActivityEvents({publisher, kv: new MockKVProvider()});
emitMessageCreated(
params(
[],
'[OPEN](https://Shop.Example.com) [docs](<https://www.docs.example.org/a>) https://[email protected]:8443/x',
),
);
await vi.waitFor(() => expect(publisher.payloads).toHaveLength(1));
const event = JSON.parse(publisher.payloads[0]!);
expect(event.data.link_domains).toEqual(['shop.example.com', 'docs.example.org', 'cdn.example.net']);
});
});
@@ -10,22 +10,20 @@ import type {Message} from '@app/api/models/Message';
import type {User} from '@app/api/models/User';
import type {IUserRepository} from '@app/api/user/IUserRepository';
import {findInvites} from '@app/api/utils/InviteUtils';
import {extractLinkHosts} from '@app/api/utils/UrlNormalizer';
import {ChannelTypes} from '@fluxer/constants/src/ChannelConstants';
import {RelationshipTypes} from '@fluxer/constants/src/UserConstants';
const CONTENT_MAX_CHARS = 2000;
const LIST_MAX = 10;
const MENTIONS_MAX = 20;
const LINK_PATTERN = /https?:\/\/([^\s/?#<>"']+)/giu;
const WWW_PREFIX_RE = /^www\./u;
function linkDomains(content: string): Array<string> {
const domains = new Set<string>();
for (const match of content.matchAll(LINK_PATTERN)) {
const host = match[1]
?.toLowerCase()
.replace(/:\d+$/u, '')
.replace(/^www\./u, '');
if (host) domains.add(host);
for (const host of extractLinkHosts(content)) {
const domain = host.replace(WWW_PREFIX_RE, '');
if (domain) domains.add(domain);
if (domains.size >= LIST_MAX) break;
}
return [...domains];
@@ -94,10 +94,14 @@ export class MessageValidationService {
contentModerationService.scanText(data.content, modCtx);
if (data.embeds) {
for (const embed of data.embeds) {
if (embed.url) contentModerationService.scanUrl(embed.url, modCtx);
contentModerationService.scanText(embed.title ?? null, modCtx);
contentModerationService.scanText(embed.description ?? null, modCtx);
if (embed.footer) contentModerationService.scanText(embed.footer.text ?? null, modCtx);
if (embed.author) contentModerationService.scanText(embed.author.name ?? null, modCtx);
if (embed.author) {
contentModerationService.scanText(embed.author.name ?? null, modCtx);
if (embed.author.url) contentModerationService.scanUrl(embed.author.url, modCtx);
}
if (embed.fields) {
for (const field of embed.fields) {
contentModerationService.scanText(field.name ?? null, modCtx);
@@ -5,7 +5,6 @@ import {Logger} from '@app/api/Logger';
import {fileShaCache} from '@app/api/middleware/FileShaCache';
import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache';
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
import {extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
import {ContentBlockedError} from '@fluxer/errors/src/domains/content/ContentBlockedError';
export interface ModerationContext {
@@ -40,16 +39,12 @@ class ContentModerationService {
);
throw new ContentBlockedError();
}
const urls = extractUrlCandidates(text);
if (urls.length === 0) return;
for (const url of urls) {
if (urlBlocklistCache.isUrlOrDomainBanned(url)) {
Logger.warn(
{surface: ctx.surface, userId: ctx.userId?.toString(), guildId: ctx.guildId?.toString()},
'content_moderation.block url match in text',
);
throw new ContentBlockedError();
}
if (urlBlocklistCache.containsBannedLink(text)) {
Logger.warn(
{surface: ctx.surface, userId: ctx.userId?.toString(), guildId: ctx.guildId?.toString()},
'content_moderation.block url match in text',
);
throw new ContentBlockedError();
}
}
@@ -4,7 +4,6 @@ import {Logger} from '@app/api/Logger';
import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache';
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
import {readRequestJsonBody} from '@app/api/utils/RequestJsonBody';
import {extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
import {ContentBlockedError} from '@fluxer/errors/src/domains/content/ContentBlockedError';
import {createMiddleware} from 'hono/factory';
@@ -73,6 +72,8 @@ const SKIP_FIELD_SUFFIXES = [
] as const;
const SKIP_CONTENT_FILTER_PATH_PARTS = [
'/admin/blocklists/phrase/',
'/admin/blocklists/url-domain/',
'/admin/blocklists/url/',
'/auth/',
'/oauth2/',
'/premium/store/',
@@ -149,15 +150,12 @@ const ContentFilterMiddleware = createMiddleware(async (ctx, next) => {
);
throw new ContentBlockedError();
}
const urls = extractUrlCandidates(text);
for (const url of urls) {
if (urlBlocklistCache.isUrlOrDomainBanned(url)) {
Logger.warn(
{surface: 'global_filter', userId: userId?.toString(), path},
'content_moderation.block url match in request body',
);
throw new ContentBlockedError();
}
if (urlBlocklistCache.containsBannedLink(text)) {
Logger.warn(
{surface: 'global_filter', userId: userId?.toString(), path},
'content_moderation.block url match in request body',
);
throw new ContentBlockedError();
}
}
return next();
@@ -7,12 +7,13 @@ import {BANNED_URL_DOMAINS_REFRESH_CHANNEL, BANNED_URLS_REFRESH_CHANNEL} from '@
import type {IStorageService} from '@app/api/infrastructure/IStorageService';
import {Logger} from '@app/api/Logger';
import {RefreshSubscription} from '@app/api/utils/RefreshSubscription';
import {canonicalizeUrl} from '@app/api/utils/UrlNormalizer';
import {UrlHostRuleSet} from '@app/api/utils/UrlHostRules';
import {canonicalizeUrl, extractLinkHosts, extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
import type {IKVProvider} from '@pkgs/kv_client/src/IKVProvider';
class UrlBlocklistCache {
private exactUrls: Set<string> = new Set();
private blockedDomains: Set<string> = new Set();
private hostRules = new UrlHostRuleSet();
private adminRepository = new AdminRepository();
private kvClient: IKVProvider | null = null;
private storageService: IStorageService | null = null;
@@ -55,15 +56,15 @@ class UrlBlocklistCache {
for (const row of manualUrls) {
if (row.url_canonical) nextUrls.add(row.url_canonical.toLowerCase());
}
const nextDomains = new Set<string>();
const nextHostRules = new UrlHostRuleSet();
for (const row of domains) {
nextDomains.add(row.domain.toLowerCase());
nextHostRules.add(row.domain, row.match_subdomains ?? true);
}
this.exactUrls = nextUrls;
this.blockedDomains = nextDomains;
this.hostRules = nextHostRules;
this.consecutiveFailures = 0;
Logger.debug(
{urls: nextUrls.size, domains: nextDomains.size, feedUrls: feedUrls.size},
{urls: nextUrls.size, ...nextHostRules.size, feedUrls: feedUrls.size},
'URL blocklist cache refreshed',
);
}
@@ -94,7 +95,17 @@ class UrlBlocklistCache {
}
isHostnameBanned(host: string): boolean {
return this.blockedDomains.has(host.toLowerCase());
return this.hostRules.matches(host);
}
containsBannedLink(text: string): boolean {
for (const url of extractUrlCandidates(text)) {
if (this.isUrlOrDomainBanned(url)) return true;
}
for (const host of extractLinkHosts(text)) {
if (this.isHostnameBanned(host)) return true;
}
return false;
}
addExactUrl(canonical: string): void {
@@ -105,21 +116,22 @@ class UrlBlocklistCache {
this.exactUrls.delete(canonical.toLowerCase());
}
addDomain(domain: string): void {
this.blockedDomains.add(domain.toLowerCase());
addDomain(domain: string, matchSubdomains = true): void {
this.hostRules.add(domain, matchSubdomains);
}
removeDomain(domain: string): void {
this.blockedDomains.delete(domain.toLowerCase());
this.hostRules.remove(domain);
}
get size(): {
urls: number;
domains: number;
patterns: number;
} {
return {
urls: this.exactUrls.size,
domains: this.blockedDomains.size,
...this.hostRules.size,
};
}
@@ -128,7 +140,7 @@ class UrlBlocklistCache {
Logger.error({error}, 'Failed to shut down URL blocklist cache');
});
this.exactUrls = new Set();
this.blockedDomains = new Set();
this.hostRules = new UrlHostRuleSet();
this.kvClient = null;
this.storageService = null;
this.consecutiveFailures = 0;
@@ -81,4 +81,14 @@ describe('shouldSkipContentFilterPath', () => {
const result = paths.map((path) => shouldSkipContentFilterPath(path));
expect(result).toEqual([false, false, false]);
});
test('skips blocklist writes whose values are the blocked content', () => {
const paths = [
'/admin/blocklists/phrase/entries',
'/admin/blocklists/url/entries',
'/admin/blocklists/url-domain/entries',
'/admin/blocklists/profile-substring/entries',
];
const result = paths.map((path) => shouldSkipContentFilterPath(path));
expect(result).toEqual([true, true, true, false]);
});
});
@@ -0,0 +1,65 @@
// SPDX-License-Identifier: AGPL-3.0-or-later
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
import {afterEach, describe, expect, it} from 'vitest';
describe('urlBlocklistCache link matching', () => {
afterEach(() => {
urlBlocklistCache.resetForTesting();
});
it('blocks a masked markdown link whose target a pattern covers', () => {
urlBlocklistCache.addDomain('*shop*.onrender.com', true);
expect(urlBlocklistCache.containsBannedLink('[open the store](https://shop-2.onrender.com)')).toBe(true);
expect(urlBlocklistCache.containsBannedLink('[open the store](<https://www.shop.onrender.com/x>)')).toBe(true);
expect(urlBlocklistCache.containsBannedLink('[https://docs.onrender.com](https://shop.onrender.com)')).toBe(true);
});
it('blocks autolinks and bare links', () => {
urlBlocklistCache.addDomain('*shop*.onrender.com', false);
expect(urlBlocklistCache.containsBannedLink('<https://myshop.onrender.com/path>')).toBe(true);
expect(urlBlocklistCache.containsBannedLink('visit myshop.onrender.com today')).toBe(true);
});
it('normalizes the link target before matching', () => {
urlBlocklistCache.addDomain('*shop*.onrender.com', false);
const variants = [
'https://user:[email protected]:8443/x',
'https://[email protected]',
'https://shop.onrender.com./',
'https://shop%2Eonrender%2Ecom/',
'https://shop。onrender。com/',
'https://shop-ü.onrender.com/',
];
for (const text of variants) {
expect(urlBlocklistCache.containsBannedLink(text), text).toBe(true);
}
});
it('leaves the bare suffix and unrelated hosts alone', () => {
urlBlocklistCache.addDomain('*shop*.onrender.com', true);
expect(urlBlocklistCache.containsBannedLink('https://onrender.com/docs')).toBe(false);
expect(urlBlocklistCache.containsBannedLink('[docs](https://docs.onrender.com)')).toBe(false);
expect(urlBlocklistCache.containsBannedLink('the shop is closed')).toBe(false);
});
it('covers subdomains of a domain entry only when it is flagged to', () => {
urlBlocklistCache.addDomain('shop.example.com', true);
urlBlocklistCache.addDomain('store.example.com', false);
expect(urlBlocklistCache.containsBannedLink('https://www.shop.example.com')).toBe(true);
expect(urlBlocklistCache.containsBannedLink('https://store.example.com')).toBe(true);
expect(urlBlocklistCache.containsBannedLink('https://www.store.example.com')).toBe(false);
});
it('stops matching after removal', () => {
urlBlocklistCache.addDomain('*shop*.onrender.com', true);
urlBlocklistCache.removeDomain('*shop*.onrender.com');
expect(urlBlocklistCache.containsBannedLink('https://shop.onrender.com')).toBe(false);
});
it('applies domain rules to a single URL', () => {
urlBlocklistCache.addDomain('*shop*.onrender.com', false);
expect(urlBlocklistCache.isUrlOrDomainBanned('https://shop.onrender.com/checkout')).toBe(true);
expect(urlBlocklistCache.isUrlOrDomainBanned('https://docs.onrender.com/')).toBe(false);
});
});
+167
View File
@@ -0,0 +1,167 @@
// SPDX-License-Identifier: AGPL-3.0-or-later
import {normalizeHostname} from '@app/api/utils/UrlNormalizer';
import {getDomain} from 'tldts';
const URL_HOST_PATTERN_MAX_WILDCARDS = 3;
const URL_HOST_PATTERN_MIN_LITERAL_CHARS = 3;
const MAX_HOSTNAME_LENGTH = 253;
const HOSTNAME_LABEL_RE = /^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/;
const GLOB_LABEL_RE = /^[a-z0-9*-]{1,63}$/;
const REPEATED_WILDCARDS_RE = /\*+/g;
export type UrlDomainEntry =
| {ok: true; value: string; pattern: boolean}
| {
ok: false;
message: string;
};
function isValidHostname(host: string): boolean {
if (host.length > MAX_HOSTNAME_LENGTH) return false;
return host.split('.').every((label) => HOSTNAME_LABEL_RE.test(label));
}
function invalid(message: string): UrlDomainEntry {
return {ok: false, message};
}
export function parseUrlDomainEntry(raw: string): UrlDomainEntry {
const value = raw.trim().toLowerCase();
if (!value.includes('*')) {
const host = normalizeHostname(value);
if (!host?.includes('.') || !isValidHostname(host)) {
return invalid('Must be a valid domain');
}
return {ok: true, value: host, pattern: false};
}
const separator = value.indexOf('.');
if (separator === -1) {
return invalid('A pattern must name a domain after its wildcard label');
}
const glob = value.slice(0, separator).replace(REPEATED_WILDCARDS_RE, '*');
const suffixValue = value.slice(separator + 1);
if (suffixValue.includes('*')) {
return invalid('Only the leftmost label of a pattern can contain *');
}
if (!GLOB_LABEL_RE.test(glob)) {
return invalid('The wildcard label can contain only a-z, 0-9, hyphens, and *');
}
const wildcards = glob.length - glob.replaceAll('*', '').length;
if (wildcards > URL_HOST_PATTERN_MAX_WILDCARDS) {
return invalid(`The wildcard label can contain at most ${URL_HOST_PATTERN_MAX_WILDCARDS} *`);
}
if (glob.length - wildcards < URL_HOST_PATTERN_MIN_LITERAL_CHARS) {
return invalid(
`The wildcard label needs at least ${URL_HOST_PATTERN_MIN_LITERAL_CHARS} characters besides *, so the pattern is too broad`,
);
}
const suffix = normalizeHostname(suffixValue);
if (!suffix || !isValidHostname(suffix)) {
return invalid('The part after the wildcard label must be a valid domain');
}
if (getDomain(suffix, {allowPrivateDomains: false}) === null) {
return invalid('The part after the wildcard label is a public suffix, so the pattern is too broad');
}
const entry = `${glob}.${suffix}`;
if (entry.length > MAX_HOSTNAME_LENGTH) {
return invalid('Must be a valid domain');
}
return {ok: true, value: entry, pattern: true};
}
interface HostPatternRule {
value: string;
segments: ReadonlyArray<string>;
matchSubdomains: boolean;
}
function globMatches(segments: ReadonlyArray<string>, label: string): boolean {
const first = segments[0] ?? '';
const last = segments[segments.length - 1] ?? '';
if (!label.startsWith(first)) return false;
let position = first.length;
for (let index = 1; index < segments.length - 1; index++) {
const segment = segments[index] ?? '';
const found = label.indexOf(segment, position);
if (found === -1) return false;
position = found + segment.length;
}
return label.length - last.length >= position && label.endsWith(last);
}
export class UrlHostRuleSet {
private readonly domains = new Map<string, boolean>();
private readonly patterns = new Map<string, Array<HostPatternRule>>();
private patternCount = 0;
add(rawValue: string, matchSubdomains: boolean): void {
if (rawValue.includes('*')) {
const entry = parseUrlDomainEntry(rawValue);
if (!entry.ok || !entry.pattern) return;
this.remove(entry.value);
const separator = entry.value.indexOf('.');
const suffix = entry.value.slice(separator + 1);
const rules = this.patterns.get(suffix) ?? [];
rules.push({
value: entry.value,
segments: entry.value.slice(0, separator).split('*'),
matchSubdomains,
});
this.patterns.set(suffix, rules);
this.patternCount++;
return;
}
const host = normalizeHostname(rawValue);
if (host) this.domains.set(host, matchSubdomains);
}
remove(rawValue: string): void {
if (!rawValue.includes('*')) {
const host = normalizeHostname(rawValue);
if (host) this.domains.delete(host);
return;
}
const entry = parseUrlDomainEntry(rawValue);
if (!entry.ok) return;
const suffix = entry.value.slice(entry.value.indexOf('.') + 1);
const rules = this.patterns.get(suffix);
if (!rules) return;
const remaining = rules.filter((rule) => rule.value !== entry.value);
this.patternCount -= rules.length - remaining.length;
if (remaining.length === 0) {
this.patterns.delete(suffix);
} else {
this.patterns.set(suffix, remaining);
}
}
matches(rawHost: string): boolean {
const host = normalizeHostname(rawHost);
if (!host) return false;
if (this.domains.has(host)) return true;
let labelStart = 0;
let isFirstLabel = true;
while (labelStart < host.length) {
const separator = host.indexOf('.', labelStart);
if (separator === -1) return false;
const suffix = host.slice(separator + 1);
if (this.domains.get(suffix) === true) return true;
const rules = this.patterns.get(suffix);
if (rules) {
const label = host.slice(labelStart, separator);
for (const rule of rules) {
if ((isFirstLabel || rule.matchSubdomains) && globMatches(rule.segments, label)) return true;
}
}
labelStart = separator + 1;
isFirstLabel = false;
}
return false;
}
get size(): {domains: number; patterns: number} {
return {domains: this.domains.size, patterns: this.patternCount};
}
}
+37
View File
@@ -58,6 +58,17 @@ export function canonicalizeUrl(raw: string): string | null {
return parsed.toString().toLowerCase();
}
const TRAILING_DOTS_RE = /\.+$/;
export function normalizeHostname(raw: string): string | null {
const trimmed = raw.trim();
if (!trimmed) return null;
const ascii = domainToASCII(trimmed);
if (!ascii) return null;
const host = ascii.toLowerCase().replace(TRAILING_DOTS_RE, '');
return host || null;
}
const URL_CANDIDATE_RE =
/(?<![a-z0-9._+-]@)((?:https?:\/\/)?(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/:?#][^\s<>\u201D']*)?)/gi;
const TRAILING_PUNCT_RE = /[.,;:!?)\]}\x22'\u00bb\u201C\u201D]+$/;
@@ -78,3 +89,29 @@ export function extractUrlCandidates(text: string | null | undefined): Array<str
}
return out;
}
const LINK_AUTHORITY_RE = /https?:\/\/([^\s/\\?#<>"'`()[\]{}|^]+)/giu;
function hostFromAuthority(authority: string): string | null {
const cleaned = authority.replace(TRAILING_PUNCT_RE, '');
if (!cleaned) return null;
let parsed: URL;
try {
parsed = new URL(`http://${cleaned}/`);
} catch {
return null;
}
return normalizeHostname(parsed.hostname);
}
export function extractLinkHosts(text: string | null | undefined): Array<string> {
if (!text) return [];
const hosts = new Set<string>();
for (const match of text.matchAll(LINK_AUTHORITY_RE)) {
const authority = match[1];
if (!authority) continue;
const host = hostFromAuthority(authority);
if (host) hosts.add(host);
}
return [...hosts];
}
@@ -0,0 +1,171 @@
// SPDX-License-Identifier: AGPL-3.0-or-later
import {parseUrlDomainEntry, UrlHostRuleSet} from '@app/api/utils/UrlHostRules';
import {describe, expect, it} from 'vitest';
function rules(entries: Array<[string, boolean]>): UrlHostRuleSet {
const set = new UrlHostRuleSet();
for (const [value, matchSubdomains] of entries) {
set.add(value, matchSubdomains);
}
return set;
}
describe('parseUrlDomainEntry', () => {
it('canonicalizes a plain domain', () => {
expect(parseUrlDomainEntry(' Shop.Example.COM. ')).toEqual({ok: true, value: 'shop.example.com', pattern: false});
});
it('stores an internationalized domain in punycode', () => {
expect(parseUrlDomainEntry('bücher.example')).toEqual({ok: true, value: 'xn--bcher-kva.example', pattern: false});
});
it('keeps an exact entry for a domain under a shared hosting suffix', () => {
expect(parseUrlDomainEntry('onrender.com')).toEqual({ok: true, value: 'onrender.com', pattern: false});
});
it('rejects malformed plain domains', () => {
for (const value of ['localhost', 'bad_label.example.com', '-lead.example.com', 'a..example.com', 'a b.com']) {
expect(parseUrlDomainEntry(value).ok).toBe(false);
}
});
it('canonicalizes a pattern and collapses repeated wildcards', () => {
expect(parseUrlDomainEntry('**Shop**.OnRender.com.')).toEqual({
ok: true,
value: '*shop*.onrender.com',
pattern: true,
});
});
it('accepts patterns under a private shared hosting suffix', () => {
expect(parseUrlDomainEntry('*shop*.github.io').ok).toBe(true);
expect(parseUrlDomainEntry('shop-*.example.co.uk').ok).toBe(true);
});
it('rejects patterns that are too broad', () => {
for (const value of [
'*',
'*.com',
'*shop*.com',
'shop*.co.uk',
'*.example.com',
'*ab*.example.com',
'a*b.example.com',
]) {
expect(parseUrlDomainEntry(value).ok).toBe(false);
}
});
it('rejects wildcards outside the leftmost label', () => {
expect(parseUrlDomainEntry('shop.*.example.com').ok).toBe(false);
expect(parseUrlDomainEntry('*shop*.example*.com').ok).toBe(false);
});
it('rejects wildcard labels with unsupported characters or too many wildcards', () => {
expect(parseUrlDomainEntry('*sh?p*.example.com').ok).toBe(false);
expect(parseUrlDomainEntry('*sh_p*.example.com').ok).toBe(false);
expect(parseUrlDomainEntry('*a*b*c*d*.example.com').ok).toBe(false);
});
it('rejects a pattern without a domain', () => {
expect(parseUrlDomainEntry('*shop*').ok).toBe(false);
});
});
describe('UrlHostRuleSet', () => {
it('matches an exact domain', () => {
const set = rules([['shop.example.com', false]]);
expect(set.matches('shop.example.com')).toBe(true);
expect(set.matches('www.shop.example.com')).toBe(false);
expect(set.matches('example.com')).toBe(false);
});
it('matches subdomains only when the entry covers them', () => {
const set = rules([['shop.example.com', true]]);
expect(set.matches('shop.example.com')).toBe(true);
expect(set.matches('a.b.shop.example.com')).toBe(true);
expect(set.matches('myshop.example.com')).toBe(false);
});
it('normalizes the checked host', () => {
const set = rules([['shop.example.com', true]]);
expect(set.matches('SHOP.Example.com.')).toBe(true);
expect(set.matches('www.shop。example。com')).toBe(true);
expect(rules([['xn--bcher-kva.example', false]]).matches('Bücher.example')).toBe(true);
});
it('matches a pattern against the label left of its suffix', () => {
const set = rules([['*shop*.onrender.com', false]]);
expect(set.matches('shop.onrender.com')).toBe(true);
expect(set.matches('best-shop-2.onrender.com')).toBe(true);
expect(set.matches('SHOPPING.onrender.com')).toBe(true);
expect(set.matches('store.onrender.com')).toBe(false);
});
it('never matches the bare suffix of a pattern', () => {
const set = rules([['*shop*.onrender.com', true]]);
expect(set.matches('onrender.com')).toBe(false);
expect(set.matches('com')).toBe(false);
expect(set.matches('shop.com')).toBe(false);
expect(set.matches('shop.onrender.com.evil.example')).toBe(false);
expect(set.matches('shoponrender.com')).toBe(false);
});
it('applies anchored pattern segments', () => {
const set = rules([
['shop-*.example.com', false],
['*-store.example.org', false],
['a*b*c.example.net', false],
]);
expect(set.matches('shop-1.example.com')).toBe(true);
expect(set.matches('myshop-1.example.com')).toBe(false);
expect(set.matches('big-store.example.org')).toBe(true);
expect(set.matches('big-store2.example.org')).toBe(false);
expect(set.matches('axxbyyc.example.net')).toBe(true);
expect(set.matches('abc.example.net')).toBe(true);
expect(set.matches('acb.example.net')).toBe(false);
});
it('extends a pattern to deeper subdomains only when the entry covers them', () => {
const exact = rules([['*shop*.onrender.com', false]]);
const covering = rules([['*shop*.onrender.com', true]]);
expect(exact.matches('www.shop.onrender.com')).toBe(false);
expect(covering.matches('www.shop.onrender.com')).toBe(true);
expect(covering.matches('shop.www.onrender.com')).toBe(false);
});
it('matches internationalized labels through their punycode form', () => {
const set = rules([['*shop*.onrender.com', false]]);
expect(set.matches('shop-ü.onrender.com')).toBe(true);
});
it('removes domains and patterns', () => {
const set = rules([
['shop.example.com', true],
['*shop*.onrender.com', true],
['*store*.onrender.com', true],
]);
set.remove('SHOP.example.com');
set.remove('**shop*.onrender.com');
expect(set.matches('shop.example.com')).toBe(false);
expect(set.matches('shop.onrender.com')).toBe(false);
expect(set.matches('store.onrender.com')).toBe(true);
expect(set.size).toEqual({domains: 0, patterns: 1});
});
it('replaces a pattern when it is added again', () => {
const set = rules([
['*shop*.onrender.com', false],
['*shop*.onrender.com', true],
]);
expect(set.size).toEqual({domains: 0, patterns: 1});
expect(set.matches('www.shop.onrender.com')).toBe(true);
});
it('ignores invalid stored patterns', () => {
const set = rules([['*.com', true]]);
expect(set.size).toEqual({domains: 0, patterns: 0});
expect(set.matches('anything.com')).toBe(false);
});
});
@@ -1,6 +1,6 @@
// SPDX-License-Identifier: AGPL-3.0-or-later
import {canonicalizeUrl, extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
import {canonicalizeUrl, extractLinkHosts, extractUrlCandidates, normalizeHostname} from '@app/api/utils/UrlNormalizer';
import {describe, expect, it} from 'vitest';
describe('canonicalizeUrl', () => {
@@ -176,3 +176,61 @@ describe('extractUrlCandidates', () => {
expect(extractUrlCandidates('hello world no links here')).toEqual([]);
});
});
describe('normalizeHostname', () => {
it('lowercases and strips trailing dots', () => {
expect(normalizeHostname('Shop.Example.COM..')).toBe('shop.example.com');
});
it('converts internationalized names to punycode', () => {
expect(normalizeHostname('bücher.example')).toBe('xn--bcher-kva.example');
});
it('maps ideographic full stops to dots', () => {
expect(normalizeHostname('shop\u3002example\u3002com')).toBe('shop.example.com');
});
it('rejects empty and invalid input', () => {
expect(normalizeHostname(' ')).toBeNull();
expect(normalizeHostname('.')).toBeNull();
expect(normalizeHostname('a b.com')).toBeNull();
});
});
describe('extractLinkHosts', () => {
it('drops the closing parenthesis of a masked markdown link', () => {
expect(extractLinkHosts('[open the shop](https://shop.example.com)')).toEqual(['shop.example.com']);
});
it('drops brackets and parentheses around nested links', () => {
expect(extractLinkHosts('[[x]](https://shop.example.com)(more)')).toEqual(['shop.example.com']);
});
it('reads angle-bracket autolinks', () => {
expect(extractLinkHosts('<https://Shop.Example.com/path>')).toEqual(['shop.example.com']);
});
it('ignores userinfo and ports', () => {
expect(extractLinkHosts('https://user:[email protected]:8443/x')).toEqual(['shop.example.com']);
expect(extractLinkHosts('[x](https://[email protected])')).toEqual(['shop.example.com']);
});
it('strips trailing dots and punctuation', () => {
expect(extractLinkHosts('see https://shop.example.com./ and https://other.example.org, ok')).toEqual([
'shop.example.com',
'other.example.org',
]);
});
it('decodes percent-encoded hosts the way a browser does', () => {
expect(extractLinkHosts('https://shop%2Eexample%2Ecom/')).toEqual(['shop.example.com']);
});
it('returns internationalized hosts in punycode', () => {
expect(extractLinkHosts('https://bücher.example/')).toEqual(['xn--bcher-kva.example']);
});
it('matches the scheme case-insensitively', () => {
expect(extractLinkHosts('HTTPS://SHOP.EXAMPLE.COM')).toEqual(['shop.example.com']);
});
it('stops the host at a backslash', () => {
expect(extractLinkHosts('https://shop.example.com\\path')).toEqual(['shop.example.com']);
});
it('deduplicates hosts', () => {
expect(extractLinkHosts('https://a.example.com https://A.example.com/x')).toEqual(['a.example.com']);
});
it('returns an empty array without links', () => {
expect(extractLinkHosts('shop.example.com')).toEqual([]);
expect(extractLinkHosts(null)).toEqual([]);
});
});
@@ -330,11 +330,11 @@ async function scanEmbedsForBannedContent(
continue;
}
for (const embed of embeds) {
const imageUrls = [embed.thumbnail?.url, embed.image?.url, embed.video?.url, embed.audio?.url].filter(
const embedUrls = [embed.url, embed.thumbnail?.url, embed.image?.url, embed.video?.url, embed.audio?.url].filter(
(u): u is string => u != null,
);
for (const imageUrl of imageUrls) {
contentModerationService.scanUrl(imageUrl, ctx);
for (const embedUrl of embedUrls) {
contentModerationService.scanUrl(embedUrl, ctx);
}
const children = embed.children ?? [];
for (const child of children) {