mirror of
https://github.com/fluxerapp/fluxer
synced 2026-10-07 19:22:14 +09:00
feat(blocklist): add url-domain host patterns (#3164)
This commit is contained in:
@@ -79,6 +79,7 @@
|
||||
"sharp": "catalog:",
|
||||
"stripe": "catalog:",
|
||||
"tempy": "catalog:",
|
||||
"tldts": "catalog:",
|
||||
"transliteration": "catalog:",
|
||||
"tsx": "catalog:",
|
||||
"uint8array-extras": "catalog:",
|
||||
|
||||
@@ -99,7 +99,8 @@ const BLOCKLIST_CATALOG = [
|
||||
},
|
||||
{
|
||||
list_type: 'url-domain' as const,
|
||||
description: 'Domains blocked from being linked, optionally covering every subdomain rooted at the domain.',
|
||||
description:
|
||||
'Domains blocked from being linked, optionally covering every subdomain rooted at the domain. A value whose leftmost label contains * is a pattern that matches that one label under a registrable domain.',
|
||||
value_field: 'domain',
|
||||
fields: ['match_subdomains', 'category', 'severity', 'source_url', 'notes'],
|
||||
scoped: false,
|
||||
@@ -488,7 +489,7 @@ export function BanAdminController(app: HonoApp) {
|
||||
security: ['adminApiKey'],
|
||||
tags: ['Admin'],
|
||||
description:
|
||||
'Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a URL can match a banned domain. The profile-substring blocklist requires a scope.',
|
||||
'Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a url-domain value can be a hostname or an http(s) URL that a stored domain or pattern covers. The profile-substring blocklist requires a scope.',
|
||||
}),
|
||||
async (ctx) => {
|
||||
const adminService = ctx.get('adminService');
|
||||
|
||||
@@ -24,6 +24,7 @@ import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache';
|
||||
import {profileSubstringBlocklistCache} from '@app/api/middleware/ProfileSubstringBlocklistCache';
|
||||
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
|
||||
import {canonicalizeStoredPhrase} from '@app/api/utils/PhraseBlocklistNormalization';
|
||||
import {parseUrlDomainEntry} from '@app/api/utils/UrlHostRules';
|
||||
import {canonicalizeUrl} from '@app/api/utils/UrlNormalizer';
|
||||
import {APIErrorCodes} from '@fluxer/constants/src/ApiErrorCodes';
|
||||
import {ValidationErrorCodes} from '@fluxer/constants/src/ValidationErrorCodes';
|
||||
@@ -108,6 +109,16 @@ function normalizeAvatarHashes(hashes: Array<string>): Array<string> {
|
||||
return Array.from(new Set(hashes.map((hash) => stripAvatarAnimationPrefix(hash.toLowerCase()))));
|
||||
}
|
||||
|
||||
function hostFromUrlOrHostname(value: string): string | null {
|
||||
const trimmed = value.trim();
|
||||
if (!/^https?:\/\//i.test(trimmed)) return trimmed;
|
||||
try {
|
||||
return new URL(trimmed).hostname;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function withReasonMetadata(entries: Array<[string, string]>, reason: string | undefined): Map<string, string> {
|
||||
if (!reason) {
|
||||
return new Map(entries);
|
||||
@@ -366,7 +377,9 @@ export class AdminBanManagementService {
|
||||
) {
|
||||
const {adminRepository} = this.deps;
|
||||
const {cache: cacheService} = this.deps.apiContext.services;
|
||||
const d = data.domain.toLowerCase();
|
||||
const entry = parseUrlDomainEntry(data.domain);
|
||||
if (!entry.ok) throw InputValidationError.create('domain', entry.message);
|
||||
const d = entry.value;
|
||||
const matchSubs = data.match_subdomains ?? true;
|
||||
await adminRepository.banUrlDomain({
|
||||
domain: d,
|
||||
@@ -378,7 +391,7 @@ export class AdminBanManagementService {
|
||||
added_by: adminUserId,
|
||||
notes: data.notes ?? null,
|
||||
});
|
||||
urlBlocklistCache.addDomain(d);
|
||||
urlBlocklistCache.addDomain(d, matchSubs);
|
||||
await cacheService.publish(BANNED_URL_DOMAINS_REFRESH_CHANNEL, 'refresh');
|
||||
await this.createBlocklistAuditLog({
|
||||
adminUserId,
|
||||
@@ -388,6 +401,7 @@ export class AdminBanManagementService {
|
||||
metadata: new Map([
|
||||
['domain', d],
|
||||
['match_subdomains', String(matchSubs)],
|
||||
['pattern', String(entry.pattern)],
|
||||
]),
|
||||
});
|
||||
}
|
||||
@@ -401,7 +415,8 @@ export class AdminBanManagementService {
|
||||
) {
|
||||
const {adminRepository} = this.deps;
|
||||
const {cache: cacheService} = this.deps.apiContext.services;
|
||||
const d = data.domain.toLowerCase();
|
||||
const entry = parseUrlDomainEntry(data.domain);
|
||||
const d = entry.ok ? entry.value : data.domain.trim().toLowerCase();
|
||||
await adminRepository.unbanUrlDomain(d);
|
||||
urlBlocklistCache.removeDomain(d);
|
||||
await cacheService.publish(BANNED_URL_DOMAINS_REFRESH_CHANNEL, 'refresh');
|
||||
@@ -417,7 +432,8 @@ export class AdminBanManagementService {
|
||||
async checkUrlDomainBan(data: {domain: string}): Promise<{
|
||||
banned: boolean;
|
||||
}> {
|
||||
return {banned: urlBlocklistCache.isHostnameBanned(data.domain)};
|
||||
const host = hostFromUrlOrHostname(data.domain);
|
||||
return {banned: host != null && urlBlocklistCache.isHostnameBanned(host)};
|
||||
}
|
||||
|
||||
async banFileSha(
|
||||
|
||||
@@ -0,0 +1,170 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import {createTestAccount, setUserACLs, type TestAccount} from '@app/api/auth/tests/AuthTestUtils';
|
||||
import {createChannel, createGuild} from '@app/api/channel/tests/ChannelTestUtils';
|
||||
import {ensureSessionStarted} from '@app/api/message/tests/MessageTestUtils';
|
||||
import {getAdminRepository} from '@app/api/middleware/ServiceSingletons';
|
||||
import {type ApiTestHarness, createApiTestHarness} from '@app/api/test/ApiTestHarness';
|
||||
import {createBuilder} from '@app/api/test/TestRequestBuilder';
|
||||
import {APIErrorCodes} from '@fluxer/constants/src/ApiErrorCodes';
|
||||
import {afterAll, beforeAll, beforeEach, describe, expect, it} from 'vitest';
|
||||
|
||||
interface ValidationErrorResponse {
|
||||
code: string;
|
||||
errors?: Array<{path: string; message: string}>;
|
||||
}
|
||||
|
||||
interface EntryPage {
|
||||
items: Array<{value: string; match_subdomains: boolean | null}>;
|
||||
}
|
||||
|
||||
describe('Admin url-domain blocklist patterns', () => {
|
||||
let harness: ApiTestHarness;
|
||||
let admin: TestAccount;
|
||||
|
||||
beforeAll(async () => {
|
||||
harness = await createApiTestHarness();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
await harness.reset();
|
||||
admin = await setUserACLs(harness, await createTestAccount(harness), [
|
||||
'admin:authenticate',
|
||||
'ban:url_domain:add',
|
||||
'ban:url_domain:check',
|
||||
'ban:url_domain:remove',
|
||||
]);
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
await harness?.shutdown();
|
||||
});
|
||||
|
||||
async function add(domain: string, matchSubdomains?: boolean): Promise<void> {
|
||||
await createBuilder(harness, admin.token)
|
||||
.post('/admin/blocklists/url-domain/entries')
|
||||
.body(matchSubdomains === undefined ? {domain} : {domain, match_subdomains: matchSubdomains})
|
||||
.expect(204)
|
||||
.execute();
|
||||
}
|
||||
|
||||
async function check(value: string): Promise<boolean> {
|
||||
const json = await createBuilder<{banned: boolean}>(harness, admin.token)
|
||||
.get(`/admin/blocklists/url-domain/entries/${encodeURIComponent(value)}`)
|
||||
.expect(200)
|
||||
.execute();
|
||||
return json.banned;
|
||||
}
|
||||
|
||||
async function list(): Promise<EntryPage['items']> {
|
||||
const json = await createBuilder<EntryPage>(harness, admin.token)
|
||||
.get('/admin/blocklists/url-domain/entries?limit=200')
|
||||
.expect(200)
|
||||
.execute();
|
||||
return json.items;
|
||||
}
|
||||
|
||||
it('stores a canonical pattern and reports the hosts it covers', async () => {
|
||||
await add('**Shop**.OnRender.com.');
|
||||
expect(await list()).toMatchObject([{value: '*shop*.onrender.com', match_subdomains: true}]);
|
||||
expect(await check('shop-2.onrender.com')).toBe(true);
|
||||
expect(await check('https://www.myshop.onrender.com/checkout')).toBe(true);
|
||||
expect(await check('onrender.com')).toBe(false);
|
||||
expect(await check('docs.onrender.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('records whether the entry is a pattern in the audit log', async () => {
|
||||
await add('*shop*.onrender.com', false);
|
||||
await add('shop.example.com');
|
||||
const logs = (await getAdminRepository().listAllAuditLogsPaginated(1000)).filter(
|
||||
(log) => log.action === 'ban_url_domain',
|
||||
);
|
||||
const metadata = logs.map((log) => Object.fromEntries(log.metadata));
|
||||
expect(metadata).toEqual(
|
||||
expect.arrayContaining([
|
||||
{domain: '*shop*.onrender.com', match_subdomains: 'false', pattern: 'true'},
|
||||
{domain: 'shop.example.com', match_subdomains: 'true', pattern: 'false'},
|
||||
]),
|
||||
);
|
||||
});
|
||||
|
||||
it('rejects patterns that are too broad or malformed', async () => {
|
||||
for (const domain of ['*', '*.com', '*shop*.co.uk', '*.onrender.com', '*ab*.onrender.com', 'shop.*.example.com']) {
|
||||
const json = await createBuilder<ValidationErrorResponse>(harness, admin.token)
|
||||
.post('/admin/blocklists/url-domain/entries')
|
||||
.body({domain})
|
||||
.expect(400, 'INVALID_FORM_BODY')
|
||||
.execute();
|
||||
expect(json.errors?.[0]?.path, domain).toBe('domain');
|
||||
}
|
||||
expect(await list()).toEqual([]);
|
||||
});
|
||||
|
||||
it('validates the value on update', async () => {
|
||||
const json = await createBuilder<ValidationErrorResponse>(harness, admin.token)
|
||||
.patch(`/admin/blocklists/url-domain/entries/${encodeURIComponent('*.com')}`)
|
||||
.body({})
|
||||
.expect(400, 'INVALID_FORM_BODY')
|
||||
.execute();
|
||||
expect(json.errors?.[0]?.path).toBe('domain');
|
||||
});
|
||||
|
||||
it('stores internationalized domains in ASCII form', async () => {
|
||||
await add('Bücher.Example.');
|
||||
expect((await list()).map((entry) => entry.value)).toEqual(['xn--bcher-kva.example']);
|
||||
expect(await check('www.bücher.example')).toBe(true);
|
||||
});
|
||||
|
||||
it('accepts an add for a domain that is already blocked', async () => {
|
||||
await add('shop.example.com');
|
||||
await add('shop.example.com', false);
|
||||
expect(await list()).toMatchObject([{value: 'shop.example.com', match_subdomains: false}]);
|
||||
});
|
||||
|
||||
it('removes a pattern through any spelling that canonicalizes to it', async () => {
|
||||
await add('*shop*.onrender.com');
|
||||
await createBuilder(harness, admin.token)
|
||||
.delete(`/admin/blocklists/url-domain/entries/${encodeURIComponent('*SHOP**.onrender.com')}`)
|
||||
.expect(204)
|
||||
.execute();
|
||||
expect(await list()).toEqual([]);
|
||||
expect(await check('shop.onrender.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('blocks messages whose masked links or autolinks point at a covered host', async () => {
|
||||
await add('*shop*.onrender.com');
|
||||
const member = await createTestAccount(harness);
|
||||
const guild = await createGuild(harness, member.token, 'Links');
|
||||
const channel = await createChannel(harness, member.token, guild.id, 'general');
|
||||
await ensureSessionStarted(harness, member.token);
|
||||
for (const content of [
|
||||
'[open the store](https://shop-2.onrender.com)',
|
||||
'<https://[email protected]:8443/x>',
|
||||
'https://SHOP.onrender.com./',
|
||||
]) {
|
||||
await createBuilder(harness, member.token)
|
||||
.post(`/channels/${channel.id}/messages`)
|
||||
.body({content})
|
||||
.expect(403, APIErrorCodes.CONTENT_BLOCKED)
|
||||
.execute();
|
||||
}
|
||||
await createBuilder(harness, member.token)
|
||||
.post(`/channels/${channel.id}/messages`)
|
||||
.body({content: '[docs](https://docs.onrender.com) and https://onrender.com'})
|
||||
.expect(200)
|
||||
.execute();
|
||||
});
|
||||
|
||||
it('blocks rich embeds that link to a covered host', async () => {
|
||||
await add('*shop*.onrender.com');
|
||||
const member = await createTestAccount(harness);
|
||||
const guild = await createGuild(harness, member.token, 'Embeds');
|
||||
const channel = await createChannel(harness, member.token, guild.id, 'general');
|
||||
await ensureSessionStarted(harness, member.token);
|
||||
await createBuilder(harness, member.token)
|
||||
.post(`/channels/${channel.id}/messages`)
|
||||
.body({embeds: [{title: 'Store', url: 'https://shop.onrender.com/'}]})
|
||||
.expect(403, APIErrorCodes.CONTENT_BLOCKED)
|
||||
.execute();
|
||||
});
|
||||
});
|
||||
@@ -33,7 +33,7 @@ function attachment(id: bigint, hash: string | null): MessageAttachment {
|
||||
};
|
||||
}
|
||||
|
||||
function message(attachments: Array<MessageAttachment>): Message {
|
||||
function message(attachments: Array<MessageAttachment>, content = ''): Message {
|
||||
return new Message({
|
||||
channel_id: createChannelID(10n),
|
||||
bucket: 0,
|
||||
@@ -43,7 +43,7 @@ function message(attachments: Array<MessageAttachment>): Message {
|
||||
webhook_id: null,
|
||||
webhook_name: null,
|
||||
webhook_avatar_hash: null,
|
||||
content: '',
|
||||
content,
|
||||
edited_timestamp: null,
|
||||
pinned_timestamp: null,
|
||||
flags: 0,
|
||||
@@ -75,10 +75,10 @@ describe('message activity', () => {
|
||||
resetActivityEventsForTests();
|
||||
});
|
||||
|
||||
function params(attachments: Array<MessageAttachment>) {
|
||||
function params(attachments: Array<MessageAttachment>, content = '') {
|
||||
return {
|
||||
user: {id: createUserID(3n), isBot: false} as unknown as User,
|
||||
message: message(attachments),
|
||||
message: message(attachments, content),
|
||||
channel: {id: createChannelID(10n), type: ChannelTypes.DM} as unknown as Channel,
|
||||
guildId: null,
|
||||
guildOwnerId: null,
|
||||
@@ -117,4 +117,18 @@ describe('message activity', () => {
|
||||
expect(updated.data).toMatchObject({message_id: '100', attachments: [{hash: HASH.toLowerCase()}]});
|
||||
expect(updated.id).not.toBe(created.id);
|
||||
});
|
||||
|
||||
it('records the link domain of a masked markdown link without its brackets', async () => {
|
||||
const publisher = new CapturingPublisher();
|
||||
await startActivityEvents({publisher, kv: new MockKVProvider()});
|
||||
emitMessageCreated(
|
||||
params(
|
||||
[],
|
||||
'[OPEN](https://Shop.Example.com) [docs](<https://www.docs.example.org/a>) https://[email protected]:8443/x',
|
||||
),
|
||||
);
|
||||
await vi.waitFor(() => expect(publisher.payloads).toHaveLength(1));
|
||||
const event = JSON.parse(publisher.payloads[0]!);
|
||||
expect(event.data.link_domains).toEqual(['shop.example.com', 'docs.example.org', 'cdn.example.net']);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -10,22 +10,20 @@ import type {Message} from '@app/api/models/Message';
|
||||
import type {User} from '@app/api/models/User';
|
||||
import type {IUserRepository} from '@app/api/user/IUserRepository';
|
||||
import {findInvites} from '@app/api/utils/InviteUtils';
|
||||
import {extractLinkHosts} from '@app/api/utils/UrlNormalizer';
|
||||
import {ChannelTypes} from '@fluxer/constants/src/ChannelConstants';
|
||||
import {RelationshipTypes} from '@fluxer/constants/src/UserConstants';
|
||||
|
||||
const CONTENT_MAX_CHARS = 2000;
|
||||
const LIST_MAX = 10;
|
||||
const MENTIONS_MAX = 20;
|
||||
const LINK_PATTERN = /https?:\/\/([^\s/?#<>"']+)/giu;
|
||||
const WWW_PREFIX_RE = /^www\./u;
|
||||
|
||||
function linkDomains(content: string): Array<string> {
|
||||
const domains = new Set<string>();
|
||||
for (const match of content.matchAll(LINK_PATTERN)) {
|
||||
const host = match[1]
|
||||
?.toLowerCase()
|
||||
.replace(/:\d+$/u, '')
|
||||
.replace(/^www\./u, '');
|
||||
if (host) domains.add(host);
|
||||
for (const host of extractLinkHosts(content)) {
|
||||
const domain = host.replace(WWW_PREFIX_RE, '');
|
||||
if (domain) domains.add(domain);
|
||||
if (domains.size >= LIST_MAX) break;
|
||||
}
|
||||
return [...domains];
|
||||
|
||||
@@ -94,10 +94,14 @@ export class MessageValidationService {
|
||||
contentModerationService.scanText(data.content, modCtx);
|
||||
if (data.embeds) {
|
||||
for (const embed of data.embeds) {
|
||||
if (embed.url) contentModerationService.scanUrl(embed.url, modCtx);
|
||||
contentModerationService.scanText(embed.title ?? null, modCtx);
|
||||
contentModerationService.scanText(embed.description ?? null, modCtx);
|
||||
if (embed.footer) contentModerationService.scanText(embed.footer.text ?? null, modCtx);
|
||||
if (embed.author) contentModerationService.scanText(embed.author.name ?? null, modCtx);
|
||||
if (embed.author) {
|
||||
contentModerationService.scanText(embed.author.name ?? null, modCtx);
|
||||
if (embed.author.url) contentModerationService.scanUrl(embed.author.url, modCtx);
|
||||
}
|
||||
if (embed.fields) {
|
||||
for (const field of embed.fields) {
|
||||
contentModerationService.scanText(field.name ?? null, modCtx);
|
||||
|
||||
@@ -5,7 +5,6 @@ import {Logger} from '@app/api/Logger';
|
||||
import {fileShaCache} from '@app/api/middleware/FileShaCache';
|
||||
import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache';
|
||||
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
|
||||
import {extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
|
||||
import {ContentBlockedError} from '@fluxer/errors/src/domains/content/ContentBlockedError';
|
||||
|
||||
export interface ModerationContext {
|
||||
@@ -40,16 +39,12 @@ class ContentModerationService {
|
||||
);
|
||||
throw new ContentBlockedError();
|
||||
}
|
||||
const urls = extractUrlCandidates(text);
|
||||
if (urls.length === 0) return;
|
||||
for (const url of urls) {
|
||||
if (urlBlocklistCache.isUrlOrDomainBanned(url)) {
|
||||
Logger.warn(
|
||||
{surface: ctx.surface, userId: ctx.userId?.toString(), guildId: ctx.guildId?.toString()},
|
||||
'content_moderation.block url match in text',
|
||||
);
|
||||
throw new ContentBlockedError();
|
||||
}
|
||||
if (urlBlocklistCache.containsBannedLink(text)) {
|
||||
Logger.warn(
|
||||
{surface: ctx.surface, userId: ctx.userId?.toString(), guildId: ctx.guildId?.toString()},
|
||||
'content_moderation.block url match in text',
|
||||
);
|
||||
throw new ContentBlockedError();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -4,7 +4,6 @@ import {Logger} from '@app/api/Logger';
|
||||
import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache';
|
||||
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
|
||||
import {readRequestJsonBody} from '@app/api/utils/RequestJsonBody';
|
||||
import {extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
|
||||
import {ContentBlockedError} from '@fluxer/errors/src/domains/content/ContentBlockedError';
|
||||
import {createMiddleware} from 'hono/factory';
|
||||
|
||||
@@ -73,6 +72,8 @@ const SKIP_FIELD_SUFFIXES = [
|
||||
] as const;
|
||||
const SKIP_CONTENT_FILTER_PATH_PARTS = [
|
||||
'/admin/blocklists/phrase/',
|
||||
'/admin/blocklists/url-domain/',
|
||||
'/admin/blocklists/url/',
|
||||
'/auth/',
|
||||
'/oauth2/',
|
||||
'/premium/store/',
|
||||
@@ -149,15 +150,12 @@ const ContentFilterMiddleware = createMiddleware(async (ctx, next) => {
|
||||
);
|
||||
throw new ContentBlockedError();
|
||||
}
|
||||
const urls = extractUrlCandidates(text);
|
||||
for (const url of urls) {
|
||||
if (urlBlocklistCache.isUrlOrDomainBanned(url)) {
|
||||
Logger.warn(
|
||||
{surface: 'global_filter', userId: userId?.toString(), path},
|
||||
'content_moderation.block url match in request body',
|
||||
);
|
||||
throw new ContentBlockedError();
|
||||
}
|
||||
if (urlBlocklistCache.containsBannedLink(text)) {
|
||||
Logger.warn(
|
||||
{surface: 'global_filter', userId: userId?.toString(), path},
|
||||
'content_moderation.block url match in request body',
|
||||
);
|
||||
throw new ContentBlockedError();
|
||||
}
|
||||
}
|
||||
return next();
|
||||
|
||||
@@ -7,12 +7,13 @@ import {BANNED_URL_DOMAINS_REFRESH_CHANNEL, BANNED_URLS_REFRESH_CHANNEL} from '@
|
||||
import type {IStorageService} from '@app/api/infrastructure/IStorageService';
|
||||
import {Logger} from '@app/api/Logger';
|
||||
import {RefreshSubscription} from '@app/api/utils/RefreshSubscription';
|
||||
import {canonicalizeUrl} from '@app/api/utils/UrlNormalizer';
|
||||
import {UrlHostRuleSet} from '@app/api/utils/UrlHostRules';
|
||||
import {canonicalizeUrl, extractLinkHosts, extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
|
||||
import type {IKVProvider} from '@pkgs/kv_client/src/IKVProvider';
|
||||
|
||||
class UrlBlocklistCache {
|
||||
private exactUrls: Set<string> = new Set();
|
||||
private blockedDomains: Set<string> = new Set();
|
||||
private hostRules = new UrlHostRuleSet();
|
||||
private adminRepository = new AdminRepository();
|
||||
private kvClient: IKVProvider | null = null;
|
||||
private storageService: IStorageService | null = null;
|
||||
@@ -55,15 +56,15 @@ class UrlBlocklistCache {
|
||||
for (const row of manualUrls) {
|
||||
if (row.url_canonical) nextUrls.add(row.url_canonical.toLowerCase());
|
||||
}
|
||||
const nextDomains = new Set<string>();
|
||||
const nextHostRules = new UrlHostRuleSet();
|
||||
for (const row of domains) {
|
||||
nextDomains.add(row.domain.toLowerCase());
|
||||
nextHostRules.add(row.domain, row.match_subdomains ?? true);
|
||||
}
|
||||
this.exactUrls = nextUrls;
|
||||
this.blockedDomains = nextDomains;
|
||||
this.hostRules = nextHostRules;
|
||||
this.consecutiveFailures = 0;
|
||||
Logger.debug(
|
||||
{urls: nextUrls.size, domains: nextDomains.size, feedUrls: feedUrls.size},
|
||||
{urls: nextUrls.size, ...nextHostRules.size, feedUrls: feedUrls.size},
|
||||
'URL blocklist cache refreshed',
|
||||
);
|
||||
}
|
||||
@@ -94,7 +95,17 @@ class UrlBlocklistCache {
|
||||
}
|
||||
|
||||
isHostnameBanned(host: string): boolean {
|
||||
return this.blockedDomains.has(host.toLowerCase());
|
||||
return this.hostRules.matches(host);
|
||||
}
|
||||
|
||||
containsBannedLink(text: string): boolean {
|
||||
for (const url of extractUrlCandidates(text)) {
|
||||
if (this.isUrlOrDomainBanned(url)) return true;
|
||||
}
|
||||
for (const host of extractLinkHosts(text)) {
|
||||
if (this.isHostnameBanned(host)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
addExactUrl(canonical: string): void {
|
||||
@@ -105,21 +116,22 @@ class UrlBlocklistCache {
|
||||
this.exactUrls.delete(canonical.toLowerCase());
|
||||
}
|
||||
|
||||
addDomain(domain: string): void {
|
||||
this.blockedDomains.add(domain.toLowerCase());
|
||||
addDomain(domain: string, matchSubdomains = true): void {
|
||||
this.hostRules.add(domain, matchSubdomains);
|
||||
}
|
||||
|
||||
removeDomain(domain: string): void {
|
||||
this.blockedDomains.delete(domain.toLowerCase());
|
||||
this.hostRules.remove(domain);
|
||||
}
|
||||
|
||||
get size(): {
|
||||
urls: number;
|
||||
domains: number;
|
||||
patterns: number;
|
||||
} {
|
||||
return {
|
||||
urls: this.exactUrls.size,
|
||||
domains: this.blockedDomains.size,
|
||||
...this.hostRules.size,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -128,7 +140,7 @@ class UrlBlocklistCache {
|
||||
Logger.error({error}, 'Failed to shut down URL blocklist cache');
|
||||
});
|
||||
this.exactUrls = new Set();
|
||||
this.blockedDomains = new Set();
|
||||
this.hostRules = new UrlHostRuleSet();
|
||||
this.kvClient = null;
|
||||
this.storageService = null;
|
||||
this.consecutiveFailures = 0;
|
||||
|
||||
@@ -81,4 +81,14 @@ describe('shouldSkipContentFilterPath', () => {
|
||||
const result = paths.map((path) => shouldSkipContentFilterPath(path));
|
||||
expect(result).toEqual([false, false, false]);
|
||||
});
|
||||
test('skips blocklist writes whose values are the blocked content', () => {
|
||||
const paths = [
|
||||
'/admin/blocklists/phrase/entries',
|
||||
'/admin/blocklists/url/entries',
|
||||
'/admin/blocklists/url-domain/entries',
|
||||
'/admin/blocklists/profile-substring/entries',
|
||||
];
|
||||
const result = paths.map((path) => shouldSkipContentFilterPath(path));
|
||||
expect(result).toEqual([true, true, true, false]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache';
|
||||
import {afterEach, describe, expect, it} from 'vitest';
|
||||
|
||||
describe('urlBlocklistCache link matching', () => {
|
||||
afterEach(() => {
|
||||
urlBlocklistCache.resetForTesting();
|
||||
});
|
||||
|
||||
it('blocks a masked markdown link whose target a pattern covers', () => {
|
||||
urlBlocklistCache.addDomain('*shop*.onrender.com', true);
|
||||
expect(urlBlocklistCache.containsBannedLink('[open the store](https://shop-2.onrender.com)')).toBe(true);
|
||||
expect(urlBlocklistCache.containsBannedLink('[open the store](<https://www.shop.onrender.com/x>)')).toBe(true);
|
||||
expect(urlBlocklistCache.containsBannedLink('[https://docs.onrender.com](https://shop.onrender.com)')).toBe(true);
|
||||
});
|
||||
|
||||
it('blocks autolinks and bare links', () => {
|
||||
urlBlocklistCache.addDomain('*shop*.onrender.com', false);
|
||||
expect(urlBlocklistCache.containsBannedLink('<https://myshop.onrender.com/path>')).toBe(true);
|
||||
expect(urlBlocklistCache.containsBannedLink('visit myshop.onrender.com today')).toBe(true);
|
||||
});
|
||||
|
||||
it('normalizes the link target before matching', () => {
|
||||
urlBlocklistCache.addDomain('*shop*.onrender.com', false);
|
||||
const variants = [
|
||||
'https://user:[email protected]:8443/x',
|
||||
'https://[email protected]',
|
||||
'https://shop.onrender.com./',
|
||||
'https://shop%2Eonrender%2Ecom/',
|
||||
'https://shop。onrender。com/',
|
||||
'https://shop-ü.onrender.com/',
|
||||
];
|
||||
for (const text of variants) {
|
||||
expect(urlBlocklistCache.containsBannedLink(text), text).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('leaves the bare suffix and unrelated hosts alone', () => {
|
||||
urlBlocklistCache.addDomain('*shop*.onrender.com', true);
|
||||
expect(urlBlocklistCache.containsBannedLink('https://onrender.com/docs')).toBe(false);
|
||||
expect(urlBlocklistCache.containsBannedLink('[docs](https://docs.onrender.com)')).toBe(false);
|
||||
expect(urlBlocklistCache.containsBannedLink('the shop is closed')).toBe(false);
|
||||
});
|
||||
|
||||
it('covers subdomains of a domain entry only when it is flagged to', () => {
|
||||
urlBlocklistCache.addDomain('shop.example.com', true);
|
||||
urlBlocklistCache.addDomain('store.example.com', false);
|
||||
expect(urlBlocklistCache.containsBannedLink('https://www.shop.example.com')).toBe(true);
|
||||
expect(urlBlocklistCache.containsBannedLink('https://store.example.com')).toBe(true);
|
||||
expect(urlBlocklistCache.containsBannedLink('https://www.store.example.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('stops matching after removal', () => {
|
||||
urlBlocklistCache.addDomain('*shop*.onrender.com', true);
|
||||
urlBlocklistCache.removeDomain('*shop*.onrender.com');
|
||||
expect(urlBlocklistCache.containsBannedLink('https://shop.onrender.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('applies domain rules to a single URL', () => {
|
||||
urlBlocklistCache.addDomain('*shop*.onrender.com', false);
|
||||
expect(urlBlocklistCache.isUrlOrDomainBanned('https://shop.onrender.com/checkout')).toBe(true);
|
||||
expect(urlBlocklistCache.isUrlOrDomainBanned('https://docs.onrender.com/')).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,167 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import {normalizeHostname} from '@app/api/utils/UrlNormalizer';
|
||||
import {getDomain} from 'tldts';
|
||||
|
||||
const URL_HOST_PATTERN_MAX_WILDCARDS = 3;
|
||||
const URL_HOST_PATTERN_MIN_LITERAL_CHARS = 3;
|
||||
|
||||
const MAX_HOSTNAME_LENGTH = 253;
|
||||
const HOSTNAME_LABEL_RE = /^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/;
|
||||
const GLOB_LABEL_RE = /^[a-z0-9*-]{1,63}$/;
|
||||
const REPEATED_WILDCARDS_RE = /\*+/g;
|
||||
|
||||
export type UrlDomainEntry =
|
||||
| {ok: true; value: string; pattern: boolean}
|
||||
| {
|
||||
ok: false;
|
||||
message: string;
|
||||
};
|
||||
|
||||
function isValidHostname(host: string): boolean {
|
||||
if (host.length > MAX_HOSTNAME_LENGTH) return false;
|
||||
return host.split('.').every((label) => HOSTNAME_LABEL_RE.test(label));
|
||||
}
|
||||
|
||||
function invalid(message: string): UrlDomainEntry {
|
||||
return {ok: false, message};
|
||||
}
|
||||
|
||||
export function parseUrlDomainEntry(raw: string): UrlDomainEntry {
|
||||
const value = raw.trim().toLowerCase();
|
||||
if (!value.includes('*')) {
|
||||
const host = normalizeHostname(value);
|
||||
if (!host?.includes('.') || !isValidHostname(host)) {
|
||||
return invalid('Must be a valid domain');
|
||||
}
|
||||
return {ok: true, value: host, pattern: false};
|
||||
}
|
||||
const separator = value.indexOf('.');
|
||||
if (separator === -1) {
|
||||
return invalid('A pattern must name a domain after its wildcard label');
|
||||
}
|
||||
const glob = value.slice(0, separator).replace(REPEATED_WILDCARDS_RE, '*');
|
||||
const suffixValue = value.slice(separator + 1);
|
||||
if (suffixValue.includes('*')) {
|
||||
return invalid('Only the leftmost label of a pattern can contain *');
|
||||
}
|
||||
if (!GLOB_LABEL_RE.test(glob)) {
|
||||
return invalid('The wildcard label can contain only a-z, 0-9, hyphens, and *');
|
||||
}
|
||||
const wildcards = glob.length - glob.replaceAll('*', '').length;
|
||||
if (wildcards > URL_HOST_PATTERN_MAX_WILDCARDS) {
|
||||
return invalid(`The wildcard label can contain at most ${URL_HOST_PATTERN_MAX_WILDCARDS} *`);
|
||||
}
|
||||
if (glob.length - wildcards < URL_HOST_PATTERN_MIN_LITERAL_CHARS) {
|
||||
return invalid(
|
||||
`The wildcard label needs at least ${URL_HOST_PATTERN_MIN_LITERAL_CHARS} characters besides *, so the pattern is too broad`,
|
||||
);
|
||||
}
|
||||
const suffix = normalizeHostname(suffixValue);
|
||||
if (!suffix || !isValidHostname(suffix)) {
|
||||
return invalid('The part after the wildcard label must be a valid domain');
|
||||
}
|
||||
if (getDomain(suffix, {allowPrivateDomains: false}) === null) {
|
||||
return invalid('The part after the wildcard label is a public suffix, so the pattern is too broad');
|
||||
}
|
||||
const entry = `${glob}.${suffix}`;
|
||||
if (entry.length > MAX_HOSTNAME_LENGTH) {
|
||||
return invalid('Must be a valid domain');
|
||||
}
|
||||
return {ok: true, value: entry, pattern: true};
|
||||
}
|
||||
|
||||
interface HostPatternRule {
|
||||
value: string;
|
||||
segments: ReadonlyArray<string>;
|
||||
matchSubdomains: boolean;
|
||||
}
|
||||
|
||||
function globMatches(segments: ReadonlyArray<string>, label: string): boolean {
|
||||
const first = segments[0] ?? '';
|
||||
const last = segments[segments.length - 1] ?? '';
|
||||
if (!label.startsWith(first)) return false;
|
||||
let position = first.length;
|
||||
for (let index = 1; index < segments.length - 1; index++) {
|
||||
const segment = segments[index] ?? '';
|
||||
const found = label.indexOf(segment, position);
|
||||
if (found === -1) return false;
|
||||
position = found + segment.length;
|
||||
}
|
||||
return label.length - last.length >= position && label.endsWith(last);
|
||||
}
|
||||
|
||||
export class UrlHostRuleSet {
|
||||
private readonly domains = new Map<string, boolean>();
|
||||
private readonly patterns = new Map<string, Array<HostPatternRule>>();
|
||||
private patternCount = 0;
|
||||
|
||||
add(rawValue: string, matchSubdomains: boolean): void {
|
||||
if (rawValue.includes('*')) {
|
||||
const entry = parseUrlDomainEntry(rawValue);
|
||||
if (!entry.ok || !entry.pattern) return;
|
||||
this.remove(entry.value);
|
||||
const separator = entry.value.indexOf('.');
|
||||
const suffix = entry.value.slice(separator + 1);
|
||||
const rules = this.patterns.get(suffix) ?? [];
|
||||
rules.push({
|
||||
value: entry.value,
|
||||
segments: entry.value.slice(0, separator).split('*'),
|
||||
matchSubdomains,
|
||||
});
|
||||
this.patterns.set(suffix, rules);
|
||||
this.patternCount++;
|
||||
return;
|
||||
}
|
||||
const host = normalizeHostname(rawValue);
|
||||
if (host) this.domains.set(host, matchSubdomains);
|
||||
}
|
||||
|
||||
remove(rawValue: string): void {
|
||||
if (!rawValue.includes('*')) {
|
||||
const host = normalizeHostname(rawValue);
|
||||
if (host) this.domains.delete(host);
|
||||
return;
|
||||
}
|
||||
const entry = parseUrlDomainEntry(rawValue);
|
||||
if (!entry.ok) return;
|
||||
const suffix = entry.value.slice(entry.value.indexOf('.') + 1);
|
||||
const rules = this.patterns.get(suffix);
|
||||
if (!rules) return;
|
||||
const remaining = rules.filter((rule) => rule.value !== entry.value);
|
||||
this.patternCount -= rules.length - remaining.length;
|
||||
if (remaining.length === 0) {
|
||||
this.patterns.delete(suffix);
|
||||
} else {
|
||||
this.patterns.set(suffix, remaining);
|
||||
}
|
||||
}
|
||||
|
||||
matches(rawHost: string): boolean {
|
||||
const host = normalizeHostname(rawHost);
|
||||
if (!host) return false;
|
||||
if (this.domains.has(host)) return true;
|
||||
let labelStart = 0;
|
||||
let isFirstLabel = true;
|
||||
while (labelStart < host.length) {
|
||||
const separator = host.indexOf('.', labelStart);
|
||||
if (separator === -1) return false;
|
||||
const suffix = host.slice(separator + 1);
|
||||
if (this.domains.get(suffix) === true) return true;
|
||||
const rules = this.patterns.get(suffix);
|
||||
if (rules) {
|
||||
const label = host.slice(labelStart, separator);
|
||||
for (const rule of rules) {
|
||||
if ((isFirstLabel || rule.matchSubdomains) && globMatches(rule.segments, label)) return true;
|
||||
}
|
||||
}
|
||||
labelStart = separator + 1;
|
||||
isFirstLabel = false;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
get size(): {domains: number; patterns: number} {
|
||||
return {domains: this.domains.size, patterns: this.patternCount};
|
||||
}
|
||||
}
|
||||
@@ -58,6 +58,17 @@ export function canonicalizeUrl(raw: string): string | null {
|
||||
return parsed.toString().toLowerCase();
|
||||
}
|
||||
|
||||
const TRAILING_DOTS_RE = /\.+$/;
|
||||
|
||||
export function normalizeHostname(raw: string): string | null {
|
||||
const trimmed = raw.trim();
|
||||
if (!trimmed) return null;
|
||||
const ascii = domainToASCII(trimmed);
|
||||
if (!ascii) return null;
|
||||
const host = ascii.toLowerCase().replace(TRAILING_DOTS_RE, '');
|
||||
return host || null;
|
||||
}
|
||||
|
||||
const URL_CANDIDATE_RE =
|
||||
/(?<![a-z0-9._+-]@)((?:https?:\/\/)?(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,}(?:[/:?#][^\s<>\u201D']*)?)/gi;
|
||||
const TRAILING_PUNCT_RE = /[.,;:!?)\]}\x22'\u00bb\u201C\u201D]+$/;
|
||||
@@ -78,3 +89,29 @@ export function extractUrlCandidates(text: string | null | undefined): Array<str
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const LINK_AUTHORITY_RE = /https?:\/\/([^\s/\\?#<>"'`()[\]{}|^]+)/giu;
|
||||
|
||||
function hostFromAuthority(authority: string): string | null {
|
||||
const cleaned = authority.replace(TRAILING_PUNCT_RE, '');
|
||||
if (!cleaned) return null;
|
||||
let parsed: URL;
|
||||
try {
|
||||
parsed = new URL(`http://${cleaned}/`);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
return normalizeHostname(parsed.hostname);
|
||||
}
|
||||
|
||||
export function extractLinkHosts(text: string | null | undefined): Array<string> {
|
||||
if (!text) return [];
|
||||
const hosts = new Set<string>();
|
||||
for (const match of text.matchAll(LINK_AUTHORITY_RE)) {
|
||||
const authority = match[1];
|
||||
if (!authority) continue;
|
||||
const host = hostFromAuthority(authority);
|
||||
if (host) hosts.add(host);
|
||||
}
|
||||
return [...hosts];
|
||||
}
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import {parseUrlDomainEntry, UrlHostRuleSet} from '@app/api/utils/UrlHostRules';
|
||||
import {describe, expect, it} from 'vitest';
|
||||
|
||||
function rules(entries: Array<[string, boolean]>): UrlHostRuleSet {
|
||||
const set = new UrlHostRuleSet();
|
||||
for (const [value, matchSubdomains] of entries) {
|
||||
set.add(value, matchSubdomains);
|
||||
}
|
||||
return set;
|
||||
}
|
||||
|
||||
describe('parseUrlDomainEntry', () => {
|
||||
it('canonicalizes a plain domain', () => {
|
||||
expect(parseUrlDomainEntry(' Shop.Example.COM. ')).toEqual({ok: true, value: 'shop.example.com', pattern: false});
|
||||
});
|
||||
|
||||
it('stores an internationalized domain in punycode', () => {
|
||||
expect(parseUrlDomainEntry('bücher.example')).toEqual({ok: true, value: 'xn--bcher-kva.example', pattern: false});
|
||||
});
|
||||
|
||||
it('keeps an exact entry for a domain under a shared hosting suffix', () => {
|
||||
expect(parseUrlDomainEntry('onrender.com')).toEqual({ok: true, value: 'onrender.com', pattern: false});
|
||||
});
|
||||
|
||||
it('rejects malformed plain domains', () => {
|
||||
for (const value of ['localhost', 'bad_label.example.com', '-lead.example.com', 'a..example.com', 'a b.com']) {
|
||||
expect(parseUrlDomainEntry(value).ok).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
it('canonicalizes a pattern and collapses repeated wildcards', () => {
|
||||
expect(parseUrlDomainEntry('**Shop**.OnRender.com.')).toEqual({
|
||||
ok: true,
|
||||
value: '*shop*.onrender.com',
|
||||
pattern: true,
|
||||
});
|
||||
});
|
||||
|
||||
it('accepts patterns under a private shared hosting suffix', () => {
|
||||
expect(parseUrlDomainEntry('*shop*.github.io').ok).toBe(true);
|
||||
expect(parseUrlDomainEntry('shop-*.example.co.uk').ok).toBe(true);
|
||||
});
|
||||
|
||||
it('rejects patterns that are too broad', () => {
|
||||
for (const value of [
|
||||
'*',
|
||||
'*.com',
|
||||
'*shop*.com',
|
||||
'shop*.co.uk',
|
||||
'*.example.com',
|
||||
'*ab*.example.com',
|
||||
'a*b.example.com',
|
||||
]) {
|
||||
expect(parseUrlDomainEntry(value).ok).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects wildcards outside the leftmost label', () => {
|
||||
expect(parseUrlDomainEntry('shop.*.example.com').ok).toBe(false);
|
||||
expect(parseUrlDomainEntry('*shop*.example*.com').ok).toBe(false);
|
||||
});
|
||||
|
||||
it('rejects wildcard labels with unsupported characters or too many wildcards', () => {
|
||||
expect(parseUrlDomainEntry('*sh?p*.example.com').ok).toBe(false);
|
||||
expect(parseUrlDomainEntry('*sh_p*.example.com').ok).toBe(false);
|
||||
expect(parseUrlDomainEntry('*a*b*c*d*.example.com').ok).toBe(false);
|
||||
});
|
||||
|
||||
it('rejects a pattern without a domain', () => {
|
||||
expect(parseUrlDomainEntry('*shop*').ok).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('UrlHostRuleSet', () => {
|
||||
it('matches an exact domain', () => {
|
||||
const set = rules([['shop.example.com', false]]);
|
||||
expect(set.matches('shop.example.com')).toBe(true);
|
||||
expect(set.matches('www.shop.example.com')).toBe(false);
|
||||
expect(set.matches('example.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('matches subdomains only when the entry covers them', () => {
|
||||
const set = rules([['shop.example.com', true]]);
|
||||
expect(set.matches('shop.example.com')).toBe(true);
|
||||
expect(set.matches('a.b.shop.example.com')).toBe(true);
|
||||
expect(set.matches('myshop.example.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('normalizes the checked host', () => {
|
||||
const set = rules([['shop.example.com', true]]);
|
||||
expect(set.matches('SHOP.Example.com.')).toBe(true);
|
||||
expect(set.matches('www.shop。example。com')).toBe(true);
|
||||
expect(rules([['xn--bcher-kva.example', false]]).matches('Bücher.example')).toBe(true);
|
||||
});
|
||||
|
||||
it('matches a pattern against the label left of its suffix', () => {
|
||||
const set = rules([['*shop*.onrender.com', false]]);
|
||||
expect(set.matches('shop.onrender.com')).toBe(true);
|
||||
expect(set.matches('best-shop-2.onrender.com')).toBe(true);
|
||||
expect(set.matches('SHOPPING.onrender.com')).toBe(true);
|
||||
expect(set.matches('store.onrender.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('never matches the bare suffix of a pattern', () => {
|
||||
const set = rules([['*shop*.onrender.com', true]]);
|
||||
expect(set.matches('onrender.com')).toBe(false);
|
||||
expect(set.matches('com')).toBe(false);
|
||||
expect(set.matches('shop.com')).toBe(false);
|
||||
expect(set.matches('shop.onrender.com.evil.example')).toBe(false);
|
||||
expect(set.matches('shoponrender.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('applies anchored pattern segments', () => {
|
||||
const set = rules([
|
||||
['shop-*.example.com', false],
|
||||
['*-store.example.org', false],
|
||||
['a*b*c.example.net', false],
|
||||
]);
|
||||
expect(set.matches('shop-1.example.com')).toBe(true);
|
||||
expect(set.matches('myshop-1.example.com')).toBe(false);
|
||||
expect(set.matches('big-store.example.org')).toBe(true);
|
||||
expect(set.matches('big-store2.example.org')).toBe(false);
|
||||
expect(set.matches('axxbyyc.example.net')).toBe(true);
|
||||
expect(set.matches('abc.example.net')).toBe(true);
|
||||
expect(set.matches('acb.example.net')).toBe(false);
|
||||
});
|
||||
|
||||
it('extends a pattern to deeper subdomains only when the entry covers them', () => {
|
||||
const exact = rules([['*shop*.onrender.com', false]]);
|
||||
const covering = rules([['*shop*.onrender.com', true]]);
|
||||
expect(exact.matches('www.shop.onrender.com')).toBe(false);
|
||||
expect(covering.matches('www.shop.onrender.com')).toBe(true);
|
||||
expect(covering.matches('shop.www.onrender.com')).toBe(false);
|
||||
});
|
||||
|
||||
it('matches internationalized labels through their punycode form', () => {
|
||||
const set = rules([['*shop*.onrender.com', false]]);
|
||||
expect(set.matches('shop-ü.onrender.com')).toBe(true);
|
||||
});
|
||||
|
||||
it('removes domains and patterns', () => {
|
||||
const set = rules([
|
||||
['shop.example.com', true],
|
||||
['*shop*.onrender.com', true],
|
||||
['*store*.onrender.com', true],
|
||||
]);
|
||||
set.remove('SHOP.example.com');
|
||||
set.remove('**shop*.onrender.com');
|
||||
expect(set.matches('shop.example.com')).toBe(false);
|
||||
expect(set.matches('shop.onrender.com')).toBe(false);
|
||||
expect(set.matches('store.onrender.com')).toBe(true);
|
||||
expect(set.size).toEqual({domains: 0, patterns: 1});
|
||||
});
|
||||
|
||||
it('replaces a pattern when it is added again', () => {
|
||||
const set = rules([
|
||||
['*shop*.onrender.com', false],
|
||||
['*shop*.onrender.com', true],
|
||||
]);
|
||||
expect(set.size).toEqual({domains: 0, patterns: 1});
|
||||
expect(set.matches('www.shop.onrender.com')).toBe(true);
|
||||
});
|
||||
|
||||
it('ignores invalid stored patterns', () => {
|
||||
const set = rules([['*.com', true]]);
|
||||
expect(set.size).toEqual({domains: 0, patterns: 0});
|
||||
expect(set.matches('anything.com')).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -1,6 +1,6 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import {canonicalizeUrl, extractUrlCandidates} from '@app/api/utils/UrlNormalizer';
|
||||
import {canonicalizeUrl, extractLinkHosts, extractUrlCandidates, normalizeHostname} from '@app/api/utils/UrlNormalizer';
|
||||
import {describe, expect, it} from 'vitest';
|
||||
|
||||
describe('canonicalizeUrl', () => {
|
||||
@@ -176,3 +176,61 @@ describe('extractUrlCandidates', () => {
|
||||
expect(extractUrlCandidates('hello world no links here')).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('normalizeHostname', () => {
|
||||
it('lowercases and strips trailing dots', () => {
|
||||
expect(normalizeHostname('Shop.Example.COM..')).toBe('shop.example.com');
|
||||
});
|
||||
it('converts internationalized names to punycode', () => {
|
||||
expect(normalizeHostname('bücher.example')).toBe('xn--bcher-kva.example');
|
||||
});
|
||||
it('maps ideographic full stops to dots', () => {
|
||||
expect(normalizeHostname('shop\u3002example\u3002com')).toBe('shop.example.com');
|
||||
});
|
||||
it('rejects empty and invalid input', () => {
|
||||
expect(normalizeHostname(' ')).toBeNull();
|
||||
expect(normalizeHostname('.')).toBeNull();
|
||||
expect(normalizeHostname('a b.com')).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractLinkHosts', () => {
|
||||
it('drops the closing parenthesis of a masked markdown link', () => {
|
||||
expect(extractLinkHosts('[open the shop](https://shop.example.com)')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('drops brackets and parentheses around nested links', () => {
|
||||
expect(extractLinkHosts('[[x]](https://shop.example.com)(more)')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('reads angle-bracket autolinks', () => {
|
||||
expect(extractLinkHosts('<https://Shop.Example.com/path>')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('ignores userinfo and ports', () => {
|
||||
expect(extractLinkHosts('https://user:[email protected]:8443/x')).toEqual(['shop.example.com']);
|
||||
expect(extractLinkHosts('[x](https://[email protected])')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('strips trailing dots and punctuation', () => {
|
||||
expect(extractLinkHosts('see https://shop.example.com./ and https://other.example.org, ok')).toEqual([
|
||||
'shop.example.com',
|
||||
'other.example.org',
|
||||
]);
|
||||
});
|
||||
it('decodes percent-encoded hosts the way a browser does', () => {
|
||||
expect(extractLinkHosts('https://shop%2Eexample%2Ecom/')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('returns internationalized hosts in punycode', () => {
|
||||
expect(extractLinkHosts('https://bücher.example/')).toEqual(['xn--bcher-kva.example']);
|
||||
});
|
||||
it('matches the scheme case-insensitively', () => {
|
||||
expect(extractLinkHosts('HTTPS://SHOP.EXAMPLE.COM')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('stops the host at a backslash', () => {
|
||||
expect(extractLinkHosts('https://shop.example.com\\path')).toEqual(['shop.example.com']);
|
||||
});
|
||||
it('deduplicates hosts', () => {
|
||||
expect(extractLinkHosts('https://a.example.com https://A.example.com/x')).toEqual(['a.example.com']);
|
||||
});
|
||||
it('returns an empty array without links', () => {
|
||||
expect(extractLinkHosts('shop.example.com')).toEqual([]);
|
||||
expect(extractLinkHosts(null)).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -330,11 +330,11 @@ async function scanEmbedsForBannedContent(
|
||||
continue;
|
||||
}
|
||||
for (const embed of embeds) {
|
||||
const imageUrls = [embed.thumbnail?.url, embed.image?.url, embed.video?.url, embed.audio?.url].filter(
|
||||
const embedUrls = [embed.url, embed.thumbnail?.url, embed.image?.url, embed.video?.url, embed.audio?.url].filter(
|
||||
(u): u is string => u != null,
|
||||
);
|
||||
for (const imageUrl of imageUrls) {
|
||||
contentModerationService.scanUrl(imageUrl, ctx);
|
||||
for (const embedUrl of embedUrls) {
|
||||
contentModerationService.scanUrl(embedUrl, ctx);
|
||||
}
|
||||
const children = embed.children ?? [];
|
||||
for (const child of children) {
|
||||
|
||||
Reference in New Issue
Block a user