From 7eebfca20b9ca31b8aa7dd2a6775b0f5a5a21f1e Mon Sep 17 00:00:00 2001 From: Hampus Date: Sat, 3 Oct 2026 14:46:08 +0200 Subject: [PATCH] feat(blocklist): add url-domain host patterns (#3164) --- fluxer_admin/openapi-admin.json | 11 +- fluxer_admin/src/api/bans.rs | 17 +- fluxer_admin/src/api/types/common.rs | 19 ++ fluxer_admin/src/routes/bans.rs | 91 +++++++--- fluxer_admin/src/routes/bans_actions.rs | 21 +-- .../src/templates/pages/url_domain_bans.rs | 154 ++++++++++++++-- fluxer_api/package.json | 1 + .../admin/controllers/BanAdminController.ts | 5 +- .../services/AdminBanManagementService.ts | 24 ++- .../tests/AdminBlocklistUrlDomain.test.ts | 170 +++++++++++++++++ .../services/message/MessageActivity.test.ts | 22 ++- .../services/message/MessageActivity.ts | 12 +- .../message/MessageValidationService.ts | 6 +- .../ContentModerationService.ts | 17 +- .../api/middleware/ContentFilterMiddleware.ts | 18 +- .../src/api/middleware/UrlBlocklistCache.ts | 36 ++-- .../tests/ContentFilterMiddleware.test.ts | 10 + .../tests/UrlBlocklistCache.test.ts | 65 +++++++ fluxer_api/src/api/utils/UrlHostRules.ts | 167 +++++++++++++++++ fluxer_api/src/api/utils/UrlNormalizer.ts | 37 ++++ .../src/api/utils/tests/UrlHostRules.test.ts | 171 ++++++++++++++++++ .../src/api/utils/tests/UrlNormalizer.test.ts | 60 +++++- .../src/api/worker/tasks/ExtractEmbeds.ts | 6 +- .../src/content/docs/admin-api/blocklists.mdx | 45 +++-- .../src/content/docs/http-api/webhooks.mdx | 2 +- .../schema/src/domains/admin/AdminSchemas.ts | 14 +- pnpm-lock.yaml | 6 + pnpm-workspace.yaml | 1 + 28 files changed, 1074 insertions(+), 134 deletions(-) create mode 100644 fluxer_api/src/api/admin/tests/AdminBlocklistUrlDomain.test.ts create mode 100644 fluxer_api/src/api/middleware/tests/UrlBlocklistCache.test.ts create mode 100644 fluxer_api/src/api/utils/UrlHostRules.ts create mode 100644 fluxer_api/src/api/utils/tests/UrlHostRules.test.ts diff --git a/fluxer_admin/openapi-admin.json b/fluxer_admin/openapi-admin.json index d341e4472..801a2f664 100644 --- a/fluxer_admin/openapi-admin.json +++ b/fluxer_admin/openapi-admin.json @@ -1451,7 +1451,7 @@ "content": {"application/json": {"schema": {"$ref": "#/components/schemas/Error"}}} } }, - "description": "Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a URL can match a banned domain. The profile-substring blocklist requires a scope.", + "description": "Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a url-domain value can be a hostname or an http(s) URL that a stored domain or pattern covers. The profile-substring blocklist requires a scope.", "security": [{"adminApiKey": []}], "parameters": [ { @@ -12968,10 +12968,13 @@ "BanUrlDomainRequest": { "type": "object", "properties": { - "domain": {"description": "Domain to ban (e.g. example.com)", "type": "string"}, + "domain": { + "description": "Domain to ban (e.g. example.com), or a pattern whose leftmost label contains * under a registrable domain (e.g. *shop*.example.com). Internationalized names are stored in ASCII form.", + "type": "string" + }, "match_subdomains": { "default": true, - "description": "If true, any subdomain rooted at this domain is also banned", + "description": "If true, any subdomain rooted at this domain, or at a host the pattern matches, is also banned", "type": "boolean" }, "category": {"description": "Category / source slug (defaults to \"manual\")", "type": "string"}, @@ -13075,7 +13078,7 @@ "properties": { "match_subdomains": { "default": true, - "description": "If true, any subdomain rooted at this domain is also banned", + "description": "If true, any subdomain rooted at this domain, or at a host the pattern matches, is also banned", "type": "boolean" }, "category": {"description": "Category / source slug (defaults to \"manual\")", "type": "string"}, diff --git a/fluxer_admin/src/api/bans.rs b/fluxer_admin/src/api/bans.rs index 171b4115d..1522a7dc4 100644 --- a/fluxer_admin/src/api/bans.rs +++ b/fluxer_admin/src/api/bans.rs @@ -3,7 +3,7 @@ use crate::api::generated::{snowflake, types as generated_types}; use super::client::{AdminApiClient, ApiError, ApiResult}; -use super::types::{BanAvatarResult, BanCheckResult, BulkBanResult}; +use super::types::{BanAvatarResult, BanCheckResult, BlocklistEntryPage, BulkBanResult}; impl AdminApiClient { pub async fn ban_email(&self, email: &str, audit_log_reason: Option<&str>) -> ApiResult<()> { @@ -148,6 +148,19 @@ impl AdminApiClient { self.check_blocklist_entry("url-domain", domain, None).await } + pub async fn list_url_domain_entries( + &self, + after: Option<&str>, + ) -> ApiResult { + let list_type = blocklist_list_type("url-domain")?; + let response = self + .generated() + .list_admin_blocklist_entries(list_type, after, Some(BLOCKLIST_PAGE_SIZE), None) + .await + .map_err(|e| self.generated_error(e))?; + self.generated_value(response.into_inner()) + } + pub async fn ban_file_sha( &self, sha256_hex: &str, @@ -335,6 +348,8 @@ impl AdminApiClient { const PROFILE_SUBSTRING_LIST: &str = "profile-substring"; +const BLOCKLIST_PAGE_SIZE: &str = "200"; + fn blocklist_list_type(list_type: &str) -> ApiResult { generated_types::AdminBlocklistListType::try_from(list_type) .map_err(|e| ApiError::Parse(e.to_string())) diff --git a/fluxer_admin/src/api/types/common.rs b/fluxer_admin/src/api/types/common.rs index 397297733..8fe3677dc 100644 --- a/fluxer_admin/src/api/types/common.rs +++ b/fluxer_admin/src/api/types/common.rs @@ -260,6 +260,25 @@ pub struct BanCheckResult { pub entries: Vec, } +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct BlocklistEntry { + pub value: String, + #[serde(default)] + pub match_subdomains: Option, + #[serde(default)] + pub category: Option, + #[serde(default)] + pub created_at: Option, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct BlocklistEntryPage { + pub items: Vec, + pub has_more: bool, + #[serde(default)] + pub next_after: Option, +} + #[derive(Clone, Debug, Deserialize, Serialize)] pub struct BulkBanResult { pub job_id: String, diff --git a/fluxer_admin/src/routes/bans.rs b/fluxer_admin/src/routes/bans.rs index 9ba21e701..85e8ebc99 100644 --- a/fluxer_admin/src/routes/bans.rs +++ b/fluxer_admin/src/routes/bans.rs @@ -1,7 +1,10 @@ // SPDX-License-Identifier: AGPL-3.0-or-later use crate::{ - api::client::AdminApiClient, + api::{ + client::{AdminApiClient, ApiError}, + types::FlashMessage, + }, middleware::{auth::AuthContext, csrf, htmx}, state::AppState, templates, @@ -13,10 +16,12 @@ use axum::{ response::{Html, IntoResponse, Response}, routing::get, }; +use serde::Deserialize; use super::ActionQuery; use super::bans_actions::{ BanFormData, custom_flash, execute_ban, extract_value, flash_response, render_inline_flash, + to_flash, }; pub fn router() -> Router { @@ -132,16 +137,56 @@ ban_post!(url_bans_post, "url-bans"); ban_post!(file_sha_bans_post, "file-sha-bans"); ban_post!(avatar_hash_bans_post, "avatar-hash-bans"); +#[derive(Deserialize)] +struct UrlDomainListQuery { + after: Option, +} + +async fn render_url_domain_page( + state: &AppState, + auth: &AuthContext, + flash: Option<&FlashMessage>, + csrf_token: &str, + after: Option<&str>, +) -> Response { + let config = state.config(); + let client = AdminApiClient::new(state.http_client(), config, &auth.session); + let entries = match client.list_url_domain_entries(after).await { + Ok(page) => Some(page), + Err(error) => { + tracing::warn!(%error, "admin API request failed: list URL domain blocklist"); + None + } + }; + let markup = templates::pages::url_domain_bans::url_domain_bans_page( + config, + auth, + flash, + csrf_token, + entries.as_ref(), + ); + Html(markup.into_string()).into_response() +} + +fn ban_url_domain_error(domain: &str, error: &ApiError) -> String { + match error { + ApiError::Http { status: 400, .. } => { + format!("Failed to ban {domain}: not a valid domain, or the pattern is too broad") + } + _ => format!("Failed to ban {domain}"), + } +} + async fn url_domain_bans( State(state): State, auth: axum::Extension, request: Request, ) -> Response { - let config = state.config(); let csrf_token = csrf::get_csrf_token(&request); - let markup = - templates::pages::url_domain_bans::url_domain_bans_page(config, &auth.0, None, &csrf_token); - Html(markup.into_string()).into_response() + let Query(query): Query = + Query::try_from_uri(request.uri()).unwrap_or(Query(UrlDomainListQuery { after: None })); + let after = query.after.as_deref().filter(|value| !value.is_empty()); + render_url_domain_page(&state, &auth.0, None, &csrf_token, after).await } async fn url_domain_bans_post( @@ -172,10 +217,10 @@ async fn url_domain_bans_post( .ban_url_domain(&domain, m_sub, form.audit_log_reason.as_deref()) .await { - Ok(()) => ("success", format!("Domain {domain} banned successfully")), + Ok(()) => ("success", format!("{domain} banned successfully")), Err(error) => { tracing::warn!(%error, domain, "admin API request failed: ban URL domain"); - ("error", format!("Failed to ban domain {domain}")) + ("error", ban_url_domain_error(&domain, &error)) } } } @@ -183,15 +228,15 @@ async fn url_domain_bans_post( .unban_url_domain(&domain, form.audit_log_reason.as_deref()) .await { - Ok(()) => ("success", format!("Domain {domain} unbanned")), + Ok(()) => ("success", format!("{domain} unbanned")), Err(error) => { tracing::warn!(%error, domain, "admin API request failed: unban URL domain"); - ("error", format!("Failed to unban domain {domain}")) + ("error", format!("Failed to unban {domain}")) } }, "check" => match client.check_url_domain_ban(&domain).await { - Ok(r) if r.banned => ("info", format!("Domain {domain} is banned")), - Ok(_) => ("info", format!("Domain {domain} is NOT banned")), + Ok(r) if r.banned => ("info", format!("{domain} is blocked")), + Ok(_) => ("info", format!("{domain} is NOT blocked")), Err(error) => { tracing::warn!(%error, domain, "admin API request failed: check URL domain ban"); ("error", "Error checking ban status".into()) @@ -199,15 +244,11 @@ async fn url_domain_bans_post( }, _ => ("error", "Unknown action".into()), }; - custom_flash( - config, - &auth.0, - is_htmx, - level, - &msg, - &csrf_token, - "url-domain", - ) + if is_htmx { + return render_inline_flash(level, &msg); + } + let flash = to_flash(level, &msg); + render_url_domain_page(&state, &auth.0, Some(&flash), &csrf_token, None).await } async fn profile_substring_bans( @@ -279,13 +320,5 @@ async fn profile_substring_bans_post( }, _ => ("error", "Unknown action".into()), }; - custom_flash( - config, - &auth.0, - is_htmx, - level, - &msg, - &csrf_token, - "profile-substring", - ) + custom_flash(config, &auth.0, is_htmx, level, &msg, &csrf_token) } diff --git a/fluxer_admin/src/routes/bans_actions.rs b/fluxer_admin/src/routes/bans_actions.rs index d827fce62..ac6b23962 100644 --- a/fluxer_admin/src/routes/bans_actions.rs +++ b/fluxer_admin/src/routes/bans_actions.rs @@ -280,25 +280,16 @@ pub fn custom_flash( level: &str, message: &str, csrf_token: &str, - page_type: &str, ) -> Response { if is_htmx { return render_inline_flash(level, message); } let flash = to_flash(level, message); - let markup = match page_type { - "url-domain" => templates::pages::url_domain_bans::url_domain_bans_page( - config, - auth, - Some(&flash), - csrf_token, - ), - _ => templates::pages::profile_substring_bans::profile_substring_bans_page( - config, - auth, - Some(&flash), - csrf_token, - ), - }; + let markup = templates::pages::profile_substring_bans::profile_substring_bans_page( + config, + auth, + Some(&flash), + csrf_token, + ); Html(markup.into_string()).into_response() } diff --git a/fluxer_admin/src/templates/pages/url_domain_bans.rs b/fluxer_admin/src/templates/pages/url_domain_bans.rs index a4e504c92..66110e477 100644 --- a/fluxer_admin/src/templates/pages/url_domain_bans.rs +++ b/fluxer_admin/src/templates/pages/url_domain_bans.rs @@ -1,10 +1,16 @@ // SPDX-License-Identifier: AGPL-3.0-or-later use crate::{ + api::types::{BlocklistEntry, BlocklistEntryPage}, config::AdminConfig, middleware::auth::AuthContext, templates::{ - components::{form::checkbox, page_container::page_header}, + components::{ + badge::{BadgeVariant, badge}, + form::{checkbox, csrf_input}, + page_container::page_header, + table::{data_table, empty_state, table_cell, table_row}, + }, layout::admin_layout, pages::blocklist_helpers::{ BlocklistActionVariant, blocklist_action_card, blocklist_text_field, @@ -13,15 +19,21 @@ use crate::{ }; use maud::{Markup, html}; +const PAGE_DESCRIPTION: &str = "A domain entry blocks that host and, when it matches subdomains, every host under it. \ + A pattern such as *shop*.example.com matches the one label left of a registrable domain, so it blocks \ + shop.example.com and my-shop-2.example.com but never example.com itself. Patterns are matched against the \ + ASCII form of a host."; + pub fn url_domain_bans_page( config: &AdminConfig, auth: &AuthContext, flash: Option<&crate::api::types::FlashMessage>, csrf_token: &str, + entries: Option<&BlocklistEntryPage>, ) -> Markup { let base = &config.base_path; let content = html! { - (page_header("URL Domain Blocklist", None)) + (page_header("URL Domain Blocklist", Some(PAGE_DESCRIPTION))) div class="grid gap-6 lg:grid-cols-2" { (ban_card(base, csrf_token)) (check_card(base, csrf_token)) @@ -29,6 +41,9 @@ pub fn url_domain_bans_page( div class="mt-6" { (unban_card(base, csrf_token)) } + div class="mt-6" { + (entries_card(base, csrf_token, entries)) + } }; admin_layout( config, @@ -43,15 +58,15 @@ pub fn url_domain_bans_page( fn ban_card(base: &str, csrf_token: &str) -> Markup { let action_url = format!("{base}/url-domain-bans?action=ban&_csrf={csrf_token}"); blocklist_action_card( - "Ban URL Domain", + "Ban URL Domain or Pattern", &action_url, csrf_token, html! { - (blocklist_text_field("domain", "Domain", "example.com", true)) + (blocklist_text_field("domain", "Domain or pattern", "example.com or *shop*.example.com", true)) (checkbox("match_subdomains", "true", "Match subdomains (e.g. sub.example.com)", true, true)) (blocklist_text_field("audit_log_reason", "Private reason (audit log, optional)", "Why is this ban being applied?", false)) }, - "Ban Domain", + "Ban", BlocklistActionVariant::Primary, ) } @@ -59,13 +74,13 @@ fn ban_card(base: &str, csrf_token: &str) -> Markup { fn check_card(base: &str, csrf_token: &str) -> Markup { let action_url = format!("{base}/url-domain-bans?action=check&_csrf={csrf_token}"); blocklist_action_card( - "Check Domain Ban Status", + "Test a Host or URL", &action_url, csrf_token, html! { - (blocklist_text_field("domain", "Domain", "example.com", true)) + (blocklist_text_field("domain", "Host or URL", "shop-2.example.com or https://shop.example.com/x", true)) }, - "Check Status", + "Test", BlocklistActionVariant::Primary, ) } @@ -73,14 +88,131 @@ fn check_card(base: &str, csrf_token: &str) -> Markup { fn unban_card(base: &str, csrf_token: &str) -> Markup { let action_url = format!("{base}/url-domain-bans?action=unban&_csrf={csrf_token}"); blocklist_action_card( - "Remove Domain Ban", + "Remove Domain or Pattern", &action_url, csrf_token, html! { - (blocklist_text_field("domain", "Domain", "example.com", true)) + (blocklist_text_field("domain", "Domain or pattern", "example.com or *shop*.example.com", true)) (blocklist_text_field("audit_log_reason", "Private reason (audit log, optional)", "Why is this ban being removed?", false)) }, - "Unban Domain", + "Unban", BlocklistActionVariant::Danger, ) } + +fn entries_card(base: &str, csrf_token: &str, entries: Option<&BlocklistEntryPage>) -> Markup { + html! { + div class="rounded-lg border border-neutral-200 bg-white p-4 shadow-sm sm:p-6" { + div class="mb-4 flex items-center justify-between gap-4" { + h3 class="text-base font-medium text-neutral-900" { "Blocked Domains and Patterns" } + a href={(base) "/url-domain-bans"} class="text-sm text-brand-primary hover:underline" { "Refresh" } + } + @match entries { + None => { + p class="text-sm text-red-700" { "Failed to load the blocklist entries" } + } + Some(page) => { + (entries_table(base, csrf_token, page)) + } + } + } + } +} + +fn entries_table(base: &str, csrf_token: &str, page: &BlocklistEntryPage) -> Markup { + if page.items.is_empty() { + return empty_state("No domains or patterns are blocked"); + } + let next_after = page.next_after.as_deref().filter(|_| page.has_more); + html! { + (data_table( + &["Value", "Kind", "Subdomains", "Category", "Added", ""], + html! { + @for entry in &page.items { + (entry_row(base, csrf_token, entry)) + } + }, + )) + @if let Some(next) = next_after { + div class="mt-4" { + a href={(base) "/url-domain-bans?after=" (urlencoding::encode(next))} + class="text-sm text-brand-primary hover:underline" { + "Next page" + } + } + } + } +} + +fn entry_row(base: &str, csrf_token: &str, entry: &BlocklistEntry) -> Markup { + let action_url = format!("{base}/url-domain-bans?action=unban&_csrf={csrf_token}"); + let is_pattern = entry.value.contains('*'); + table_row(html! { + (table_cell(false, html! { code class="break-all" { (entry.value) } })) + (table_cell(false, html! { + @if is_pattern { + (badge("Pattern", BadgeVariant::Info)) + } @else { + (badge("Domain", BadgeVariant::Default)) + } + })) + (table_cell(true, html! { + @if entry.match_subdomains.unwrap_or(true) { "Yes" } @else { "No" } + })) + (table_cell(true, html! { (entry.category.as_deref().unwrap_or("")) })) + (table_cell(true, html! { (entry.created_at.as_deref().unwrap_or("")) })) + (table_cell(false, html! { + form method="post" action=(action_url) { + (csrf_input(csrf_token)) + input type="hidden" name="domain" value=(entry.value); + button type="submit" class="text-sm font-medium text-red-600 hover:text-red-700" { + "Remove" + } + } + })) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry(value: &str, match_subdomains: bool) -> BlocklistEntry { + BlocklistEntry { + value: value.to_owned(), + match_subdomains: Some(match_subdomains), + category: Some("manual".to_owned()), + created_at: None, + } + } + + #[test] + fn entries_table_lists_patterns_with_remove_forms() { + let page = BlocklistEntryPage { + items: vec![ + entry("*shop*.example.com", false), + entry("store.example.com", true), + ], + has_more: true, + next_after: Some("store.example.com".to_owned()), + }; + let markup = entries_table("/admin", "token", &page).into_string(); + assert!(markup.contains("*shop*.example.com")); + assert!(markup.contains(">Pattern")); + assert!(markup.contains(">Domain")); + assert!(markup.contains(r#"name="domain" value="*shop*.example.com""#)); + assert!(markup.contains("/admin/url-domain-bans?action=unban&_csrf=token")); + assert!(markup.contains("/admin/url-domain-bans?after=store.example.com")); + } + + #[test] + fn entries_table_reports_an_empty_list() { + let page = BlocklistEntryPage { + items: Vec::new(), + has_more: false, + next_after: None, + }; + let markup = entries_table("/admin", "token", &page).into_string(); + assert!(markup.contains("No domains or patterns are blocked")); + } +} diff --git a/fluxer_api/package.json b/fluxer_api/package.json index d6f31247a..491965393 100644 --- a/fluxer_api/package.json +++ b/fluxer_api/package.json @@ -79,6 +79,7 @@ "sharp": "catalog:", "stripe": "catalog:", "tempy": "catalog:", + "tldts": "catalog:", "transliteration": "catalog:", "tsx": "catalog:", "uint8array-extras": "catalog:", diff --git a/fluxer_api/src/api/admin/controllers/BanAdminController.ts b/fluxer_api/src/api/admin/controllers/BanAdminController.ts index 3b50c6451..30caba6d6 100644 --- a/fluxer_api/src/api/admin/controllers/BanAdminController.ts +++ b/fluxer_api/src/api/admin/controllers/BanAdminController.ts @@ -99,7 +99,8 @@ const BLOCKLIST_CATALOG = [ }, { list_type: 'url-domain' as const, - description: 'Domains blocked from being linked, optionally covering every subdomain rooted at the domain.', + description: + 'Domains blocked from being linked, optionally covering every subdomain rooted at the domain. A value whose leftmost label contains * is a pattern that matches that one label under a registrable domain.', value_field: 'domain', fields: ['match_subdomains', 'category', 'severity', 'source_url', 'notes'], scoped: false, @@ -488,7 +489,7 @@ export function BanAdminController(app: HonoApp) { security: ['adminApiKey'], tags: ['Admin'], description: - 'Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a URL can match a banned domain. The profile-substring blocklist requires a scope.', + 'Report whether a value is currently blocked by a blocklist. The value is percent-encoded in the path. An IP address can still match a broader stored CIDR entry, and a url-domain value can be a hostname or an http(s) URL that a stored domain or pattern covers. The profile-substring blocklist requires a scope.', }), async (ctx) => { const adminService = ctx.get('adminService'); diff --git a/fluxer_api/src/api/admin/services/AdminBanManagementService.ts b/fluxer_api/src/api/admin/services/AdminBanManagementService.ts index a1e5a30dd..ee47c9f95 100644 --- a/fluxer_api/src/api/admin/services/AdminBanManagementService.ts +++ b/fluxer_api/src/api/admin/services/AdminBanManagementService.ts @@ -24,6 +24,7 @@ import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache'; import {profileSubstringBlocklistCache} from '@app/api/middleware/ProfileSubstringBlocklistCache'; import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache'; import {canonicalizeStoredPhrase} from '@app/api/utils/PhraseBlocklistNormalization'; +import {parseUrlDomainEntry} from '@app/api/utils/UrlHostRules'; import {canonicalizeUrl} from '@app/api/utils/UrlNormalizer'; import {APIErrorCodes} from '@fluxer/constants/src/ApiErrorCodes'; import {ValidationErrorCodes} from '@fluxer/constants/src/ValidationErrorCodes'; @@ -108,6 +109,16 @@ function normalizeAvatarHashes(hashes: Array): Array { return Array.from(new Set(hashes.map((hash) => stripAvatarAnimationPrefix(hash.toLowerCase())))); } +function hostFromUrlOrHostname(value: string): string | null { + const trimmed = value.trim(); + if (!/^https?:\/\//i.test(trimmed)) return trimmed; + try { + return new URL(trimmed).hostname; + } catch { + return null; + } +} + function withReasonMetadata(entries: Array<[string, string]>, reason: string | undefined): Map { if (!reason) { return new Map(entries); @@ -366,7 +377,9 @@ export class AdminBanManagementService { ) { const {adminRepository} = this.deps; const {cache: cacheService} = this.deps.apiContext.services; - const d = data.domain.toLowerCase(); + const entry = parseUrlDomainEntry(data.domain); + if (!entry.ok) throw InputValidationError.create('domain', entry.message); + const d = entry.value; const matchSubs = data.match_subdomains ?? true; await adminRepository.banUrlDomain({ domain: d, @@ -378,7 +391,7 @@ export class AdminBanManagementService { added_by: adminUserId, notes: data.notes ?? null, }); - urlBlocklistCache.addDomain(d); + urlBlocklistCache.addDomain(d, matchSubs); await cacheService.publish(BANNED_URL_DOMAINS_REFRESH_CHANNEL, 'refresh'); await this.createBlocklistAuditLog({ adminUserId, @@ -388,6 +401,7 @@ export class AdminBanManagementService { metadata: new Map([ ['domain', d], ['match_subdomains', String(matchSubs)], + ['pattern', String(entry.pattern)], ]), }); } @@ -401,7 +415,8 @@ export class AdminBanManagementService { ) { const {adminRepository} = this.deps; const {cache: cacheService} = this.deps.apiContext.services; - const d = data.domain.toLowerCase(); + const entry = parseUrlDomainEntry(data.domain); + const d = entry.ok ? entry.value : data.domain.trim().toLowerCase(); await adminRepository.unbanUrlDomain(d); urlBlocklistCache.removeDomain(d); await cacheService.publish(BANNED_URL_DOMAINS_REFRESH_CHANNEL, 'refresh'); @@ -417,7 +432,8 @@ export class AdminBanManagementService { async checkUrlDomainBan(data: {domain: string}): Promise<{ banned: boolean; }> { - return {banned: urlBlocklistCache.isHostnameBanned(data.domain)}; + const host = hostFromUrlOrHostname(data.domain); + return {banned: host != null && urlBlocklistCache.isHostnameBanned(host)}; } async banFileSha( diff --git a/fluxer_api/src/api/admin/tests/AdminBlocklistUrlDomain.test.ts b/fluxer_api/src/api/admin/tests/AdminBlocklistUrlDomain.test.ts new file mode 100644 index 000000000..791bb6192 --- /dev/null +++ b/fluxer_api/src/api/admin/tests/AdminBlocklistUrlDomain.test.ts @@ -0,0 +1,170 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +import {createTestAccount, setUserACLs, type TestAccount} from '@app/api/auth/tests/AuthTestUtils'; +import {createChannel, createGuild} from '@app/api/channel/tests/ChannelTestUtils'; +import {ensureSessionStarted} from '@app/api/message/tests/MessageTestUtils'; +import {getAdminRepository} from '@app/api/middleware/ServiceSingletons'; +import {type ApiTestHarness, createApiTestHarness} from '@app/api/test/ApiTestHarness'; +import {createBuilder} from '@app/api/test/TestRequestBuilder'; +import {APIErrorCodes} from '@fluxer/constants/src/ApiErrorCodes'; +import {afterAll, beforeAll, beforeEach, describe, expect, it} from 'vitest'; + +interface ValidationErrorResponse { + code: string; + errors?: Array<{path: string; message: string}>; +} + +interface EntryPage { + items: Array<{value: string; match_subdomains: boolean | null}>; +} + +describe('Admin url-domain blocklist patterns', () => { + let harness: ApiTestHarness; + let admin: TestAccount; + + beforeAll(async () => { + harness = await createApiTestHarness(); + }); + + beforeEach(async () => { + await harness.reset(); + admin = await setUserACLs(harness, await createTestAccount(harness), [ + 'admin:authenticate', + 'ban:url_domain:add', + 'ban:url_domain:check', + 'ban:url_domain:remove', + ]); + }); + + afterAll(async () => { + await harness?.shutdown(); + }); + + async function add(domain: string, matchSubdomains?: boolean): Promise { + await createBuilder(harness, admin.token) + .post('/admin/blocklists/url-domain/entries') + .body(matchSubdomains === undefined ? {domain} : {domain, match_subdomains: matchSubdomains}) + .expect(204) + .execute(); + } + + async function check(value: string): Promise { + const json = await createBuilder<{banned: boolean}>(harness, admin.token) + .get(`/admin/blocklists/url-domain/entries/${encodeURIComponent(value)}`) + .expect(200) + .execute(); + return json.banned; + } + + async function list(): Promise { + const json = await createBuilder(harness, admin.token) + .get('/admin/blocklists/url-domain/entries?limit=200') + .expect(200) + .execute(); + return json.items; + } + + it('stores a canonical pattern and reports the hosts it covers', async () => { + await add('**Shop**.OnRender.com.'); + expect(await list()).toMatchObject([{value: '*shop*.onrender.com', match_subdomains: true}]); + expect(await check('shop-2.onrender.com')).toBe(true); + expect(await check('https://www.myshop.onrender.com/checkout')).toBe(true); + expect(await check('onrender.com')).toBe(false); + expect(await check('docs.onrender.com')).toBe(false); + }); + + it('records whether the entry is a pattern in the audit log', async () => { + await add('*shop*.onrender.com', false); + await add('shop.example.com'); + const logs = (await getAdminRepository().listAllAuditLogsPaginated(1000)).filter( + (log) => log.action === 'ban_url_domain', + ); + const metadata = logs.map((log) => Object.fromEntries(log.metadata)); + expect(metadata).toEqual( + expect.arrayContaining([ + {domain: '*shop*.onrender.com', match_subdomains: 'false', pattern: 'true'}, + {domain: 'shop.example.com', match_subdomains: 'true', pattern: 'false'}, + ]), + ); + }); + + it('rejects patterns that are too broad or malformed', async () => { + for (const domain of ['*', '*.com', '*shop*.co.uk', '*.onrender.com', '*ab*.onrender.com', 'shop.*.example.com']) { + const json = await createBuilder(harness, admin.token) + .post('/admin/blocklists/url-domain/entries') + .body({domain}) + .expect(400, 'INVALID_FORM_BODY') + .execute(); + expect(json.errors?.[0]?.path, domain).toBe('domain'); + } + expect(await list()).toEqual([]); + }); + + it('validates the value on update', async () => { + const json = await createBuilder(harness, admin.token) + .patch(`/admin/blocklists/url-domain/entries/${encodeURIComponent('*.com')}`) + .body({}) + .expect(400, 'INVALID_FORM_BODY') + .execute(); + expect(json.errors?.[0]?.path).toBe('domain'); + }); + + it('stores internationalized domains in ASCII form', async () => { + await add('Bücher.Example.'); + expect((await list()).map((entry) => entry.value)).toEqual(['xn--bcher-kva.example']); + expect(await check('www.bücher.example')).toBe(true); + }); + + it('accepts an add for a domain that is already blocked', async () => { + await add('shop.example.com'); + await add('shop.example.com', false); + expect(await list()).toMatchObject([{value: 'shop.example.com', match_subdomains: false}]); + }); + + it('removes a pattern through any spelling that canonicalizes to it', async () => { + await add('*shop*.onrender.com'); + await createBuilder(harness, admin.token) + .delete(`/admin/blocklists/url-domain/entries/${encodeURIComponent('*SHOP**.onrender.com')}`) + .expect(204) + .execute(); + expect(await list()).toEqual([]); + expect(await check('shop.onrender.com')).toBe(false); + }); + + it('blocks messages whose masked links or autolinks point at a covered host', async () => { + await add('*shop*.onrender.com'); + const member = await createTestAccount(harness); + const guild = await createGuild(harness, member.token, 'Links'); + const channel = await createChannel(harness, member.token, guild.id, 'general'); + await ensureSessionStarted(harness, member.token); + for (const content of [ + '[open the store](https://shop-2.onrender.com)', + '', + 'https://SHOP.onrender.com./', + ]) { + await createBuilder(harness, member.token) + .post(`/channels/${channel.id}/messages`) + .body({content}) + .expect(403, APIErrorCodes.CONTENT_BLOCKED) + .execute(); + } + await createBuilder(harness, member.token) + .post(`/channels/${channel.id}/messages`) + .body({content: '[docs](https://docs.onrender.com) and https://onrender.com'}) + .expect(200) + .execute(); + }); + + it('blocks rich embeds that link to a covered host', async () => { + await add('*shop*.onrender.com'); + const member = await createTestAccount(harness); + const guild = await createGuild(harness, member.token, 'Embeds'); + const channel = await createChannel(harness, member.token, guild.id, 'general'); + await ensureSessionStarted(harness, member.token); + await createBuilder(harness, member.token) + .post(`/channels/${channel.id}/messages`) + .body({embeds: [{title: 'Store', url: 'https://shop.onrender.com/'}]}) + .expect(403, APIErrorCodes.CONTENT_BLOCKED) + .execute(); + }); +}); diff --git a/fluxer_api/src/api/channel/services/message/MessageActivity.test.ts b/fluxer_api/src/api/channel/services/message/MessageActivity.test.ts index e42325df0..8b45eed1b 100644 --- a/fluxer_api/src/api/channel/services/message/MessageActivity.test.ts +++ b/fluxer_api/src/api/channel/services/message/MessageActivity.test.ts @@ -33,7 +33,7 @@ function attachment(id: bigint, hash: string | null): MessageAttachment { }; } -function message(attachments: Array): Message { +function message(attachments: Array, content = ''): Message { return new Message({ channel_id: createChannelID(10n), bucket: 0, @@ -43,7 +43,7 @@ function message(attachments: Array): Message { webhook_id: null, webhook_name: null, webhook_avatar_hash: null, - content: '', + content, edited_timestamp: null, pinned_timestamp: null, flags: 0, @@ -75,10 +75,10 @@ describe('message activity', () => { resetActivityEventsForTests(); }); - function params(attachments: Array) { + function params(attachments: Array, content = '') { return { user: {id: createUserID(3n), isBot: false} as unknown as User, - message: message(attachments), + message: message(attachments, content), channel: {id: createChannelID(10n), type: ChannelTypes.DM} as unknown as Channel, guildId: null, guildOwnerId: null, @@ -117,4 +117,18 @@ describe('message activity', () => { expect(updated.data).toMatchObject({message_id: '100', attachments: [{hash: HASH.toLowerCase()}]}); expect(updated.id).not.toBe(created.id); }); + + it('records the link domain of a masked markdown link without its brackets', async () => { + const publisher = new CapturingPublisher(); + await startActivityEvents({publisher, kv: new MockKVProvider()}); + emitMessageCreated( + params( + [], + '[OPEN](https://Shop.Example.com) [docs]() https://user@cdn.example.net:8443/x', + ), + ); + await vi.waitFor(() => expect(publisher.payloads).toHaveLength(1)); + const event = JSON.parse(publisher.payloads[0]!); + expect(event.data.link_domains).toEqual(['shop.example.com', 'docs.example.org', 'cdn.example.net']); + }); }); diff --git a/fluxer_api/src/api/channel/services/message/MessageActivity.ts b/fluxer_api/src/api/channel/services/message/MessageActivity.ts index 0d186ecce..37fb759ff 100644 --- a/fluxer_api/src/api/channel/services/message/MessageActivity.ts +++ b/fluxer_api/src/api/channel/services/message/MessageActivity.ts @@ -10,22 +10,20 @@ import type {Message} from '@app/api/models/Message'; import type {User} from '@app/api/models/User'; import type {IUserRepository} from '@app/api/user/IUserRepository'; import {findInvites} from '@app/api/utils/InviteUtils'; +import {extractLinkHosts} from '@app/api/utils/UrlNormalizer'; import {ChannelTypes} from '@fluxer/constants/src/ChannelConstants'; import {RelationshipTypes} from '@fluxer/constants/src/UserConstants'; const CONTENT_MAX_CHARS = 2000; const LIST_MAX = 10; const MENTIONS_MAX = 20; -const LINK_PATTERN = /https?:\/\/([^\s/?#<>"']+)/giu; +const WWW_PREFIX_RE = /^www\./u; function linkDomains(content: string): Array { const domains = new Set(); - for (const match of content.matchAll(LINK_PATTERN)) { - const host = match[1] - ?.toLowerCase() - .replace(/:\d+$/u, '') - .replace(/^www\./u, ''); - if (host) domains.add(host); + for (const host of extractLinkHosts(content)) { + const domain = host.replace(WWW_PREFIX_RE, ''); + if (domain) domains.add(domain); if (domains.size >= LIST_MAX) break; } return [...domains]; diff --git a/fluxer_api/src/api/channel/services/message/MessageValidationService.ts b/fluxer_api/src/api/channel/services/message/MessageValidationService.ts index c4b4dd98f..5ae79fd24 100644 --- a/fluxer_api/src/api/channel/services/message/MessageValidationService.ts +++ b/fluxer_api/src/api/channel/services/message/MessageValidationService.ts @@ -94,10 +94,14 @@ export class MessageValidationService { contentModerationService.scanText(data.content, modCtx); if (data.embeds) { for (const embed of data.embeds) { + if (embed.url) contentModerationService.scanUrl(embed.url, modCtx); contentModerationService.scanText(embed.title ?? null, modCtx); contentModerationService.scanText(embed.description ?? null, modCtx); if (embed.footer) contentModerationService.scanText(embed.footer.text ?? null, modCtx); - if (embed.author) contentModerationService.scanText(embed.author.name ?? null, modCtx); + if (embed.author) { + contentModerationService.scanText(embed.author.name ?? null, modCtx); + if (embed.author.url) contentModerationService.scanUrl(embed.author.url, modCtx); + } if (embed.fields) { for (const field of embed.fields) { contentModerationService.scanText(field.name ?? null, modCtx); diff --git a/fluxer_api/src/api/infrastructure/ContentModerationService.ts b/fluxer_api/src/api/infrastructure/ContentModerationService.ts index c04d531e7..0dd3fb358 100644 --- a/fluxer_api/src/api/infrastructure/ContentModerationService.ts +++ b/fluxer_api/src/api/infrastructure/ContentModerationService.ts @@ -5,7 +5,6 @@ import {Logger} from '@app/api/Logger'; import {fileShaCache} from '@app/api/middleware/FileShaCache'; import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache'; import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache'; -import {extractUrlCandidates} from '@app/api/utils/UrlNormalizer'; import {ContentBlockedError} from '@fluxer/errors/src/domains/content/ContentBlockedError'; export interface ModerationContext { @@ -40,16 +39,12 @@ class ContentModerationService { ); throw new ContentBlockedError(); } - const urls = extractUrlCandidates(text); - if (urls.length === 0) return; - for (const url of urls) { - if (urlBlocklistCache.isUrlOrDomainBanned(url)) { - Logger.warn( - {surface: ctx.surface, userId: ctx.userId?.toString(), guildId: ctx.guildId?.toString()}, - 'content_moderation.block url match in text', - ); - throw new ContentBlockedError(); - } + if (urlBlocklistCache.containsBannedLink(text)) { + Logger.warn( + {surface: ctx.surface, userId: ctx.userId?.toString(), guildId: ctx.guildId?.toString()}, + 'content_moderation.block url match in text', + ); + throw new ContentBlockedError(); } } diff --git a/fluxer_api/src/api/middleware/ContentFilterMiddleware.ts b/fluxer_api/src/api/middleware/ContentFilterMiddleware.ts index 7822d8402..4a0a2da9c 100644 --- a/fluxer_api/src/api/middleware/ContentFilterMiddleware.ts +++ b/fluxer_api/src/api/middleware/ContentFilterMiddleware.ts @@ -4,7 +4,6 @@ import {Logger} from '@app/api/Logger'; import {phraseBlocklistCache} from '@app/api/middleware/PhraseBlocklistCache'; import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache'; import {readRequestJsonBody} from '@app/api/utils/RequestJsonBody'; -import {extractUrlCandidates} from '@app/api/utils/UrlNormalizer'; import {ContentBlockedError} from '@fluxer/errors/src/domains/content/ContentBlockedError'; import {createMiddleware} from 'hono/factory'; @@ -73,6 +72,8 @@ const SKIP_FIELD_SUFFIXES = [ ] as const; const SKIP_CONTENT_FILTER_PATH_PARTS = [ '/admin/blocklists/phrase/', + '/admin/blocklists/url-domain/', + '/admin/blocklists/url/', '/auth/', '/oauth2/', '/premium/store/', @@ -149,15 +150,12 @@ const ContentFilterMiddleware = createMiddleware(async (ctx, next) => { ); throw new ContentBlockedError(); } - const urls = extractUrlCandidates(text); - for (const url of urls) { - if (urlBlocklistCache.isUrlOrDomainBanned(url)) { - Logger.warn( - {surface: 'global_filter', userId: userId?.toString(), path}, - 'content_moderation.block url match in request body', - ); - throw new ContentBlockedError(); - } + if (urlBlocklistCache.containsBannedLink(text)) { + Logger.warn( + {surface: 'global_filter', userId: userId?.toString(), path}, + 'content_moderation.block url match in request body', + ); + throw new ContentBlockedError(); } } return next(); diff --git a/fluxer_api/src/api/middleware/UrlBlocklistCache.ts b/fluxer_api/src/api/middleware/UrlBlocklistCache.ts index 236afe25d..6c83aee0e 100644 --- a/fluxer_api/src/api/middleware/UrlBlocklistCache.ts +++ b/fluxer_api/src/api/middleware/UrlBlocklistCache.ts @@ -7,12 +7,13 @@ import {BANNED_URL_DOMAINS_REFRESH_CHANNEL, BANNED_URLS_REFRESH_CHANNEL} from '@ import type {IStorageService} from '@app/api/infrastructure/IStorageService'; import {Logger} from '@app/api/Logger'; import {RefreshSubscription} from '@app/api/utils/RefreshSubscription'; -import {canonicalizeUrl} from '@app/api/utils/UrlNormalizer'; +import {UrlHostRuleSet} from '@app/api/utils/UrlHostRules'; +import {canonicalizeUrl, extractLinkHosts, extractUrlCandidates} from '@app/api/utils/UrlNormalizer'; import type {IKVProvider} from '@pkgs/kv_client/src/IKVProvider'; class UrlBlocklistCache { private exactUrls: Set = new Set(); - private blockedDomains: Set = new Set(); + private hostRules = new UrlHostRuleSet(); private adminRepository = new AdminRepository(); private kvClient: IKVProvider | null = null; private storageService: IStorageService | null = null; @@ -55,15 +56,15 @@ class UrlBlocklistCache { for (const row of manualUrls) { if (row.url_canonical) nextUrls.add(row.url_canonical.toLowerCase()); } - const nextDomains = new Set(); + const nextHostRules = new UrlHostRuleSet(); for (const row of domains) { - nextDomains.add(row.domain.toLowerCase()); + nextHostRules.add(row.domain, row.match_subdomains ?? true); } this.exactUrls = nextUrls; - this.blockedDomains = nextDomains; + this.hostRules = nextHostRules; this.consecutiveFailures = 0; Logger.debug( - {urls: nextUrls.size, domains: nextDomains.size, feedUrls: feedUrls.size}, + {urls: nextUrls.size, ...nextHostRules.size, feedUrls: feedUrls.size}, 'URL blocklist cache refreshed', ); } @@ -94,7 +95,17 @@ class UrlBlocklistCache { } isHostnameBanned(host: string): boolean { - return this.blockedDomains.has(host.toLowerCase()); + return this.hostRules.matches(host); + } + + containsBannedLink(text: string): boolean { + for (const url of extractUrlCandidates(text)) { + if (this.isUrlOrDomainBanned(url)) return true; + } + for (const host of extractLinkHosts(text)) { + if (this.isHostnameBanned(host)) return true; + } + return false; } addExactUrl(canonical: string): void { @@ -105,21 +116,22 @@ class UrlBlocklistCache { this.exactUrls.delete(canonical.toLowerCase()); } - addDomain(domain: string): void { - this.blockedDomains.add(domain.toLowerCase()); + addDomain(domain: string, matchSubdomains = true): void { + this.hostRules.add(domain, matchSubdomains); } removeDomain(domain: string): void { - this.blockedDomains.delete(domain.toLowerCase()); + this.hostRules.remove(domain); } get size(): { urls: number; domains: number; + patterns: number; } { return { urls: this.exactUrls.size, - domains: this.blockedDomains.size, + ...this.hostRules.size, }; } @@ -128,7 +140,7 @@ class UrlBlocklistCache { Logger.error({error}, 'Failed to shut down URL blocklist cache'); }); this.exactUrls = new Set(); - this.blockedDomains = new Set(); + this.hostRules = new UrlHostRuleSet(); this.kvClient = null; this.storageService = null; this.consecutiveFailures = 0; diff --git a/fluxer_api/src/api/middleware/tests/ContentFilterMiddleware.test.ts b/fluxer_api/src/api/middleware/tests/ContentFilterMiddleware.test.ts index 61e67f6a7..0b7bac2b1 100644 --- a/fluxer_api/src/api/middleware/tests/ContentFilterMiddleware.test.ts +++ b/fluxer_api/src/api/middleware/tests/ContentFilterMiddleware.test.ts @@ -81,4 +81,14 @@ describe('shouldSkipContentFilterPath', () => { const result = paths.map((path) => shouldSkipContentFilterPath(path)); expect(result).toEqual([false, false, false]); }); + test('skips blocklist writes whose values are the blocked content', () => { + const paths = [ + '/admin/blocklists/phrase/entries', + '/admin/blocklists/url/entries', + '/admin/blocklists/url-domain/entries', + '/admin/blocklists/profile-substring/entries', + ]; + const result = paths.map((path) => shouldSkipContentFilterPath(path)); + expect(result).toEqual([true, true, true, false]); + }); }); diff --git a/fluxer_api/src/api/middleware/tests/UrlBlocklistCache.test.ts b/fluxer_api/src/api/middleware/tests/UrlBlocklistCache.test.ts new file mode 100644 index 000000000..529282279 --- /dev/null +++ b/fluxer_api/src/api/middleware/tests/UrlBlocklistCache.test.ts @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +import {urlBlocklistCache} from '@app/api/middleware/UrlBlocklistCache'; +import {afterEach, describe, expect, it} from 'vitest'; + +describe('urlBlocklistCache link matching', () => { + afterEach(() => { + urlBlocklistCache.resetForTesting(); + }); + + it('blocks a masked markdown link whose target a pattern covers', () => { + urlBlocklistCache.addDomain('*shop*.onrender.com', true); + expect(urlBlocklistCache.containsBannedLink('[open the store](https://shop-2.onrender.com)')).toBe(true); + expect(urlBlocklistCache.containsBannedLink('[open the store]()')).toBe(true); + expect(urlBlocklistCache.containsBannedLink('[https://docs.onrender.com](https://shop.onrender.com)')).toBe(true); + }); + + it('blocks autolinks and bare links', () => { + urlBlocklistCache.addDomain('*shop*.onrender.com', false); + expect(urlBlocklistCache.containsBannedLink('')).toBe(true); + expect(urlBlocklistCache.containsBannedLink('visit myshop.onrender.com today')).toBe(true); + }); + + it('normalizes the link target before matching', () => { + urlBlocklistCache.addDomain('*shop*.onrender.com', false); + const variants = [ + 'https://user:pass@Shop.OnRender.com:8443/x', + 'https://login@shop.onrender.com', + 'https://shop.onrender.com./', + 'https://shop%2Eonrender%2Ecom/', + 'https://shop。onrender。com/', + 'https://shop-ü.onrender.com/', + ]; + for (const text of variants) { + expect(urlBlocklistCache.containsBannedLink(text), text).toBe(true); + } + }); + + it('leaves the bare suffix and unrelated hosts alone', () => { + urlBlocklistCache.addDomain('*shop*.onrender.com', true); + expect(urlBlocklistCache.containsBannedLink('https://onrender.com/docs')).toBe(false); + expect(urlBlocklistCache.containsBannedLink('[docs](https://docs.onrender.com)')).toBe(false); + expect(urlBlocklistCache.containsBannedLink('the shop is closed')).toBe(false); + }); + + it('covers subdomains of a domain entry only when it is flagged to', () => { + urlBlocklistCache.addDomain('shop.example.com', true); + urlBlocklistCache.addDomain('store.example.com', false); + expect(urlBlocklistCache.containsBannedLink('https://www.shop.example.com')).toBe(true); + expect(urlBlocklistCache.containsBannedLink('https://store.example.com')).toBe(true); + expect(urlBlocklistCache.containsBannedLink('https://www.store.example.com')).toBe(false); + }); + + it('stops matching after removal', () => { + urlBlocklistCache.addDomain('*shop*.onrender.com', true); + urlBlocklistCache.removeDomain('*shop*.onrender.com'); + expect(urlBlocklistCache.containsBannedLink('https://shop.onrender.com')).toBe(false); + }); + + it('applies domain rules to a single URL', () => { + urlBlocklistCache.addDomain('*shop*.onrender.com', false); + expect(urlBlocklistCache.isUrlOrDomainBanned('https://shop.onrender.com/checkout')).toBe(true); + expect(urlBlocklistCache.isUrlOrDomainBanned('https://docs.onrender.com/')).toBe(false); + }); +}); diff --git a/fluxer_api/src/api/utils/UrlHostRules.ts b/fluxer_api/src/api/utils/UrlHostRules.ts new file mode 100644 index 000000000..2cfde4d38 --- /dev/null +++ b/fluxer_api/src/api/utils/UrlHostRules.ts @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +import {normalizeHostname} from '@app/api/utils/UrlNormalizer'; +import {getDomain} from 'tldts'; + +const URL_HOST_PATTERN_MAX_WILDCARDS = 3; +const URL_HOST_PATTERN_MIN_LITERAL_CHARS = 3; + +const MAX_HOSTNAME_LENGTH = 253; +const HOSTNAME_LABEL_RE = /^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$/; +const GLOB_LABEL_RE = /^[a-z0-9*-]{1,63}$/; +const REPEATED_WILDCARDS_RE = /\*+/g; + +export type UrlDomainEntry = + | {ok: true; value: string; pattern: boolean} + | { + ok: false; + message: string; + }; + +function isValidHostname(host: string): boolean { + if (host.length > MAX_HOSTNAME_LENGTH) return false; + return host.split('.').every((label) => HOSTNAME_LABEL_RE.test(label)); +} + +function invalid(message: string): UrlDomainEntry { + return {ok: false, message}; +} + +export function parseUrlDomainEntry(raw: string): UrlDomainEntry { + const value = raw.trim().toLowerCase(); + if (!value.includes('*')) { + const host = normalizeHostname(value); + if (!host?.includes('.') || !isValidHostname(host)) { + return invalid('Must be a valid domain'); + } + return {ok: true, value: host, pattern: false}; + } + const separator = value.indexOf('.'); + if (separator === -1) { + return invalid('A pattern must name a domain after its wildcard label'); + } + const glob = value.slice(0, separator).replace(REPEATED_WILDCARDS_RE, '*'); + const suffixValue = value.slice(separator + 1); + if (suffixValue.includes('*')) { + return invalid('Only the leftmost label of a pattern can contain *'); + } + if (!GLOB_LABEL_RE.test(glob)) { + return invalid('The wildcard label can contain only a-z, 0-9, hyphens, and *'); + } + const wildcards = glob.length - glob.replaceAll('*', '').length; + if (wildcards > URL_HOST_PATTERN_MAX_WILDCARDS) { + return invalid(`The wildcard label can contain at most ${URL_HOST_PATTERN_MAX_WILDCARDS} *`); + } + if (glob.length - wildcards < URL_HOST_PATTERN_MIN_LITERAL_CHARS) { + return invalid( + `The wildcard label needs at least ${URL_HOST_PATTERN_MIN_LITERAL_CHARS} characters besides *, so the pattern is too broad`, + ); + } + const suffix = normalizeHostname(suffixValue); + if (!suffix || !isValidHostname(suffix)) { + return invalid('The part after the wildcard label must be a valid domain'); + } + if (getDomain(suffix, {allowPrivateDomains: false}) === null) { + return invalid('The part after the wildcard label is a public suffix, so the pattern is too broad'); + } + const entry = `${glob}.${suffix}`; + if (entry.length > MAX_HOSTNAME_LENGTH) { + return invalid('Must be a valid domain'); + } + return {ok: true, value: entry, pattern: true}; +} + +interface HostPatternRule { + value: string; + segments: ReadonlyArray; + matchSubdomains: boolean; +} + +function globMatches(segments: ReadonlyArray, label: string): boolean { + const first = segments[0] ?? ''; + const last = segments[segments.length - 1] ?? ''; + if (!label.startsWith(first)) return false; + let position = first.length; + for (let index = 1; index < segments.length - 1; index++) { + const segment = segments[index] ?? ''; + const found = label.indexOf(segment, position); + if (found === -1) return false; + position = found + segment.length; + } + return label.length - last.length >= position && label.endsWith(last); +} + +export class UrlHostRuleSet { + private readonly domains = new Map(); + private readonly patterns = new Map>(); + private patternCount = 0; + + add(rawValue: string, matchSubdomains: boolean): void { + if (rawValue.includes('*')) { + const entry = parseUrlDomainEntry(rawValue); + if (!entry.ok || !entry.pattern) return; + this.remove(entry.value); + const separator = entry.value.indexOf('.'); + const suffix = entry.value.slice(separator + 1); + const rules = this.patterns.get(suffix) ?? []; + rules.push({ + value: entry.value, + segments: entry.value.slice(0, separator).split('*'), + matchSubdomains, + }); + this.patterns.set(suffix, rules); + this.patternCount++; + return; + } + const host = normalizeHostname(rawValue); + if (host) this.domains.set(host, matchSubdomains); + } + + remove(rawValue: string): void { + if (!rawValue.includes('*')) { + const host = normalizeHostname(rawValue); + if (host) this.domains.delete(host); + return; + } + const entry = parseUrlDomainEntry(rawValue); + if (!entry.ok) return; + const suffix = entry.value.slice(entry.value.indexOf('.') + 1); + const rules = this.patterns.get(suffix); + if (!rules) return; + const remaining = rules.filter((rule) => rule.value !== entry.value); + this.patternCount -= rules.length - remaining.length; + if (remaining.length === 0) { + this.patterns.delete(suffix); + } else { + this.patterns.set(suffix, remaining); + } + } + + matches(rawHost: string): boolean { + const host = normalizeHostname(rawHost); + if (!host) return false; + if (this.domains.has(host)) return true; + let labelStart = 0; + let isFirstLabel = true; + while (labelStart < host.length) { + const separator = host.indexOf('.', labelStart); + if (separator === -1) return false; + const suffix = host.slice(separator + 1); + if (this.domains.get(suffix) === true) return true; + const rules = this.patterns.get(suffix); + if (rules) { + const label = host.slice(labelStart, separator); + for (const rule of rules) { + if ((isFirstLabel || rule.matchSubdomains) && globMatches(rule.segments, label)) return true; + } + } + labelStart = separator + 1; + isFirstLabel = false; + } + return false; + } + + get size(): {domains: number; patterns: number} { + return {domains: this.domains.size, patterns: this.patternCount}; + } +} diff --git a/fluxer_api/src/api/utils/UrlNormalizer.ts b/fluxer_api/src/api/utils/UrlNormalizer.ts index 81843a40e..b9724b568 100644 --- a/fluxer_api/src/api/utils/UrlNormalizer.ts +++ b/fluxer_api/src/api/utils/UrlNormalizer.ts @@ -58,6 +58,17 @@ export function canonicalizeUrl(raw: string): string | null { return parsed.toString().toLowerCase(); } +const TRAILING_DOTS_RE = /\.+$/; + +export function normalizeHostname(raw: string): string | null { + const trimmed = raw.trim(); + if (!trimmed) return null; + const ascii = domainToASCII(trimmed); + if (!ascii) return null; + const host = ascii.toLowerCase().replace(TRAILING_DOTS_RE, ''); + return host || null; +} + const URL_CANDIDATE_RE = /(?\u201D']*)?)/gi; const TRAILING_PUNCT_RE = /[.,;:!?)\]}\x22'\u00bb\u201C\u201D]+$/; @@ -78,3 +89,29 @@ export function extractUrlCandidates(text: string | null | undefined): Array"'`()[\]{}|^]+)/giu; + +function hostFromAuthority(authority: string): string | null { + const cleaned = authority.replace(TRAILING_PUNCT_RE, ''); + if (!cleaned) return null; + let parsed: URL; + try { + parsed = new URL(`http://${cleaned}/`); + } catch { + return null; + } + return normalizeHostname(parsed.hostname); +} + +export function extractLinkHosts(text: string | null | undefined): Array { + if (!text) return []; + const hosts = new Set(); + for (const match of text.matchAll(LINK_AUTHORITY_RE)) { + const authority = match[1]; + if (!authority) continue; + const host = hostFromAuthority(authority); + if (host) hosts.add(host); + } + return [...hosts]; +} diff --git a/fluxer_api/src/api/utils/tests/UrlHostRules.test.ts b/fluxer_api/src/api/utils/tests/UrlHostRules.test.ts new file mode 100644 index 000000000..ec7e4855d --- /dev/null +++ b/fluxer_api/src/api/utils/tests/UrlHostRules.test.ts @@ -0,0 +1,171 @@ +// SPDX-License-Identifier: AGPL-3.0-or-later + +import {parseUrlDomainEntry, UrlHostRuleSet} from '@app/api/utils/UrlHostRules'; +import {describe, expect, it} from 'vitest'; + +function rules(entries: Array<[string, boolean]>): UrlHostRuleSet { + const set = new UrlHostRuleSet(); + for (const [value, matchSubdomains] of entries) { + set.add(value, matchSubdomains); + } + return set; +} + +describe('parseUrlDomainEntry', () => { + it('canonicalizes a plain domain', () => { + expect(parseUrlDomainEntry(' Shop.Example.COM. ')).toEqual({ok: true, value: 'shop.example.com', pattern: false}); + }); + + it('stores an internationalized domain in punycode', () => { + expect(parseUrlDomainEntry('bücher.example')).toEqual({ok: true, value: 'xn--bcher-kva.example', pattern: false}); + }); + + it('keeps an exact entry for a domain under a shared hosting suffix', () => { + expect(parseUrlDomainEntry('onrender.com')).toEqual({ok: true, value: 'onrender.com', pattern: false}); + }); + + it('rejects malformed plain domains', () => { + for (const value of ['localhost', 'bad_label.example.com', '-lead.example.com', 'a..example.com', 'a b.com']) { + expect(parseUrlDomainEntry(value).ok).toBe(false); + } + }); + + it('canonicalizes a pattern and collapses repeated wildcards', () => { + expect(parseUrlDomainEntry('**Shop**.OnRender.com.')).toEqual({ + ok: true, + value: '*shop*.onrender.com', + pattern: true, + }); + }); + + it('accepts patterns under a private shared hosting suffix', () => { + expect(parseUrlDomainEntry('*shop*.github.io').ok).toBe(true); + expect(parseUrlDomainEntry('shop-*.example.co.uk').ok).toBe(true); + }); + + it('rejects patterns that are too broad', () => { + for (const value of [ + '*', + '*.com', + '*shop*.com', + 'shop*.co.uk', + '*.example.com', + '*ab*.example.com', + 'a*b.example.com', + ]) { + expect(parseUrlDomainEntry(value).ok).toBe(false); + } + }); + + it('rejects wildcards outside the leftmost label', () => { + expect(parseUrlDomainEntry('shop.*.example.com').ok).toBe(false); + expect(parseUrlDomainEntry('*shop*.example*.com').ok).toBe(false); + }); + + it('rejects wildcard labels with unsupported characters or too many wildcards', () => { + expect(parseUrlDomainEntry('*sh?p*.example.com').ok).toBe(false); + expect(parseUrlDomainEntry('*sh_p*.example.com').ok).toBe(false); + expect(parseUrlDomainEntry('*a*b*c*d*.example.com').ok).toBe(false); + }); + + it('rejects a pattern without a domain', () => { + expect(parseUrlDomainEntry('*shop*').ok).toBe(false); + }); +}); + +describe('UrlHostRuleSet', () => { + it('matches an exact domain', () => { + const set = rules([['shop.example.com', false]]); + expect(set.matches('shop.example.com')).toBe(true); + expect(set.matches('www.shop.example.com')).toBe(false); + expect(set.matches('example.com')).toBe(false); + }); + + it('matches subdomains only when the entry covers them', () => { + const set = rules([['shop.example.com', true]]); + expect(set.matches('shop.example.com')).toBe(true); + expect(set.matches('a.b.shop.example.com')).toBe(true); + expect(set.matches('myshop.example.com')).toBe(false); + }); + + it('normalizes the checked host', () => { + const set = rules([['shop.example.com', true]]); + expect(set.matches('SHOP.Example.com.')).toBe(true); + expect(set.matches('www.shop。example。com')).toBe(true); + expect(rules([['xn--bcher-kva.example', false]]).matches('Bücher.example')).toBe(true); + }); + + it('matches a pattern against the label left of its suffix', () => { + const set = rules([['*shop*.onrender.com', false]]); + expect(set.matches('shop.onrender.com')).toBe(true); + expect(set.matches('best-shop-2.onrender.com')).toBe(true); + expect(set.matches('SHOPPING.onrender.com')).toBe(true); + expect(set.matches('store.onrender.com')).toBe(false); + }); + + it('never matches the bare suffix of a pattern', () => { + const set = rules([['*shop*.onrender.com', true]]); + expect(set.matches('onrender.com')).toBe(false); + expect(set.matches('com')).toBe(false); + expect(set.matches('shop.com')).toBe(false); + expect(set.matches('shop.onrender.com.evil.example')).toBe(false); + expect(set.matches('shoponrender.com')).toBe(false); + }); + + it('applies anchored pattern segments', () => { + const set = rules([ + ['shop-*.example.com', false], + ['*-store.example.org', false], + ['a*b*c.example.net', false], + ]); + expect(set.matches('shop-1.example.com')).toBe(true); + expect(set.matches('myshop-1.example.com')).toBe(false); + expect(set.matches('big-store.example.org')).toBe(true); + expect(set.matches('big-store2.example.org')).toBe(false); + expect(set.matches('axxbyyc.example.net')).toBe(true); + expect(set.matches('abc.example.net')).toBe(true); + expect(set.matches('acb.example.net')).toBe(false); + }); + + it('extends a pattern to deeper subdomains only when the entry covers them', () => { + const exact = rules([['*shop*.onrender.com', false]]); + const covering = rules([['*shop*.onrender.com', true]]); + expect(exact.matches('www.shop.onrender.com')).toBe(false); + expect(covering.matches('www.shop.onrender.com')).toBe(true); + expect(covering.matches('shop.www.onrender.com')).toBe(false); + }); + + it('matches internationalized labels through their punycode form', () => { + const set = rules([['*shop*.onrender.com', false]]); + expect(set.matches('shop-ü.onrender.com')).toBe(true); + }); + + it('removes domains and patterns', () => { + const set = rules([ + ['shop.example.com', true], + ['*shop*.onrender.com', true], + ['*store*.onrender.com', true], + ]); + set.remove('SHOP.example.com'); + set.remove('**shop*.onrender.com'); + expect(set.matches('shop.example.com')).toBe(false); + expect(set.matches('shop.onrender.com')).toBe(false); + expect(set.matches('store.onrender.com')).toBe(true); + expect(set.size).toEqual({domains: 0, patterns: 1}); + }); + + it('replaces a pattern when it is added again', () => { + const set = rules([ + ['*shop*.onrender.com', false], + ['*shop*.onrender.com', true], + ]); + expect(set.size).toEqual({domains: 0, patterns: 1}); + expect(set.matches('www.shop.onrender.com')).toBe(true); + }); + + it('ignores invalid stored patterns', () => { + const set = rules([['*.com', true]]); + expect(set.size).toEqual({domains: 0, patterns: 0}); + expect(set.matches('anything.com')).toBe(false); + }); +}); diff --git a/fluxer_api/src/api/utils/tests/UrlNormalizer.test.ts b/fluxer_api/src/api/utils/tests/UrlNormalizer.test.ts index 446154f37..971f64b38 100644 --- a/fluxer_api/src/api/utils/tests/UrlNormalizer.test.ts +++ b/fluxer_api/src/api/utils/tests/UrlNormalizer.test.ts @@ -1,6 +1,6 @@ // SPDX-License-Identifier: AGPL-3.0-or-later -import {canonicalizeUrl, extractUrlCandidates} from '@app/api/utils/UrlNormalizer'; +import {canonicalizeUrl, extractLinkHosts, extractUrlCandidates, normalizeHostname} from '@app/api/utils/UrlNormalizer'; import {describe, expect, it} from 'vitest'; describe('canonicalizeUrl', () => { @@ -176,3 +176,61 @@ describe('extractUrlCandidates', () => { expect(extractUrlCandidates('hello world no links here')).toEqual([]); }); }); + +describe('normalizeHostname', () => { + it('lowercases and strips trailing dots', () => { + expect(normalizeHostname('Shop.Example.COM..')).toBe('shop.example.com'); + }); + it('converts internationalized names to punycode', () => { + expect(normalizeHostname('bücher.example')).toBe('xn--bcher-kva.example'); + }); + it('maps ideographic full stops to dots', () => { + expect(normalizeHostname('shop\u3002example\u3002com')).toBe('shop.example.com'); + }); + it('rejects empty and invalid input', () => { + expect(normalizeHostname(' ')).toBeNull(); + expect(normalizeHostname('.')).toBeNull(); + expect(normalizeHostname('a b.com')).toBeNull(); + }); +}); + +describe('extractLinkHosts', () => { + it('drops the closing parenthesis of a masked markdown link', () => { + expect(extractLinkHosts('[open the shop](https://shop.example.com)')).toEqual(['shop.example.com']); + }); + it('drops brackets and parentheses around nested links', () => { + expect(extractLinkHosts('[[x]](https://shop.example.com)(more)')).toEqual(['shop.example.com']); + }); + it('reads angle-bracket autolinks', () => { + expect(extractLinkHosts('')).toEqual(['shop.example.com']); + }); + it('ignores userinfo and ports', () => { + expect(extractLinkHosts('https://user:pass@shop.example.com:8443/x')).toEqual(['shop.example.com']); + expect(extractLinkHosts('[x](https://login@shop.example.com)')).toEqual(['shop.example.com']); + }); + it('strips trailing dots and punctuation', () => { + expect(extractLinkHosts('see https://shop.example.com./ and https://other.example.org, ok')).toEqual([ + 'shop.example.com', + 'other.example.org', + ]); + }); + it('decodes percent-encoded hosts the way a browser does', () => { + expect(extractLinkHosts('https://shop%2Eexample%2Ecom/')).toEqual(['shop.example.com']); + }); + it('returns internationalized hosts in punycode', () => { + expect(extractLinkHosts('https://bücher.example/')).toEqual(['xn--bcher-kva.example']); + }); + it('matches the scheme case-insensitively', () => { + expect(extractLinkHosts('HTTPS://SHOP.EXAMPLE.COM')).toEqual(['shop.example.com']); + }); + it('stops the host at a backslash', () => { + expect(extractLinkHosts('https://shop.example.com\\path')).toEqual(['shop.example.com']); + }); + it('deduplicates hosts', () => { + expect(extractLinkHosts('https://a.example.com https://A.example.com/x')).toEqual(['a.example.com']); + }); + it('returns an empty array without links', () => { + expect(extractLinkHosts('shop.example.com')).toEqual([]); + expect(extractLinkHosts(null)).toEqual([]); + }); +}); diff --git a/fluxer_api/src/api/worker/tasks/ExtractEmbeds.ts b/fluxer_api/src/api/worker/tasks/ExtractEmbeds.ts index 5dd3ea835..b52d2c1a9 100644 --- a/fluxer_api/src/api/worker/tasks/ExtractEmbeds.ts +++ b/fluxer_api/src/api/worker/tasks/ExtractEmbeds.ts @@ -330,11 +330,11 @@ async function scanEmbedsForBannedContent( continue; } for (const embed of embeds) { - const imageUrls = [embed.thumbnail?.url, embed.image?.url, embed.video?.url, embed.audio?.url].filter( + const embedUrls = [embed.url, embed.thumbnail?.url, embed.image?.url, embed.video?.url, embed.audio?.url].filter( (u): u is string => u != null, ); - for (const imageUrl of imageUrls) { - contentModerationService.scanUrl(imageUrl, ctx); + for (const embedUrl of embedUrls) { + contentModerationService.scanUrl(embedUrl, ctx); } const children = embed.children ?? []; for (const child of children) { diff --git a/fluxer_docs/src/content/docs/admin-api/blocklists.mdx b/fluxer_docs/src/content/docs/admin-api/blocklists.mdx index b4e2f9e00..025691c2c 100644 --- a/fluxer_docs/src/content/docs/admin-api/blocklists.mdx +++ b/fluxer_docs/src/content/docs/admin-api/blocklists.mdx @@ -35,7 +35,7 @@ Fluxer builds each permission name from `ban:`, the list name with each hyphen w 4 Canonicalised before storage. A value Fluxer cannot canonicalise returns 400 `INVALID_FORM_BODY` naming `url` -5 Stored lowercased and matched against the lowercased hostname of a submitted URL. `match_subdomains` is stored on the row and defaults to true, and the hostname match is exact whatever its value +5 Stored lowercased, in ASCII form, and without a trailing dot. A row matches that hostname, and with `match_subdomains` true it also matches every subdomain. A value whose leftmost label has `*` is a [domain pattern](#domain-patterns) 6 Stored as lowercase hexadecimal @@ -45,6 +45,27 @@ Fluxer builds each permission name from `ban:`, the list name with each hyphen w The fields and the operations a list accepts differ from list to list. Read the `fields` array and the `supports_` flags of a [blocklist object](#blocklist-object) before writing to a list. +## Domain patterns + +A `url-domain` value whose leftmost label has `*` is a pattern. Each `*` stands for any run of characters inside that one label, so `*shop*.example.com` matches `shop.example.com` and `my-shop-2.example.com`. The rest of the value is a literal domain, called the suffix. + +A pattern only matches the label directly left of its suffix. It never matches the suffix itself, so `*shop*.example.com` does not block `example.com`. With `match_subdomains` true, the pattern also matches every subdomain of a matching host, such as `www.shop.example.com`. Blocking a whole suffix takes a plain domain row. + +Fluxer refuses a pattern with 400 `INVALID_FORM_BODY` naming `domain` when: + +- `*` appears outside the leftmost label +- the leftmost label has characters other than `a-z`, `0-9`, `-`, and `*` +- the leftmost label has more than 3 `*` or fewer than 3 other characters +- the suffix is not a valid domain or is a public suffix such as `com` or `co.uk` + +A domain under a shared hosting suffix, such as `onrender.com` or `github.io`, is a valid suffix. Fluxer folds a run of `*` into one before it stores the pattern. + +Fluxer matches a pattern against the ASCII form of a hostname. An internationalised label is compared in its punycode form, so a pattern with a leading or trailing literal can miss it, while `*shop*` still matches it. + +## Link matching + +Fluxer checks every link in user content against the `url` and `url-domain` lists. This covers bare links, autolinks in angle brackets, the target of a masked link, and the link URLs of rich embeds. Fluxer reads the hostname of each link the way a browser does. It ignores user information and the port, lowercases the hostname, converts it to ASCII, decodes percent escapes, and drops trailing dots. + ## Content blocklist categories Every `url`, `url-domain`, `file-sha`, and `avatar-hash` row has a category naming where it came from. A row created through this resource without an explicit category is stored as `manual`. @@ -119,7 +140,7 @@ One entry of the blocklist catalogue returned by [List blocklists](#list-blockli ```json { "list_type": "url-domain", - "description": "Domains blocked from being linked, optionally covering every subdomain rooted at the domain.", + "description": "Domains blocked from being linked, optionally covering every subdomain rooted at the domain. A value whose leftmost label contains * is a pattern that matches that one label under a registrable domain.", "value_field": "domain", "fields": ["match_subdomains", "category", "severity", "source_url", "notes"], "scoped": false, @@ -145,7 +166,7 @@ One stored row of one blocklist. Every field is present on every entry, and a fi | source_url3 | ?string | The feed or evidence URL the row was recorded from | | notes4 | ?string | The internal note stored alongside the row | | content_type5 | ?string | The media type hint recorded alongside a `file-sha` row | -| match_subdomains6 | ?boolean | Whether a `url-domain` row is flagged as covering subdomains | +| match_subdomains6 | ?boolean | Whether a `url-domain` row also covers subdomains | | reason7 | ?string | The reason stored on the row | | expires_at7 | ?ISO8601 timestamp | When the row expires | | created_at8 | ?ISO8601 timestamp | When the row was added | @@ -222,7 +243,7 @@ The body of [Add blocklist entry](#add-blocklist-entry) has one shape per blockl | email | email | Email address of 1 through 254 characters | | phrase | phrase | Phrase of 1 through 500 characters | | url | url | Absolute `http` or `https` URL of 1 through 2048 characters | -| url-domain | domain | Domain of 1 through 253 characters | +| url-domain | domain | Domain or [domain pattern](#domain-patterns) of 1 through 253 characters | | file-sha | sha256_hex | Exactly 64 hexadecimal characters | | avatar-hash1 | hashes | 1 through 1000 hashes, each 8 through 10 characters matching `^(a_)?[0-9a-fA-F]{8}$` | | profile-substring1 2 | substrings | 1 through 1000 substrings, each 1 through 500 characters | @@ -242,7 +263,7 @@ Each field is a string, except `hashes` and `substrings`, which are `array[strin | scope3 | string | The [profile substring scope](#profile-substring-scopes) to store the rows under | | category?4 | string | The [content blocklist category](#content-blocklist-categories) (1-64 characters, default `manual`) | | severity?4 | integer | The [content blocklist severity](#content-blocklist-severities) (0-3, default 2) | -| match_subdomains?5 | boolean | Whether the row is flagged as covering subdomains (default true) | +| match_subdomains?5 | boolean | Whether the row also covers subdomains (default true) | | content_type?6 | string | The media type hint (1-128 characters) | | source_url?4 | string | The feed or evidence URL (1-2048 characters) | | reason?7 | string | The reason (1-1024 characters) | @@ -371,7 +392,7 @@ Every write is an upsert on the canonical value. An omitted optional field is wr | 204 | empty | Values were written | | 400 | [error response](/admin-api/#error-response) | `INVALID_FORM_BODY` because the value is not valid for the list, or a field the blocklist does not accept was supplied | -A body field the selected blocklist does not accept is stripped and never produces that 400. A `url` Fluxer cannot canonicalise returns 400 `INVALID_FORM_BODY` naming `url` in the `errors` array. +A body field the selected blocklist does not accept is stripped and never produces that 400. A `url` Fluxer cannot canonicalise returns 400 `INVALID_FORM_BODY` naming `url` in the `errors` array. A `url-domain` value that is not a valid domain or [domain pattern](#domain-patterns) returns the same 400 naming `domain`. Fluxer refuses to add an `ip` with 400 `IP_BAN_DECLINED` when the address is on the instance exemption list. @@ -381,7 +402,7 @@ The response has no body, so it does not report the canonical form that was stor Fluxer checks later requests against the written rows. For every list except `email`, other nodes see the rows after a short propagation delay. No Gateway Dispatch is emitted. -Fluxer records one [Admin audit entry](/admin-api/#admin-audit-entry-object) per written value, with that value in its metadata. An `ip` added with a positive `duration_hours` adds the metadata keys `duration_hours` and `expires_at`. Fluxer records an entry for a refused `ip` too, under the action `ban_ip_skipped_exempt`, before it returns the 400. +Fluxer records one [Admin audit entry](/admin-api/#admin-audit-entry-object) per written value, with that value in its metadata. An `ip` added with a positive `duration_hours` adds the metadata keys `duration_hours` and `expires_at`. A `url-domain` entry also records `match_subdomains` and `pattern`, which is `true` for a [domain pattern](#domain-patterns). Fluxer records an entry for a refused `ip` too, under the action `ban_ip_skipped_exempt`, before it returns the 400. ### Rate limit @@ -518,7 +539,7 @@ Reports whether one value is currently blocked by the selected blocklist and ret | email | Exact match on the lowercased address | | phrase | Normalised phrase matching, so a disguised form of a stored phrase still reads as blocked | | url5 | Exact match on the canonicalised URL | -| url-domain | Exact match on the lowercased hostname | +| url-domain6 | The hostname, each parent domain stored with `match_subdomains` true, and each [domain pattern](#domain-patterns) covering the hostname | | file-sha | Exact match on the lowercased hexadecimal digest | | avatar-hash | Exact match after the `a_` prefix is stripped and the hash is lowercased | | profile-substring | Normalised substring matching within the named scope | @@ -529,6 +550,8 @@ Reports whether one value is currently blocked by the selected blocklist and ret 5 This check does not read the `url-domain` list, so a URL that a stored domain blocks reads as not blocked. Check the hostname separately +6 The value can be a hostname or an absolute `http` or `https` URL. Fluxer normalises the hostname as it does for [link matching](#link-matching) + ### Response | Status | Body | Condition | @@ -572,7 +595,7 @@ The body is the creation shape of the selected blocklist with the value field re | scope1 | string | The [profile substring scope](#profile-substring-scopes) to write the row under | | category?2 | string | The [content blocklist category](#content-blocklist-categories) (1-64 characters, default `manual`) | | severity?2 | integer | The [content blocklist severity](#content-blocklist-severities) (0-3, default 2) | -| match_subdomains?3 | boolean | Whether the row is flagged as covering subdomains (default true) | +| match_subdomains?3 | boolean | Whether the row also covers subdomains (default true) | | content_type?4 | string | The media type hint (1-128 characters) | | source_url?2 | string | The feed or evidence URL (1-2048 characters) | | reason?5 | string | The reason (1-1024 characters) | @@ -601,7 +624,7 @@ Sending `{"severity": 3}` on a `url` row also resets its category to `manual` an | 204 | empty | Row was written | | 400 | [error response](/admin-api/#error-response) | The list accepts no update, or the body fails validation, and the request returns `INVALID_FORM_BODY` | -On the `url` blocklist, an `entry_value` Fluxer cannot canonicalise returns 400 `INVALID_FORM_BODY` naming `url` in the `errors` array. +On the `url` blocklist, an `entry_value` Fluxer cannot canonicalise returns 400 `INVALID_FORM_BODY` naming `url` in the `errors` array. On the `url-domain` blocklist, an `entry_value` that is not a valid domain or [domain pattern](#domain-patterns) returns the same 400 naming `domain`. The write is an upsert on the canonical value, so a value with no existing row is created. This operation never returns 404. A `url` blocked only by a stored `url-domain` row has no row of its own, and updating it through the URL silently creates a new exact-URL row. @@ -630,7 +653,7 @@ Removes one row from the selected blocklist. Returns 204 with an empty body. Req | list_type | string | The [blocklist type](#blocklist-types) | | entry_value1 | string | The percent-encoded canonical value of the row to remove (1-2048 characters) | -1 Canonicalised the same way it was on the add path and matched exactly, so a `url` blocked only by a broader `url-domain` row cannot be removed through the URL +1 Canonicalised the same way it was on the add path and matched exactly, so a `url` blocked only by a broader `url-domain` row cannot be removed through the URL. A `url-domain` value that fails validation is lowercased and removed as written ### Query parameters diff --git a/fluxer_docs/src/content/docs/http-api/webhooks.mdx b/fluxer_docs/src/content/docs/http-api/webhooks.mdx index f789b47ce..c0e0347d0 100644 --- a/fluxer_docs/src/content/docs/http-api/webhooks.mdx +++ b/fluxer_docs/src/content/docs/http-api/webhooks.mdx @@ -20,7 +20,7 @@ Fluxer resolves the guild before every management operation. A guild that does n [List guild webhooks](#list-guild-webhooks), [List channel webhooks](#list-channel-webhooks), and [Create webhook](#create-webhook) name a guild or a channel in their paths, so the availability gate applies. A guild with [UNAVAILABLE_FOR_EVERYONE](/http-api/guilds/#guild-features) refuses an authenticated request with 403 `MISSING_ACCESS` before the operation runs. [UNAVAILABLE_FOR_EVERYONE_BUT_STAFF](/http-api/guilds/#guild-features) does the same for an account without the instance staff flag. The remaining management routes name only a webhook ID and are not gated. -Fluxer scans a submitted webhook `name` against the instance phrase and URL blocklists, and a match returns 403 `CONTENT_BLOCKED`. The same lists scan the resolved content and embed text of a created or edited message, and a match returns the same code. The `avatar` member is exempt from that scan, and Fluxer checks its decoded bytes against the banned asset hash list when it stores them. +Fluxer scans a submitted webhook `name` against the instance phrase and URL blocklists, and a match returns 403 `CONTENT_BLOCKED`. The same lists scan the resolved content, embed text, and embed link URLs of a created or edited message, and a match returns the same code. The `avatar` member is exempt from that scan, and Fluxer checks its decoded bytes against the banned asset hash list when it stores them. ## Rate limit keying diff --git a/packages/schema/src/domains/admin/AdminSchemas.ts b/packages/schema/src/domains/admin/AdminSchemas.ts index fff0f291d..9e6dcbe56 100644 --- a/packages/schema/src/domains/admin/AdminSchemas.ts +++ b/packages/schema/src/domains/admin/AdminSchemas.ts @@ -340,13 +340,13 @@ export const BanUrlRequest = z.object({ export type BanUrlRequest = z.infer; export const BanUrlDomainRequest = z.object({ - domain: createStringType(1, 253) - .refine( - (v) => /^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)+$/i.test(v), - 'Must be a valid domain', - ) - .describe('Domain to ban (e.g. example.com)'), - match_subdomains: z.boolean().default(true).describe('If true, any subdomain rooted at this domain is also banned'), + domain: createStringType(1, 253).describe( + 'Domain to ban (e.g. example.com), or a pattern whose leftmost label contains * under a registrable domain (e.g. *shop*.example.com). Internationalized names are stored in ASCII form.', + ), + match_subdomains: z + .boolean() + .default(true) + .describe('If true, any subdomain rooted at this domain, or at a host the pattern matches, is also banned'), category: createStringType(1, 64).optional().describe('Category / source slug (defaults to "manual")'), severity: z .number() diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index ec36fe5ef..860e7c9ec 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -393,6 +393,9 @@ catalogs: tiny-invariant: specifier: 1.3.3 version: 1.3.3 + tldts: + specifier: 7.4.13 + version: 7.4.13 transliteration: specifier: 2.6.1 version: 2.6.1 @@ -710,6 +713,9 @@ importers: tempy: specifier: 'catalog:' version: 3.2.0 + tldts: + specifier: 'catalog:' + version: 7.4.13 transliteration: specifier: 'catalog:' version: 2.6.1 diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index 4c2724b32..c9e1e2690 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -166,6 +166,7 @@ catalog: tempy: 3.2.0 thumbhash: 0.1.1 tiny-invariant: 1.3.3 + tldts: 7.4.13 transliteration: 2.6.1 ts-morph: 28.0.0 tsx: 4.23.13