feat(emoji): add Unicode 17 emoji and fix mixed skin tones (#2883)
@@ -52,69 +52,13 @@ async function renderSVGToBuffer(svgContent: string, size: number): Promise<Buff
|
||||
return sharp(Buffer.from(fixed)).resize(size, size).png().toBuffer();
|
||||
}
|
||||
|
||||
function hslToRgb(h: number, s: number, l: number): [number, number, number] {
|
||||
h = ((h % 360) + 360) % 360;
|
||||
h /= 360;
|
||||
let r: number, g: number, b: number;
|
||||
if (s === 0) {
|
||||
r = g = b = l;
|
||||
} else {
|
||||
const q = l < 0.5 ? l * (1 + s) : l + s - l * s;
|
||||
const p = 2 * l - q;
|
||||
const hueToRgb = (p: number, q: number, t: number): number => {
|
||||
if (t < 0) t += 1;
|
||||
if (t > 1) t -= 1;
|
||||
if (t < 1 / 6) return p + (q - p) * 6 * t;
|
||||
if (t < 1 / 2) return q;
|
||||
if (t < 2 / 3) return p + (q - p) * (2 / 3 - t) * 6;
|
||||
return p;
|
||||
};
|
||||
r = hueToRgb(p, q, h + 1 / 3);
|
||||
g = hueToRgb(p, q, h);
|
||||
b = hueToRgb(p, q, h - 1 / 3);
|
||||
}
|
||||
return [
|
||||
Math.round(Math.min(1, Math.max(0, r)) * 255),
|
||||
Math.round(Math.min(1, Math.max(0, g)) * 255),
|
||||
Math.round(Math.min(1, Math.max(0, b)) * 255),
|
||||
];
|
||||
}
|
||||
|
||||
async function createPlaceholder(size: number): Promise<Buffer> {
|
||||
const h = Math.random() * 360;
|
||||
const [r, g, b] = hslToRgb(h, 0.7, 0.6);
|
||||
const radius = Math.floor(size * 0.4);
|
||||
const cx = Math.floor(size / 2);
|
||||
const cy = Math.floor(size / 2);
|
||||
const svg = `<svg width="${size}" height="${size}" xmlns="http://www.w3.org/2000/svg">
|
||||
<circle cx="${cx}" cy="${cy}" r="${radius}" fill="rgb(${r},${g},${b})"/>
|
||||
</svg>`;
|
||||
return sharp(Buffer.from(svg)).png().toBuffer();
|
||||
}
|
||||
|
||||
async function loadEmojiImage(surrogate: string, size: number): Promise<Buffer> {
|
||||
const codepoint = convertToCodePoints(surrogate);
|
||||
const svg = loadLocalTwemojiSVG(codepoint);
|
||||
if (svg) {
|
||||
try {
|
||||
return await renderSVGToBuffer(svg, size);
|
||||
} catch (error) {
|
||||
console.error(`Failed to render SVG for ${codepoint}:`, error);
|
||||
}
|
||||
if (svg == null) {
|
||||
throw new Error(`Missing SVG for ${codepoint} (${surrogate})`);
|
||||
}
|
||||
if (codepoint.includes('-200d-')) {
|
||||
const basePart = codepoint.split('-200d-')[0];
|
||||
const baseSvg = loadLocalTwemojiSVG(basePart);
|
||||
if (baseSvg) {
|
||||
try {
|
||||
return await renderSVGToBuffer(baseSvg, size);
|
||||
} catch (error) {
|
||||
console.error(`Failed to render base SVG for ${basePart}:`, error);
|
||||
}
|
||||
}
|
||||
}
|
||||
console.error(`Missing SVG for ${codepoint} (${surrogate}), using placeholder`);
|
||||
return createPlaceholder(size);
|
||||
return renderSVGToBuffer(svg, size);
|
||||
}
|
||||
|
||||
async function renderSpriteSheet(
|
||||
@@ -164,11 +108,11 @@ async function renderSpriteSheet(
|
||||
}
|
||||
|
||||
async function generateMainSpriteSheet(
|
||||
emojiData: Record<string, Array<EmojiObject>>,
|
||||
categories: Record<string, Array<EmojiObject>>,
|
||||
outputDir: string,
|
||||
): Promise<void> {
|
||||
const base: Array<EmojiEntry> = [];
|
||||
for (const objs of Object.values(emojiData)) {
|
||||
for (const objs of Object.values(categories)) {
|
||||
for (const obj of objs) {
|
||||
base.push({surrogates: obj.surrogates});
|
||||
}
|
||||
@@ -177,7 +121,7 @@ async function generateMainSpriteSheet(
|
||||
}
|
||||
|
||||
async function generateSkinToneSpriteSheets(
|
||||
emojiData: Record<string, Array<EmojiObject>>,
|
||||
categories: Record<string, Array<EmojiObject>>,
|
||||
outputDir: string,
|
||||
): Promise<void> {
|
||||
const skinTones = ['\u{1F3FB}', '\u{1F3FC}', '\u{1F3FD}', '\u{1F3FE}', '\u{1F3FF}'];
|
||||
@@ -185,7 +129,7 @@ async function generateSkinToneSpriteSheets(
|
||||
const skinTone = skinTones[skinIndex];
|
||||
const skinCodepoint = convertToCodePoints(skinTone);
|
||||
const skinEntries: Array<EmojiEntry> = [];
|
||||
for (const objs of Object.values(emojiData)) {
|
||||
for (const objs of Object.values(categories)) {
|
||||
for (const obj of objs) {
|
||||
if (obj.skins && obj.skins.length > skinIndex && obj.skins[skinIndex].surrogates) {
|
||||
skinEntries.push({surrogates: obj.skins[skinIndex].surrogates});
|
||||
@@ -242,11 +186,11 @@ async function main(): Promise<void> {
|
||||
const outputDir = join(appDir, 'src', 'media', 'images', 'emoji-sprites');
|
||||
mkdirSync(outputDir, {recursive: true});
|
||||
const emojiDataPath = join(appDir, 'src', 'media', 'data', 'emojis.json');
|
||||
const emojiData: Record<string, Array<EmojiObject>> = JSON.parse(readFileSync(emojiDataPath, 'utf-8'));
|
||||
const emojiData: {categories: Record<string, Array<EmojiObject>>} = JSON.parse(readFileSync(emojiDataPath, 'utf-8'));
|
||||
console.log('Generating main sprite sheet...');
|
||||
await generateMainSpriteSheet(emojiData, outputDir);
|
||||
await generateMainSpriteSheet(emojiData.categories, outputDir);
|
||||
console.log('Generating skin tone sprite sheets...');
|
||||
await generateSkinToneSpriteSheets(emojiData, outputDir);
|
||||
await generateSkinToneSpriteSheets(emojiData.categories, outputDir);
|
||||
console.log('Generating picker sprite sheet...');
|
||||
await generatePickerSpriteSheet(outputDir);
|
||||
console.log('Emoji sprites generated successfully.');
|
||||
|
||||
@@ -1,8 +1,11 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
const EYE_IN_SPEECH_BUBBLE = '\u{1F441}\u200D\u{1F5E8}';
|
||||
|
||||
export function convertToCodePoints(emoji: string): string {
|
||||
const containsZWJ = emoji.includes('\u200D');
|
||||
const processedEmoji = containsZWJ ? emoji : emoji.replace(/\uFE0F/g, '');
|
||||
const emojiWithoutFE0F = emoji.replace(/\uFE0F/g, '');
|
||||
const keepsFE0F = emoji.includes('\u200D') && emojiWithoutFE0F !== EYE_IN_SPEECH_BUBBLE;
|
||||
const processedEmoji = keepsFE0F ? emoji : emojiWithoutFE0F;
|
||||
return Array.from(processedEmoji)
|
||||
.map((char) => char.codePointAt(0)?.toString(16).replace(/^0+/, '') || '')
|
||||
.join('-');
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
import type {UnicodeEmoji} from '@app/features/emoji/types/EmojiTypes';
|
||||
import * as EmojiUtils from '@app/features/expressions/utils/EmojiUtils';
|
||||
import type {EmojiSurrogateMatch} from '@app/features/messaging/utils/markdown/parser/EmojiParsers';
|
||||
import * as RegexUtils from '@app/features/messaging/utils/RegexUtils';
|
||||
import emojiData from '@app/media/data/emojis.json';
|
||||
import {SKIN_TONE_SURROGATES} from '@fluxer/constants/src/EmojiConstants';
|
||||
@@ -60,6 +61,15 @@ const categories = Object.freeze(Object.keys(emojiData.categories));
|
||||
|
||||
const toCanonicalSurrogate = (surrogate: string): string => surrogate.replace(/️/g, '');
|
||||
|
||||
const EYE_IN_SPEECH_BUBBLE_FULLY_QUALIFIED = '\u{1F441}\uFE0F\u200D\u{1F5E8}\uFE0F';
|
||||
const HAIR_COMPONENT_NAMES: Record<string, string> = {
|
||||
'\u{1F9B0}': 'red_hair',
|
||||
'\u{1F9B1}': 'curly_hair',
|
||||
'\u{1F9B3}': 'white_hair',
|
||||
'\u{1F9B2}': 'bald',
|
||||
};
|
||||
const REGIONAL_INDICATOR_PAIR_PATTERN = '\\uD83C[\\uDDE6-\\uDDFF]\\uD83C[\\uDDE6-\\uDDFF]';
|
||||
|
||||
let defaultSkinTone: string = '';
|
||||
|
||||
class UnicodeEmojiClass {
|
||||
@@ -216,6 +226,11 @@ function buildEmojiIndex(): EmojiIndex {
|
||||
surrogateToName[skinToneEntry.surrogatePair] = skinToneEntry.name;
|
||||
canonicalSurrogateToName[toCanonicalSurrogate(skinToneEntry.surrogatePair)] = skinToneEntry.name;
|
||||
});
|
||||
const skins = (emojiObject as {skins?: ReadonlyArray<{names: ReadonlyArray<string>; surrogates: string}>}).skins;
|
||||
skins?.slice(SKIN_TONE_SURROGATES.length).forEach((skin) => {
|
||||
surrogateToName[skin.surrogates] = skin.names[0];
|
||||
canonicalSurrogateToName[toCanonicalSurrogate(skin.surrogates)] = skin.names[0];
|
||||
});
|
||||
categoryByEmojiName[emoji.uniqueName] = category;
|
||||
emojis.push(emojiJson);
|
||||
return emojiJson;
|
||||
@@ -228,6 +243,20 @@ function buildEmojiIndex(): EmojiIndex {
|
||||
canonicalSurrogateToName[toCanonicalSurrogate(surrogatePair)] = `skin-tone-${index + 1}`;
|
||||
});
|
||||
|
||||
Object.entries(HAIR_COMPONENT_NAMES).forEach(([surrogate, name]) => {
|
||||
surrogateToName[surrogate] = name;
|
||||
canonicalSurrogateToName[surrogate] = name;
|
||||
});
|
||||
|
||||
Object.keys(surrogateToName)
|
||||
.filter((surrogate) => surrogate.includes('\u20E3'))
|
||||
.map(toCanonicalSurrogate)
|
||||
.concat(EYE_IN_SPEECH_BUBBLE_FULLY_QUALIFIED)
|
||||
.forEach((alias) => {
|
||||
const name = canonicalSurrogateToName[toCanonicalSurrogate(alias)];
|
||||
if (name) surrogateToName[alias] = name;
|
||||
});
|
||||
|
||||
const keywordOwner: Record<string, string> = {};
|
||||
const keywordCount: Record<string, number> = {};
|
||||
|
||||
@@ -275,7 +304,7 @@ function buildEmojiIndex(): EmojiIndex {
|
||||
emojis,
|
||||
skinToneSpriteCount,
|
||||
baseSpriteCount,
|
||||
emojiSurrogateRegex: new RegExp(`(${surrogateAlternation})`, 'g'),
|
||||
emojiSurrogateRegex: new RegExp(`(${REGIONAL_INDICATOR_PAIR_PATTERN}|${surrogateAlternation})`, 'g'),
|
||||
emojiShortcutRegex: new RegExp(`^(${shortcutAlternation})`),
|
||||
};
|
||||
}
|
||||
@@ -288,6 +317,50 @@ function getEmojiIndex(): EmojiIndex {
|
||||
const lookupSurrogateName = (surrogate: string): string | null =>
|
||||
getEmojiIndex().canonicalSurrogateToName[toCanonicalSurrogate(surrogate)] ?? null;
|
||||
|
||||
const isRegionalIndicatorAt = (text: string, index: number): boolean => {
|
||||
const lowSurrogate = text.charCodeAt(index + 1);
|
||||
return text.charCodeAt(index) === 0xd83c && lowSurrogate >= 0xdde6 && lowSurrogate <= 0xddff;
|
||||
};
|
||||
|
||||
const isInsideRegionalIndicatorPair = (text: string, index: number): boolean => {
|
||||
if (!isRegionalIndicatorAt(text, index)) return false;
|
||||
let runStart = index;
|
||||
while (isRegionalIndicatorAt(text, runStart - 2)) runStart -= 2;
|
||||
return (index - runStart) % 4 === 2;
|
||||
};
|
||||
|
||||
const toEmojiSurrogateMatch = (text: string, start: number, end: number): EmojiSurrogateMatch => ({
|
||||
start,
|
||||
end,
|
||||
name: lookupSurrogateName(text.slice(start, end)),
|
||||
});
|
||||
|
||||
function* matchEmojiSurrogates(text: string, startIndex = 0): Generator<EmojiSurrogateMatch, void> {
|
||||
const regex = getEmojiIndex().emojiSurrogateRegex;
|
||||
let index = startIndex;
|
||||
if (isInsideRegionalIndicatorPair(text, index)) {
|
||||
yield toEmojiSurrogateMatch(text, index, index + 2);
|
||||
index += 2;
|
||||
}
|
||||
while (true) {
|
||||
regex.lastIndex = index;
|
||||
const match = regex.exec(text);
|
||||
if (match == null) return;
|
||||
index = match.index + match[0].length;
|
||||
const emojiMatch = toEmojiSurrogateMatch(text, match.index, index);
|
||||
const isUnknownRegionalIndicatorPair =
|
||||
emojiMatch.name == null &&
|
||||
isRegionalIndicatorAt(text, match.index) &&
|
||||
isRegionalIndicatorAt(text, match.index + 2);
|
||||
if (isUnknownRegionalIndicatorPair) {
|
||||
yield toEmojiSurrogateMatch(text, match.index, match.index + 2);
|
||||
yield toEmojiSurrogateMatch(text, match.index + 2, index);
|
||||
} else {
|
||||
yield emojiMatch;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const EMOJI_SHORTCODE_RE = /^:([^\s:]+(?:::skin-tone-[0-9])?):/;
|
||||
const categoryIcons = {
|
||||
people: SmileyIcon,
|
||||
@@ -377,6 +450,8 @@ export default {
|
||||
getSurrogateName: (surrogate: string): string | null => {
|
||||
return lookupSurrogateName(surrogate);
|
||||
},
|
||||
matchEmojiSurrogates,
|
||||
isInsideRegionalIndicatorPair,
|
||||
findEmojiByName: (emojiName: string): UnicodeEmoji | null => {
|
||||
return getEmojiIndex().nameToEmoji[emojiName] || null;
|
||||
},
|
||||
@@ -400,8 +475,5 @@ export default {
|
||||
get EMOTICON_PREFIX_RE(): RegExp {
|
||||
return getEmojiIndex().emojiShortcutRegex;
|
||||
},
|
||||
get EMOJI_SURROGATE_RE(): RegExp {
|
||||
return getEmojiIndex().emojiSurrogateRegex;
|
||||
},
|
||||
EMOJI_SPRITES,
|
||||
};
|
||||
|
||||
@@ -29,17 +29,19 @@ export function registerComposerEmojiShortcode(editor: LexicalEditor, resolve: C
|
||||
}
|
||||
|
||||
function findUnicodeEmoji(text: string, startIndex: number): TypedEmojiMatch | null {
|
||||
const pattern = new RegExp(UnicodeEmojis.EMOJI_SURROGATE_RE.source, 'g');
|
||||
pattern.lastIndex = startIndex;
|
||||
const match = pattern.exec(text);
|
||||
if (match == null) {
|
||||
const matches = UnicodeEmojis.matchEmojiSurrogates(text, startIndex);
|
||||
let match = matches.next().value;
|
||||
if (match && UnicodeEmojis.isInsideRegionalIndicatorPair(text, match.end)) {
|
||||
match = matches.next().value;
|
||||
}
|
||||
if (!match) {
|
||||
return null;
|
||||
}
|
||||
const name = UnicodeEmojis.nameForSurrogate(match[0], false);
|
||||
const name = match.name;
|
||||
if (!name) {
|
||||
return null;
|
||||
}
|
||||
return {start: match.index, end: match.index + match[0].length, name};
|
||||
return {start: match.start, end: match.end, name};
|
||||
}
|
||||
|
||||
interface EmojiToken extends TypedEmojiMatch {
|
||||
|
||||
@@ -5,7 +5,7 @@ import {type EmojiProvider, setEmojiParserConfig} from '@app/features/messaging/
|
||||
import {SKIN_TONE_SURROGATES} from '@fluxer/constants/src/EmojiConstants';
|
||||
|
||||
const emojiProvider: EmojiProvider = {
|
||||
getSurrogateName: UnicodeEmojis.getSurrogateName,
|
||||
matchEmojiSurrogates: UnicodeEmojis.matchEmojiSurrogates,
|
||||
findEmojiByName: UnicodeEmojis.findEmojiByShortcodeName,
|
||||
findEmojiWithSkinTone: UnicodeEmojis.findEmojiWithSkinTone,
|
||||
};
|
||||
@@ -13,9 +13,6 @@ const emojiProvider: EmojiProvider = {
|
||||
export function initializeEmojiParser(): void {
|
||||
setEmojiParserConfig({
|
||||
emojiProvider,
|
||||
get emojiRegex() {
|
||||
return UnicodeEmojis.EMOJI_SURROGATE_RE;
|
||||
},
|
||||
skinToneSurrogates: SKIN_TONE_SURROGATES,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -4,17 +4,21 @@ export interface UnicodeEmoji {
|
||||
surrogates: string;
|
||||
}
|
||||
|
||||
export interface EmojiSurrogateMatch {
|
||||
start: number;
|
||||
end: number;
|
||||
name: string | null;
|
||||
}
|
||||
|
||||
export interface EmojiProvider {
|
||||
getSurrogateName(surrogate: string): string | null;
|
||||
matchEmojiSurrogates(text: string): Iterable<EmojiSurrogateMatch>;
|
||||
findEmojiByName(name: string): UnicodeEmoji | null;
|
||||
findEmojiWithSkinTone(baseName: string, skinToneSurrogate: string): UnicodeEmoji | null;
|
||||
}
|
||||
|
||||
export interface EmojiParserConfig {
|
||||
emojiProvider?: EmojiProvider;
|
||||
emojiRegex?: RegExp;
|
||||
skinToneSurrogates?: ReadonlyArray<string>;
|
||||
convertToCodePoints?: (emoji: string) => string;
|
||||
}
|
||||
|
||||
let globalEmojiConfig: EmojiParserConfig | null = null;
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
import {convertToCodePoints} from '@app/features/expressions/utils/EmojiCodepointUtils';
|
||||
import {flattenAST} from '@app/features/messaging/utils/markdown/parser/AstUtils';
|
||||
import {getEmojiParserConfig} from '@app/features/messaging/utils/markdown/parser/EmojiParsers';
|
||||
import {MARKDOWN_PARSER_WASM_BASE64} from '@app/features/messaging/utils/markdown/parser/MarkdownParserWasmBytes';
|
||||
@@ -202,14 +203,6 @@ class Utf8OffsetTracker {
|
||||
}
|
||||
}
|
||||
|
||||
function defaultCodepoints(emoji: string): string {
|
||||
const containsZwJ = emoji.includes('');
|
||||
const processed = containsZwJ ? emoji : emoji.replace(/️/g, '');
|
||||
return Array.from(processed)
|
||||
.map((char) => char.codePointAt(0)?.toString(16).replace(/^0+/, '') || '')
|
||||
.join('-');
|
||||
}
|
||||
|
||||
const PLAINTEXT_SYMBOLS = new Set(['™', '™️', '©', '©️', '®', '®️']);
|
||||
const SPECIAL_SHORTCODES: Record<string, string> = {
|
||||
tm: '™',
|
||||
@@ -226,29 +219,23 @@ function buildEmojiContext(input: string): string {
|
||||
const provider = config?.emojiProvider;
|
||||
let context = '';
|
||||
if (!provider) return context;
|
||||
const convertToCodePoints = config.convertToCodePoints || defaultCodepoints;
|
||||
const emojiRegex = config.emojiRegex;
|
||||
if (emojiRegex) {
|
||||
const offsetTracker = new Utf8OffsetTracker();
|
||||
emojiRegex.lastIndex = 0;
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = emojiRegex.exec(input)) !== null) {
|
||||
const candidate = match[0];
|
||||
if (!candidate || PLAINTEXT_SYMBOLS.has(candidate)) continue;
|
||||
const name = provider.getSurrogateName(candidate);
|
||||
if (!name) continue;
|
||||
const candidateBytes = textEncoder.encode(candidate).byteLength;
|
||||
const byteOffset = offsetTracker.offsetFor(input, match.index);
|
||||
context += appendContextLine([
|
||||
'S',
|
||||
String(byteOffset),
|
||||
String(candidateBytes),
|
||||
candidate,
|
||||
name,
|
||||
convertToCodePoints(candidate),
|
||||
]);
|
||||
offsetTracker.advance(candidate.length, candidateBytes);
|
||||
}
|
||||
const offsetTracker = new Utf8OffsetTracker();
|
||||
for (const match of provider.matchEmojiSurrogates(input)) {
|
||||
const candidate = input.slice(match.start, match.end);
|
||||
if (PLAINTEXT_SYMBOLS.has(candidate)) continue;
|
||||
const name = match.name;
|
||||
if (!name) continue;
|
||||
const candidateBytes = textEncoder.encode(candidate).byteLength;
|
||||
const byteOffset = offsetTracker.offsetFor(input, match.start);
|
||||
context += appendContextLine([
|
||||
'S',
|
||||
String(byteOffset),
|
||||
String(candidateBytes),
|
||||
candidate,
|
||||
name,
|
||||
convertToCodePoints(candidate),
|
||||
]);
|
||||
offsetTracker.advance(candidate.length, candidateBytes);
|
||||
}
|
||||
const shortcodeRegex = /:([\p{L}\p{N}_-]+):/gu;
|
||||
const seen = new Set<string>();
|
||||
|
||||
|
Before Width: | Height: | Size: 270 KiB After Width: | Height: | Size: 279 KiB |
|
Before Width: | Height: | Size: 592 KiB After Width: | Height: | Size: 612 KiB |
|
Before Width: | Height: | Size: 268 KiB After Width: | Height: | Size: 277 KiB |
|
Before Width: | Height: | Size: 589 KiB After Width: | Height: | Size: 608 KiB |
|
Before Width: | Height: | Size: 274 KiB After Width: | Height: | Size: 283 KiB |
|
Before Width: | Height: | Size: 602 KiB After Width: | Height: | Size: 621 KiB |
|
Before Width: | Height: | Size: 269 KiB After Width: | Height: | Size: 277 KiB |
|
Before Width: | Height: | Size: 590 KiB After Width: | Height: | Size: 608 KiB |
|
Before Width: | Height: | Size: 267 KiB After Width: | Height: | Size: 275 KiB |
|
Before Width: | Height: | Size: 585 KiB After Width: | Height: | Size: 603 KiB |
|
Before Width: | Height: | Size: 1.4 MiB After Width: | Height: | Size: 1.4 MiB |
|
Before Width: | Height: | Size: 3.1 MiB After Width: | Height: | Size: 3.2 MiB |
@@ -2,6 +2,7 @@
|
||||
|
||||
use crate::ast::{EmojiKind, Node, ParserFlags, ParserResult};
|
||||
use crate::constants::{CODE_FENCE_LENGTH, MAX_INLINE_DEPTH, MAX_LINE_LENGTH};
|
||||
use crate::emoji::{EmojiContext, StandardEmoji};
|
||||
use crate::links;
|
||||
use crate::normalize::{
|
||||
combine_adjacent_text, compact_empty_text_nodes, flatten_top_level_formatting,
|
||||
@@ -168,7 +169,13 @@ fn parse_inline_with_context(
|
||||
|
||||
if let Some(result) = parse_regional_indicator_flag(remaining) {
|
||||
flush_accumulated_text(&mut nodes, &mut accumulated);
|
||||
nodes.push(result.node);
|
||||
push_regional_indicator_pair(
|
||||
&mut nodes,
|
||||
parser.emoji_context(),
|
||||
base_offset + position,
|
||||
&remaining[..result.advance],
|
||||
result.node,
|
||||
);
|
||||
position += result.advance;
|
||||
continue;
|
||||
}
|
||||
@@ -349,6 +356,49 @@ fn parse_regional_indicator_flag(text: &str) -> Option<ParserResult> {
|
||||
})
|
||||
}
|
||||
|
||||
fn push_regional_indicator_pair(
|
||||
nodes: &mut Vec<Node>,
|
||||
emoji_context: &EmojiContext,
|
||||
offset: usize,
|
||||
pair: &str,
|
||||
flag: Node,
|
||||
) {
|
||||
let (first, second) = pair.split_at(pair.len() / 2);
|
||||
match emoji_context.standard_at(offset) {
|
||||
Some(emoji) if emoji.raw == pair => nodes.push(standard_emoji_node(emoji)),
|
||||
Some(emoji) if emoji.raw == first => {
|
||||
nodes.push(standard_emoji_node(emoji));
|
||||
match emoji_context.standard_at(offset + first.len()) {
|
||||
Some(emoji) if emoji.raw == second => nodes.push(standard_emoji_node(emoji)),
|
||||
_ => nodes.extend(regional_indicator_emoji(second)),
|
||||
}
|
||||
}
|
||||
_ => nodes.push(flag),
|
||||
}
|
||||
}
|
||||
|
||||
fn standard_emoji_node(emoji: &StandardEmoji) -> Node {
|
||||
Node::Emoji {
|
||||
kind: EmojiKind::Standard {
|
||||
raw: emoji.raw.clone(),
|
||||
codepoints: emoji.codepoints.clone(),
|
||||
name: emoji.name.clone(),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
fn regional_indicator_emoji(text: &str) -> Option<Node> {
|
||||
let ch = text.chars().next()?;
|
||||
let letter = regional_indicator_letter(ch)?;
|
||||
Some(Node::Emoji {
|
||||
kind: EmojiKind::Standard {
|
||||
raw: ch.to_string(),
|
||||
codepoints: format!("{:x}", ch as u32),
|
||||
name: format!("regional_indicator_{letter}"),
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
fn regional_indicator_letter(ch: char) -> Option<char> {
|
||||
let codepoint = ch as u32;
|
||||
if !(0x1f1e6..=0x1f1ff).contains(&codepoint) {
|
||||
|
||||
@@ -151,6 +151,67 @@ fn native_parser_fixtures_cover_typescript_suite_surface() {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn native_parser_pairs_regional_indicators_with_emoji_context() {
|
||||
assert_eq!(
|
||||
parse(
|
||||
"🇦🇧🇸🇪",
|
||||
0,
|
||||
"S\t0\t4\t🇦\tregional_indicator_a\t1f1e6\nS\t4\t4\t🇧\tregional_indicator_b\t1f1e7\nS\t8\t8\t🇸🇪\tflag_se\t1f1f8-1f1ea\n"
|
||||
),
|
||||
json!({"nodes":[
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇧","codepoints":"1f1e7","name":"regional_indicator_b"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇸🇪","codepoints":"1f1f8-1f1ea","name":"flag_se"}}
|
||||
]})
|
||||
);
|
||||
assert_eq!(
|
||||
parse(
|
||||
"🇦🇧🇦🇸",
|
||||
0,
|
||||
"S\t0\t4\t🇦\tregional_indicator_a\t1f1e6\nS\t4\t4\t🇧\tregional_indicator_b\t1f1e7\nS\t8\t8\t🇦🇸\tflag_as\t1f1e6-1f1f8\n"
|
||||
),
|
||||
json!({"nodes":[
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇧","codepoints":"1f1e7","name":"regional_indicator_b"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦🇸","codepoints":"1f1e6-1f1f8","name":"flag_as"}}
|
||||
]})
|
||||
);
|
||||
assert_eq!(
|
||||
parse("Sark 🇨🇶", 0, "S\t5\t8\t🇨🇶\tflag_sark\t1f1e8-1f1f6\n"),
|
||||
json!({"nodes":[
|
||||
{"type":"Text","content":"Sark "},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇨🇶","codepoints":"1f1e8-1f1f6","name":"flag_sark"}}
|
||||
]})
|
||||
);
|
||||
assert_eq!(
|
||||
parse("x 🇦", 0, "S\t2\t4\t🇦\tregional_indicator_a\t1f1e6\n"),
|
||||
json!({"nodes":[
|
||||
{"type":"Text","content":"x "},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}}
|
||||
]})
|
||||
);
|
||||
assert_eq!(
|
||||
parse(
|
||||
"🇦🇧🇸🇪",
|
||||
0,
|
||||
"S\t0\t4\t🇦\tregional_indicator_a\t1f1e6\nS\t4\t8\t🇧🇸\tflag_bs\t1f1e7-1f1f8\nS\t12\t4\t🇪\tregional_indicator_e\t1f1ea\n"
|
||||
),
|
||||
json!({"nodes":[
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇧","codepoints":"1f1e7","name":"regional_indicator_b"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇸🇪","codepoints":"1f1f8-1f1ea","name":"flag_se"}}
|
||||
]})
|
||||
);
|
||||
assert_eq!(
|
||||
parse("🇦🇧🇸🇪", 0, ""),
|
||||
json!({"nodes":[
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦🇧","codepoints":"1f1e6-1f1e7","name":"flag_ab"}},
|
||||
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇸🇪","codepoints":"1f1f8-1f1ea","name":"flag_se"}}
|
||||
]})
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn native_parser_allows_apostrophe_in_masked_link_destination() {
|
||||
let url = "https://docs.example.test/help/faq/#why-can't-this-link-parse%3F";
|
||||
|
||||