feat(emoji): add Unicode 17 emoji and fix mixed skin tones (#2883)
@@ -52,69 +52,13 @@ async function renderSVGToBuffer(svgContent: string, size: number): Promise<Buff
|
|||||||
return sharp(Buffer.from(fixed)).resize(size, size).png().toBuffer();
|
return sharp(Buffer.from(fixed)).resize(size, size).png().toBuffer();
|
||||||
}
|
}
|
||||||
|
|
||||||
function hslToRgb(h: number, s: number, l: number): [number, number, number] {
|
|
||||||
h = ((h % 360) + 360) % 360;
|
|
||||||
h /= 360;
|
|
||||||
let r: number, g: number, b: number;
|
|
||||||
if (s === 0) {
|
|
||||||
r = g = b = l;
|
|
||||||
} else {
|
|
||||||
const q = l < 0.5 ? l * (1 + s) : l + s - l * s;
|
|
||||||
const p = 2 * l - q;
|
|
||||||
const hueToRgb = (p: number, q: number, t: number): number => {
|
|
||||||
if (t < 0) t += 1;
|
|
||||||
if (t > 1) t -= 1;
|
|
||||||
if (t < 1 / 6) return p + (q - p) * 6 * t;
|
|
||||||
if (t < 1 / 2) return q;
|
|
||||||
if (t < 2 / 3) return p + (q - p) * (2 / 3 - t) * 6;
|
|
||||||
return p;
|
|
||||||
};
|
|
||||||
r = hueToRgb(p, q, h + 1 / 3);
|
|
||||||
g = hueToRgb(p, q, h);
|
|
||||||
b = hueToRgb(p, q, h - 1 / 3);
|
|
||||||
}
|
|
||||||
return [
|
|
||||||
Math.round(Math.min(1, Math.max(0, r)) * 255),
|
|
||||||
Math.round(Math.min(1, Math.max(0, g)) * 255),
|
|
||||||
Math.round(Math.min(1, Math.max(0, b)) * 255),
|
|
||||||
];
|
|
||||||
}
|
|
||||||
|
|
||||||
async function createPlaceholder(size: number): Promise<Buffer> {
|
|
||||||
const h = Math.random() * 360;
|
|
||||||
const [r, g, b] = hslToRgb(h, 0.7, 0.6);
|
|
||||||
const radius = Math.floor(size * 0.4);
|
|
||||||
const cx = Math.floor(size / 2);
|
|
||||||
const cy = Math.floor(size / 2);
|
|
||||||
const svg = `<svg width="${size}" height="${size}" xmlns="http://www.w3.org/2000/svg">
|
|
||||||
<circle cx="${cx}" cy="${cy}" r="${radius}" fill="rgb(${r},${g},${b})"/>
|
|
||||||
</svg>`;
|
|
||||||
return sharp(Buffer.from(svg)).png().toBuffer();
|
|
||||||
}
|
|
||||||
|
|
||||||
async function loadEmojiImage(surrogate: string, size: number): Promise<Buffer> {
|
async function loadEmojiImage(surrogate: string, size: number): Promise<Buffer> {
|
||||||
const codepoint = convertToCodePoints(surrogate);
|
const codepoint = convertToCodePoints(surrogate);
|
||||||
const svg = loadLocalTwemojiSVG(codepoint);
|
const svg = loadLocalTwemojiSVG(codepoint);
|
||||||
if (svg) {
|
if (svg == null) {
|
||||||
try {
|
throw new Error(`Missing SVG for ${codepoint} (${surrogate})`);
|
||||||
return await renderSVGToBuffer(svg, size);
|
|
||||||
} catch (error) {
|
|
||||||
console.error(`Failed to render SVG for ${codepoint}:`, error);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
if (codepoint.includes('-200d-')) {
|
return renderSVGToBuffer(svg, size);
|
||||||
const basePart = codepoint.split('-200d-')[0];
|
|
||||||
const baseSvg = loadLocalTwemojiSVG(basePart);
|
|
||||||
if (baseSvg) {
|
|
||||||
try {
|
|
||||||
return await renderSVGToBuffer(baseSvg, size);
|
|
||||||
} catch (error) {
|
|
||||||
console.error(`Failed to render base SVG for ${basePart}:`, error);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
console.error(`Missing SVG for ${codepoint} (${surrogate}), using placeholder`);
|
|
||||||
return createPlaceholder(size);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async function renderSpriteSheet(
|
async function renderSpriteSheet(
|
||||||
@@ -164,11 +108,11 @@ async function renderSpriteSheet(
|
|||||||
}
|
}
|
||||||
|
|
||||||
async function generateMainSpriteSheet(
|
async function generateMainSpriteSheet(
|
||||||
emojiData: Record<string, Array<EmojiObject>>,
|
categories: Record<string, Array<EmojiObject>>,
|
||||||
outputDir: string,
|
outputDir: string,
|
||||||
): Promise<void> {
|
): Promise<void> {
|
||||||
const base: Array<EmojiEntry> = [];
|
const base: Array<EmojiEntry> = [];
|
||||||
for (const objs of Object.values(emojiData)) {
|
for (const objs of Object.values(categories)) {
|
||||||
for (const obj of objs) {
|
for (const obj of objs) {
|
||||||
base.push({surrogates: obj.surrogates});
|
base.push({surrogates: obj.surrogates});
|
||||||
}
|
}
|
||||||
@@ -177,7 +121,7 @@ async function generateMainSpriteSheet(
|
|||||||
}
|
}
|
||||||
|
|
||||||
async function generateSkinToneSpriteSheets(
|
async function generateSkinToneSpriteSheets(
|
||||||
emojiData: Record<string, Array<EmojiObject>>,
|
categories: Record<string, Array<EmojiObject>>,
|
||||||
outputDir: string,
|
outputDir: string,
|
||||||
): Promise<void> {
|
): Promise<void> {
|
||||||
const skinTones = ['\u{1F3FB}', '\u{1F3FC}', '\u{1F3FD}', '\u{1F3FE}', '\u{1F3FF}'];
|
const skinTones = ['\u{1F3FB}', '\u{1F3FC}', '\u{1F3FD}', '\u{1F3FE}', '\u{1F3FF}'];
|
||||||
@@ -185,7 +129,7 @@ async function generateSkinToneSpriteSheets(
|
|||||||
const skinTone = skinTones[skinIndex];
|
const skinTone = skinTones[skinIndex];
|
||||||
const skinCodepoint = convertToCodePoints(skinTone);
|
const skinCodepoint = convertToCodePoints(skinTone);
|
||||||
const skinEntries: Array<EmojiEntry> = [];
|
const skinEntries: Array<EmojiEntry> = [];
|
||||||
for (const objs of Object.values(emojiData)) {
|
for (const objs of Object.values(categories)) {
|
||||||
for (const obj of objs) {
|
for (const obj of objs) {
|
||||||
if (obj.skins && obj.skins.length > skinIndex && obj.skins[skinIndex].surrogates) {
|
if (obj.skins && obj.skins.length > skinIndex && obj.skins[skinIndex].surrogates) {
|
||||||
skinEntries.push({surrogates: obj.skins[skinIndex].surrogates});
|
skinEntries.push({surrogates: obj.skins[skinIndex].surrogates});
|
||||||
@@ -242,11 +186,11 @@ async function main(): Promise<void> {
|
|||||||
const outputDir = join(appDir, 'src', 'media', 'images', 'emoji-sprites');
|
const outputDir = join(appDir, 'src', 'media', 'images', 'emoji-sprites');
|
||||||
mkdirSync(outputDir, {recursive: true});
|
mkdirSync(outputDir, {recursive: true});
|
||||||
const emojiDataPath = join(appDir, 'src', 'media', 'data', 'emojis.json');
|
const emojiDataPath = join(appDir, 'src', 'media', 'data', 'emojis.json');
|
||||||
const emojiData: Record<string, Array<EmojiObject>> = JSON.parse(readFileSync(emojiDataPath, 'utf-8'));
|
const emojiData: {categories: Record<string, Array<EmojiObject>>} = JSON.parse(readFileSync(emojiDataPath, 'utf-8'));
|
||||||
console.log('Generating main sprite sheet...');
|
console.log('Generating main sprite sheet...');
|
||||||
await generateMainSpriteSheet(emojiData, outputDir);
|
await generateMainSpriteSheet(emojiData.categories, outputDir);
|
||||||
console.log('Generating skin tone sprite sheets...');
|
console.log('Generating skin tone sprite sheets...');
|
||||||
await generateSkinToneSpriteSheets(emojiData, outputDir);
|
await generateSkinToneSpriteSheets(emojiData.categories, outputDir);
|
||||||
console.log('Generating picker sprite sheet...');
|
console.log('Generating picker sprite sheet...');
|
||||||
await generatePickerSpriteSheet(outputDir);
|
await generatePickerSpriteSheet(outputDir);
|
||||||
console.log('Emoji sprites generated successfully.');
|
console.log('Emoji sprites generated successfully.');
|
||||||
|
|||||||
@@ -1,8 +1,11 @@
|
|||||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
|
|
||||||
|
const EYE_IN_SPEECH_BUBBLE = '\u{1F441}\u200D\u{1F5E8}';
|
||||||
|
|
||||||
export function convertToCodePoints(emoji: string): string {
|
export function convertToCodePoints(emoji: string): string {
|
||||||
const containsZWJ = emoji.includes('\u200D');
|
const emojiWithoutFE0F = emoji.replace(/\uFE0F/g, '');
|
||||||
const processedEmoji = containsZWJ ? emoji : emoji.replace(/\uFE0F/g, '');
|
const keepsFE0F = emoji.includes('\u200D') && emojiWithoutFE0F !== EYE_IN_SPEECH_BUBBLE;
|
||||||
|
const processedEmoji = keepsFE0F ? emoji : emojiWithoutFE0F;
|
||||||
return Array.from(processedEmoji)
|
return Array.from(processedEmoji)
|
||||||
.map((char) => char.codePointAt(0)?.toString(16).replace(/^0+/, '') || '')
|
.map((char) => char.codePointAt(0)?.toString(16).replace(/^0+/, '') || '')
|
||||||
.join('-');
|
.join('-');
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
import type {UnicodeEmoji} from '@app/features/emoji/types/EmojiTypes';
|
import type {UnicodeEmoji} from '@app/features/emoji/types/EmojiTypes';
|
||||||
import * as EmojiUtils from '@app/features/expressions/utils/EmojiUtils';
|
import * as EmojiUtils from '@app/features/expressions/utils/EmojiUtils';
|
||||||
|
import type {EmojiSurrogateMatch} from '@app/features/messaging/utils/markdown/parser/EmojiParsers';
|
||||||
import * as RegexUtils from '@app/features/messaging/utils/RegexUtils';
|
import * as RegexUtils from '@app/features/messaging/utils/RegexUtils';
|
||||||
import emojiData from '@app/media/data/emojis.json';
|
import emojiData from '@app/media/data/emojis.json';
|
||||||
import {SKIN_TONE_SURROGATES} from '@fluxer/constants/src/EmojiConstants';
|
import {SKIN_TONE_SURROGATES} from '@fluxer/constants/src/EmojiConstants';
|
||||||
@@ -60,6 +61,15 @@ const categories = Object.freeze(Object.keys(emojiData.categories));
|
|||||||
|
|
||||||
const toCanonicalSurrogate = (surrogate: string): string => surrogate.replace(/️/g, '');
|
const toCanonicalSurrogate = (surrogate: string): string => surrogate.replace(/️/g, '');
|
||||||
|
|
||||||
|
const EYE_IN_SPEECH_BUBBLE_FULLY_QUALIFIED = '\u{1F441}\uFE0F\u200D\u{1F5E8}\uFE0F';
|
||||||
|
const HAIR_COMPONENT_NAMES: Record<string, string> = {
|
||||||
|
'\u{1F9B0}': 'red_hair',
|
||||||
|
'\u{1F9B1}': 'curly_hair',
|
||||||
|
'\u{1F9B3}': 'white_hair',
|
||||||
|
'\u{1F9B2}': 'bald',
|
||||||
|
};
|
||||||
|
const REGIONAL_INDICATOR_PAIR_PATTERN = '\\uD83C[\\uDDE6-\\uDDFF]\\uD83C[\\uDDE6-\\uDDFF]';
|
||||||
|
|
||||||
let defaultSkinTone: string = '';
|
let defaultSkinTone: string = '';
|
||||||
|
|
||||||
class UnicodeEmojiClass {
|
class UnicodeEmojiClass {
|
||||||
@@ -216,6 +226,11 @@ function buildEmojiIndex(): EmojiIndex {
|
|||||||
surrogateToName[skinToneEntry.surrogatePair] = skinToneEntry.name;
|
surrogateToName[skinToneEntry.surrogatePair] = skinToneEntry.name;
|
||||||
canonicalSurrogateToName[toCanonicalSurrogate(skinToneEntry.surrogatePair)] = skinToneEntry.name;
|
canonicalSurrogateToName[toCanonicalSurrogate(skinToneEntry.surrogatePair)] = skinToneEntry.name;
|
||||||
});
|
});
|
||||||
|
const skins = (emojiObject as {skins?: ReadonlyArray<{names: ReadonlyArray<string>; surrogates: string}>}).skins;
|
||||||
|
skins?.slice(SKIN_TONE_SURROGATES.length).forEach((skin) => {
|
||||||
|
surrogateToName[skin.surrogates] = skin.names[0];
|
||||||
|
canonicalSurrogateToName[toCanonicalSurrogate(skin.surrogates)] = skin.names[0];
|
||||||
|
});
|
||||||
categoryByEmojiName[emoji.uniqueName] = category;
|
categoryByEmojiName[emoji.uniqueName] = category;
|
||||||
emojis.push(emojiJson);
|
emojis.push(emojiJson);
|
||||||
return emojiJson;
|
return emojiJson;
|
||||||
@@ -228,6 +243,20 @@ function buildEmojiIndex(): EmojiIndex {
|
|||||||
canonicalSurrogateToName[toCanonicalSurrogate(surrogatePair)] = `skin-tone-${index + 1}`;
|
canonicalSurrogateToName[toCanonicalSurrogate(surrogatePair)] = `skin-tone-${index + 1}`;
|
||||||
});
|
});
|
||||||
|
|
||||||
|
Object.entries(HAIR_COMPONENT_NAMES).forEach(([surrogate, name]) => {
|
||||||
|
surrogateToName[surrogate] = name;
|
||||||
|
canonicalSurrogateToName[surrogate] = name;
|
||||||
|
});
|
||||||
|
|
||||||
|
Object.keys(surrogateToName)
|
||||||
|
.filter((surrogate) => surrogate.includes('\u20E3'))
|
||||||
|
.map(toCanonicalSurrogate)
|
||||||
|
.concat(EYE_IN_SPEECH_BUBBLE_FULLY_QUALIFIED)
|
||||||
|
.forEach((alias) => {
|
||||||
|
const name = canonicalSurrogateToName[toCanonicalSurrogate(alias)];
|
||||||
|
if (name) surrogateToName[alias] = name;
|
||||||
|
});
|
||||||
|
|
||||||
const keywordOwner: Record<string, string> = {};
|
const keywordOwner: Record<string, string> = {};
|
||||||
const keywordCount: Record<string, number> = {};
|
const keywordCount: Record<string, number> = {};
|
||||||
|
|
||||||
@@ -275,7 +304,7 @@ function buildEmojiIndex(): EmojiIndex {
|
|||||||
emojis,
|
emojis,
|
||||||
skinToneSpriteCount,
|
skinToneSpriteCount,
|
||||||
baseSpriteCount,
|
baseSpriteCount,
|
||||||
emojiSurrogateRegex: new RegExp(`(${surrogateAlternation})`, 'g'),
|
emojiSurrogateRegex: new RegExp(`(${REGIONAL_INDICATOR_PAIR_PATTERN}|${surrogateAlternation})`, 'g'),
|
||||||
emojiShortcutRegex: new RegExp(`^(${shortcutAlternation})`),
|
emojiShortcutRegex: new RegExp(`^(${shortcutAlternation})`),
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -288,6 +317,50 @@ function getEmojiIndex(): EmojiIndex {
|
|||||||
const lookupSurrogateName = (surrogate: string): string | null =>
|
const lookupSurrogateName = (surrogate: string): string | null =>
|
||||||
getEmojiIndex().canonicalSurrogateToName[toCanonicalSurrogate(surrogate)] ?? null;
|
getEmojiIndex().canonicalSurrogateToName[toCanonicalSurrogate(surrogate)] ?? null;
|
||||||
|
|
||||||
|
const isRegionalIndicatorAt = (text: string, index: number): boolean => {
|
||||||
|
const lowSurrogate = text.charCodeAt(index + 1);
|
||||||
|
return text.charCodeAt(index) === 0xd83c && lowSurrogate >= 0xdde6 && lowSurrogate <= 0xddff;
|
||||||
|
};
|
||||||
|
|
||||||
|
const isInsideRegionalIndicatorPair = (text: string, index: number): boolean => {
|
||||||
|
if (!isRegionalIndicatorAt(text, index)) return false;
|
||||||
|
let runStart = index;
|
||||||
|
while (isRegionalIndicatorAt(text, runStart - 2)) runStart -= 2;
|
||||||
|
return (index - runStart) % 4 === 2;
|
||||||
|
};
|
||||||
|
|
||||||
|
const toEmojiSurrogateMatch = (text: string, start: number, end: number): EmojiSurrogateMatch => ({
|
||||||
|
start,
|
||||||
|
end,
|
||||||
|
name: lookupSurrogateName(text.slice(start, end)),
|
||||||
|
});
|
||||||
|
|
||||||
|
function* matchEmojiSurrogates(text: string, startIndex = 0): Generator<EmojiSurrogateMatch, void> {
|
||||||
|
const regex = getEmojiIndex().emojiSurrogateRegex;
|
||||||
|
let index = startIndex;
|
||||||
|
if (isInsideRegionalIndicatorPair(text, index)) {
|
||||||
|
yield toEmojiSurrogateMatch(text, index, index + 2);
|
||||||
|
index += 2;
|
||||||
|
}
|
||||||
|
while (true) {
|
||||||
|
regex.lastIndex = index;
|
||||||
|
const match = regex.exec(text);
|
||||||
|
if (match == null) return;
|
||||||
|
index = match.index + match[0].length;
|
||||||
|
const emojiMatch = toEmojiSurrogateMatch(text, match.index, index);
|
||||||
|
const isUnknownRegionalIndicatorPair =
|
||||||
|
emojiMatch.name == null &&
|
||||||
|
isRegionalIndicatorAt(text, match.index) &&
|
||||||
|
isRegionalIndicatorAt(text, match.index + 2);
|
||||||
|
if (isUnknownRegionalIndicatorPair) {
|
||||||
|
yield toEmojiSurrogateMatch(text, match.index, match.index + 2);
|
||||||
|
yield toEmojiSurrogateMatch(text, match.index + 2, index);
|
||||||
|
} else {
|
||||||
|
yield emojiMatch;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
const EMOJI_SHORTCODE_RE = /^:([^\s:]+(?:::skin-tone-[0-9])?):/;
|
const EMOJI_SHORTCODE_RE = /^:([^\s:]+(?:::skin-tone-[0-9])?):/;
|
||||||
const categoryIcons = {
|
const categoryIcons = {
|
||||||
people: SmileyIcon,
|
people: SmileyIcon,
|
||||||
@@ -377,6 +450,8 @@ export default {
|
|||||||
getSurrogateName: (surrogate: string): string | null => {
|
getSurrogateName: (surrogate: string): string | null => {
|
||||||
return lookupSurrogateName(surrogate);
|
return lookupSurrogateName(surrogate);
|
||||||
},
|
},
|
||||||
|
matchEmojiSurrogates,
|
||||||
|
isInsideRegionalIndicatorPair,
|
||||||
findEmojiByName: (emojiName: string): UnicodeEmoji | null => {
|
findEmojiByName: (emojiName: string): UnicodeEmoji | null => {
|
||||||
return getEmojiIndex().nameToEmoji[emojiName] || null;
|
return getEmojiIndex().nameToEmoji[emojiName] || null;
|
||||||
},
|
},
|
||||||
@@ -400,8 +475,5 @@ export default {
|
|||||||
get EMOTICON_PREFIX_RE(): RegExp {
|
get EMOTICON_PREFIX_RE(): RegExp {
|
||||||
return getEmojiIndex().emojiShortcutRegex;
|
return getEmojiIndex().emojiShortcutRegex;
|
||||||
},
|
},
|
||||||
get EMOJI_SURROGATE_RE(): RegExp {
|
|
||||||
return getEmojiIndex().emojiSurrogateRegex;
|
|
||||||
},
|
|
||||||
EMOJI_SPRITES,
|
EMOJI_SPRITES,
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -29,17 +29,19 @@ export function registerComposerEmojiShortcode(editor: LexicalEditor, resolve: C
|
|||||||
}
|
}
|
||||||
|
|
||||||
function findUnicodeEmoji(text: string, startIndex: number): TypedEmojiMatch | null {
|
function findUnicodeEmoji(text: string, startIndex: number): TypedEmojiMatch | null {
|
||||||
const pattern = new RegExp(UnicodeEmojis.EMOJI_SURROGATE_RE.source, 'g');
|
const matches = UnicodeEmojis.matchEmojiSurrogates(text, startIndex);
|
||||||
pattern.lastIndex = startIndex;
|
let match = matches.next().value;
|
||||||
const match = pattern.exec(text);
|
if (match && UnicodeEmojis.isInsideRegionalIndicatorPair(text, match.end)) {
|
||||||
if (match == null) {
|
match = matches.next().value;
|
||||||
|
}
|
||||||
|
if (!match) {
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
const name = UnicodeEmojis.nameForSurrogate(match[0], false);
|
const name = match.name;
|
||||||
if (!name) {
|
if (!name) {
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
return {start: match.index, end: match.index + match[0].length, name};
|
return {start: match.start, end: match.end, name};
|
||||||
}
|
}
|
||||||
|
|
||||||
interface EmojiToken extends TypedEmojiMatch {
|
interface EmojiToken extends TypedEmojiMatch {
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ import {type EmojiProvider, setEmojiParserConfig} from '@app/features/messaging/
|
|||||||
import {SKIN_TONE_SURROGATES} from '@fluxer/constants/src/EmojiConstants';
|
import {SKIN_TONE_SURROGATES} from '@fluxer/constants/src/EmojiConstants';
|
||||||
|
|
||||||
const emojiProvider: EmojiProvider = {
|
const emojiProvider: EmojiProvider = {
|
||||||
getSurrogateName: UnicodeEmojis.getSurrogateName,
|
matchEmojiSurrogates: UnicodeEmojis.matchEmojiSurrogates,
|
||||||
findEmojiByName: UnicodeEmojis.findEmojiByShortcodeName,
|
findEmojiByName: UnicodeEmojis.findEmojiByShortcodeName,
|
||||||
findEmojiWithSkinTone: UnicodeEmojis.findEmojiWithSkinTone,
|
findEmojiWithSkinTone: UnicodeEmojis.findEmojiWithSkinTone,
|
||||||
};
|
};
|
||||||
@@ -13,9 +13,6 @@ const emojiProvider: EmojiProvider = {
|
|||||||
export function initializeEmojiParser(): void {
|
export function initializeEmojiParser(): void {
|
||||||
setEmojiParserConfig({
|
setEmojiParserConfig({
|
||||||
emojiProvider,
|
emojiProvider,
|
||||||
get emojiRegex() {
|
|
||||||
return UnicodeEmojis.EMOJI_SURROGATE_RE;
|
|
||||||
},
|
|
||||||
skinToneSurrogates: SKIN_TONE_SURROGATES,
|
skinToneSurrogates: SKIN_TONE_SURROGATES,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,17 +4,21 @@ export interface UnicodeEmoji {
|
|||||||
surrogates: string;
|
surrogates: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export interface EmojiSurrogateMatch {
|
||||||
|
start: number;
|
||||||
|
end: number;
|
||||||
|
name: string | null;
|
||||||
|
}
|
||||||
|
|
||||||
export interface EmojiProvider {
|
export interface EmojiProvider {
|
||||||
getSurrogateName(surrogate: string): string | null;
|
matchEmojiSurrogates(text: string): Iterable<EmojiSurrogateMatch>;
|
||||||
findEmojiByName(name: string): UnicodeEmoji | null;
|
findEmojiByName(name: string): UnicodeEmoji | null;
|
||||||
findEmojiWithSkinTone(baseName: string, skinToneSurrogate: string): UnicodeEmoji | null;
|
findEmojiWithSkinTone(baseName: string, skinToneSurrogate: string): UnicodeEmoji | null;
|
||||||
}
|
}
|
||||||
|
|
||||||
export interface EmojiParserConfig {
|
export interface EmojiParserConfig {
|
||||||
emojiProvider?: EmojiProvider;
|
emojiProvider?: EmojiProvider;
|
||||||
emojiRegex?: RegExp;
|
|
||||||
skinToneSurrogates?: ReadonlyArray<string>;
|
skinToneSurrogates?: ReadonlyArray<string>;
|
||||||
convertToCodePoints?: (emoji: string) => string;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
let globalEmojiConfig: EmojiParserConfig | null = null;
|
let globalEmojiConfig: EmojiParserConfig | null = null;
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
// SPDX-License-Identifier: AGPL-3.0-or-later
|
// SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
|
|
||||||
|
import {convertToCodePoints} from '@app/features/expressions/utils/EmojiCodepointUtils';
|
||||||
import {flattenAST} from '@app/features/messaging/utils/markdown/parser/AstUtils';
|
import {flattenAST} from '@app/features/messaging/utils/markdown/parser/AstUtils';
|
||||||
import {getEmojiParserConfig} from '@app/features/messaging/utils/markdown/parser/EmojiParsers';
|
import {getEmojiParserConfig} from '@app/features/messaging/utils/markdown/parser/EmojiParsers';
|
||||||
import {MARKDOWN_PARSER_WASM_BASE64} from '@app/features/messaging/utils/markdown/parser/MarkdownParserWasmBytes';
|
import {MARKDOWN_PARSER_WASM_BASE64} from '@app/features/messaging/utils/markdown/parser/MarkdownParserWasmBytes';
|
||||||
@@ -202,14 +203,6 @@ class Utf8OffsetTracker {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
function defaultCodepoints(emoji: string): string {
|
|
||||||
const containsZwJ = emoji.includes('');
|
|
||||||
const processed = containsZwJ ? emoji : emoji.replace(/️/g, '');
|
|
||||||
return Array.from(processed)
|
|
||||||
.map((char) => char.codePointAt(0)?.toString(16).replace(/^0+/, '') || '')
|
|
||||||
.join('-');
|
|
||||||
}
|
|
||||||
|
|
||||||
const PLAINTEXT_SYMBOLS = new Set(['™', '™️', '©', '©️', '®', '®️']);
|
const PLAINTEXT_SYMBOLS = new Set(['™', '™️', '©', '©️', '®', '®️']);
|
||||||
const SPECIAL_SHORTCODES: Record<string, string> = {
|
const SPECIAL_SHORTCODES: Record<string, string> = {
|
||||||
tm: '™',
|
tm: '™',
|
||||||
@@ -226,29 +219,23 @@ function buildEmojiContext(input: string): string {
|
|||||||
const provider = config?.emojiProvider;
|
const provider = config?.emojiProvider;
|
||||||
let context = '';
|
let context = '';
|
||||||
if (!provider) return context;
|
if (!provider) return context;
|
||||||
const convertToCodePoints = config.convertToCodePoints || defaultCodepoints;
|
const offsetTracker = new Utf8OffsetTracker();
|
||||||
const emojiRegex = config.emojiRegex;
|
for (const match of provider.matchEmojiSurrogates(input)) {
|
||||||
if (emojiRegex) {
|
const candidate = input.slice(match.start, match.end);
|
||||||
const offsetTracker = new Utf8OffsetTracker();
|
if (PLAINTEXT_SYMBOLS.has(candidate)) continue;
|
||||||
emojiRegex.lastIndex = 0;
|
const name = match.name;
|
||||||
let match: RegExpExecArray | null;
|
if (!name) continue;
|
||||||
while ((match = emojiRegex.exec(input)) !== null) {
|
const candidateBytes = textEncoder.encode(candidate).byteLength;
|
||||||
const candidate = match[0];
|
const byteOffset = offsetTracker.offsetFor(input, match.start);
|
||||||
if (!candidate || PLAINTEXT_SYMBOLS.has(candidate)) continue;
|
context += appendContextLine([
|
||||||
const name = provider.getSurrogateName(candidate);
|
'S',
|
||||||
if (!name) continue;
|
String(byteOffset),
|
||||||
const candidateBytes = textEncoder.encode(candidate).byteLength;
|
String(candidateBytes),
|
||||||
const byteOffset = offsetTracker.offsetFor(input, match.index);
|
candidate,
|
||||||
context += appendContextLine([
|
name,
|
||||||
'S',
|
convertToCodePoints(candidate),
|
||||||
String(byteOffset),
|
]);
|
||||||
String(candidateBytes),
|
offsetTracker.advance(candidate.length, candidateBytes);
|
||||||
candidate,
|
|
||||||
name,
|
|
||||||
convertToCodePoints(candidate),
|
|
||||||
]);
|
|
||||||
offsetTracker.advance(candidate.length, candidateBytes);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
const shortcodeRegex = /:([\p{L}\p{N}_-]+):/gu;
|
const shortcodeRegex = /:([\p{L}\p{N}_-]+):/gu;
|
||||||
const seen = new Set<string>();
|
const seen = new Set<string>();
|
||||||
|
|||||||
|
Before Width: | Height: | Size: 270 KiB After Width: | Height: | Size: 279 KiB |
|
Before Width: | Height: | Size: 592 KiB After Width: | Height: | Size: 612 KiB |
|
Before Width: | Height: | Size: 268 KiB After Width: | Height: | Size: 277 KiB |
|
Before Width: | Height: | Size: 589 KiB After Width: | Height: | Size: 608 KiB |
|
Before Width: | Height: | Size: 274 KiB After Width: | Height: | Size: 283 KiB |
|
Before Width: | Height: | Size: 602 KiB After Width: | Height: | Size: 621 KiB |
|
Before Width: | Height: | Size: 269 KiB After Width: | Height: | Size: 277 KiB |
|
Before Width: | Height: | Size: 590 KiB After Width: | Height: | Size: 608 KiB |
|
Before Width: | Height: | Size: 267 KiB After Width: | Height: | Size: 275 KiB |
|
Before Width: | Height: | Size: 585 KiB After Width: | Height: | Size: 603 KiB |
|
Before Width: | Height: | Size: 1.4 MiB After Width: | Height: | Size: 1.4 MiB |
|
Before Width: | Height: | Size: 3.1 MiB After Width: | Height: | Size: 3.2 MiB |
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
use crate::ast::{EmojiKind, Node, ParserFlags, ParserResult};
|
use crate::ast::{EmojiKind, Node, ParserFlags, ParserResult};
|
||||||
use crate::constants::{CODE_FENCE_LENGTH, MAX_INLINE_DEPTH, MAX_LINE_LENGTH};
|
use crate::constants::{CODE_FENCE_LENGTH, MAX_INLINE_DEPTH, MAX_LINE_LENGTH};
|
||||||
|
use crate::emoji::{EmojiContext, StandardEmoji};
|
||||||
use crate::links;
|
use crate::links;
|
||||||
use crate::normalize::{
|
use crate::normalize::{
|
||||||
combine_adjacent_text, compact_empty_text_nodes, flatten_top_level_formatting,
|
combine_adjacent_text, compact_empty_text_nodes, flatten_top_level_formatting,
|
||||||
@@ -168,7 +169,13 @@ fn parse_inline_with_context(
|
|||||||
|
|
||||||
if let Some(result) = parse_regional_indicator_flag(remaining) {
|
if let Some(result) = parse_regional_indicator_flag(remaining) {
|
||||||
flush_accumulated_text(&mut nodes, &mut accumulated);
|
flush_accumulated_text(&mut nodes, &mut accumulated);
|
||||||
nodes.push(result.node);
|
push_regional_indicator_pair(
|
||||||
|
&mut nodes,
|
||||||
|
parser.emoji_context(),
|
||||||
|
base_offset + position,
|
||||||
|
&remaining[..result.advance],
|
||||||
|
result.node,
|
||||||
|
);
|
||||||
position += result.advance;
|
position += result.advance;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -349,6 +356,49 @@ fn parse_regional_indicator_flag(text: &str) -> Option<ParserResult> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn push_regional_indicator_pair(
|
||||||
|
nodes: &mut Vec<Node>,
|
||||||
|
emoji_context: &EmojiContext,
|
||||||
|
offset: usize,
|
||||||
|
pair: &str,
|
||||||
|
flag: Node,
|
||||||
|
) {
|
||||||
|
let (first, second) = pair.split_at(pair.len() / 2);
|
||||||
|
match emoji_context.standard_at(offset) {
|
||||||
|
Some(emoji) if emoji.raw == pair => nodes.push(standard_emoji_node(emoji)),
|
||||||
|
Some(emoji) if emoji.raw == first => {
|
||||||
|
nodes.push(standard_emoji_node(emoji));
|
||||||
|
match emoji_context.standard_at(offset + first.len()) {
|
||||||
|
Some(emoji) if emoji.raw == second => nodes.push(standard_emoji_node(emoji)),
|
||||||
|
_ => nodes.extend(regional_indicator_emoji(second)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => nodes.push(flag),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn standard_emoji_node(emoji: &StandardEmoji) -> Node {
|
||||||
|
Node::Emoji {
|
||||||
|
kind: EmojiKind::Standard {
|
||||||
|
raw: emoji.raw.clone(),
|
||||||
|
codepoints: emoji.codepoints.clone(),
|
||||||
|
name: emoji.name.clone(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn regional_indicator_emoji(text: &str) -> Option<Node> {
|
||||||
|
let ch = text.chars().next()?;
|
||||||
|
let letter = regional_indicator_letter(ch)?;
|
||||||
|
Some(Node::Emoji {
|
||||||
|
kind: EmojiKind::Standard {
|
||||||
|
raw: ch.to_string(),
|
||||||
|
codepoints: format!("{:x}", ch as u32),
|
||||||
|
name: format!("regional_indicator_{letter}"),
|
||||||
|
},
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
fn regional_indicator_letter(ch: char) -> Option<char> {
|
fn regional_indicator_letter(ch: char) -> Option<char> {
|
||||||
let codepoint = ch as u32;
|
let codepoint = ch as u32;
|
||||||
if !(0x1f1e6..=0x1f1ff).contains(&codepoint) {
|
if !(0x1f1e6..=0x1f1ff).contains(&codepoint) {
|
||||||
|
|||||||
@@ -151,6 +151,67 @@ fn native_parser_fixtures_cover_typescript_suite_surface() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn native_parser_pairs_regional_indicators_with_emoji_context() {
|
||||||
|
assert_eq!(
|
||||||
|
parse(
|
||||||
|
"🇦🇧🇸🇪",
|
||||||
|
0,
|
||||||
|
"S\t0\t4\t🇦\tregional_indicator_a\t1f1e6\nS\t4\t4\t🇧\tregional_indicator_b\t1f1e7\nS\t8\t8\t🇸🇪\tflag_se\t1f1f8-1f1ea\n"
|
||||||
|
),
|
||||||
|
json!({"nodes":[
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇧","codepoints":"1f1e7","name":"regional_indicator_b"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇸🇪","codepoints":"1f1f8-1f1ea","name":"flag_se"}}
|
||||||
|
]})
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
parse(
|
||||||
|
"🇦🇧🇦🇸",
|
||||||
|
0,
|
||||||
|
"S\t0\t4\t🇦\tregional_indicator_a\t1f1e6\nS\t4\t4\t🇧\tregional_indicator_b\t1f1e7\nS\t8\t8\t🇦🇸\tflag_as\t1f1e6-1f1f8\n"
|
||||||
|
),
|
||||||
|
json!({"nodes":[
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇧","codepoints":"1f1e7","name":"regional_indicator_b"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦🇸","codepoints":"1f1e6-1f1f8","name":"flag_as"}}
|
||||||
|
]})
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
parse("Sark 🇨🇶", 0, "S\t5\t8\t🇨🇶\tflag_sark\t1f1e8-1f1f6\n"),
|
||||||
|
json!({"nodes":[
|
||||||
|
{"type":"Text","content":"Sark "},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇨🇶","codepoints":"1f1e8-1f1f6","name":"flag_sark"}}
|
||||||
|
]})
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
parse("x 🇦", 0, "S\t2\t4\t🇦\tregional_indicator_a\t1f1e6\n"),
|
||||||
|
json!({"nodes":[
|
||||||
|
{"type":"Text","content":"x "},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}}
|
||||||
|
]})
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
parse(
|
||||||
|
"🇦🇧🇸🇪",
|
||||||
|
0,
|
||||||
|
"S\t0\t4\t🇦\tregional_indicator_a\t1f1e6\nS\t4\t8\t🇧🇸\tflag_bs\t1f1e7-1f1f8\nS\t12\t4\t🇪\tregional_indicator_e\t1f1ea\n"
|
||||||
|
),
|
||||||
|
json!({"nodes":[
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦","codepoints":"1f1e6","name":"regional_indicator_a"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇧","codepoints":"1f1e7","name":"regional_indicator_b"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇸🇪","codepoints":"1f1f8-1f1ea","name":"flag_se"}}
|
||||||
|
]})
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
parse("🇦🇧🇸🇪", 0, ""),
|
||||||
|
json!({"nodes":[
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇦🇧","codepoints":"1f1e6-1f1e7","name":"flag_ab"}},
|
||||||
|
{"type":"Emoji","kind":{"kind":"Standard","raw":"🇸🇪","codepoints":"1f1f8-1f1ea","name":"flag_se"}}
|
||||||
|
]})
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn native_parser_allows_apostrophe_in_masked_link_destination() {
|
fn native_parser_allows_apostrophe_in_masked_link_destination() {
|
||||||
let url = "https://docs.example.test/help/faq/#why-can't-this-link-parse%3F";
|
let url = "https://docs.example.test/help/faq/#why-can't-this-link-parse%3F";
|
||||||
|
|||||||