122 lines
4.7 KiB
JavaScript
122 lines
4.7 KiB
JavaScript
import { createHash } from "node:crypto";
|
|
import { writeFile } from "node:fs/promises";
|
|
import { fileURLToPath } from "node:url";
|
|
|
|
const UNICODE_RELEASE = "17.0.0";
|
|
const sources = {
|
|
derived: `https://www.unicode.org/Public/${UNICODE_RELEASE}/ucd/DerivedCoreProperties.txt`,
|
|
emoji: `https://www.unicode.org/Public/${UNICODE_RELEASE}/emoji/emoji-test.txt`,
|
|
unicodeData: `https://www.unicode.org/Public/${UNICODE_RELEASE}/ucd/UnicodeData.txt`,
|
|
};
|
|
|
|
const fetchText = async (url) => {
|
|
const response = await fetch(url);
|
|
if (!response.ok) throw new Error(`Failed to fetch ${url}: ${response.status}`);
|
|
return response.text();
|
|
};
|
|
|
|
const parseRange = (value) => {
|
|
const [from, to = from] = value.split("..").map((part) => Number.parseInt(part, 16));
|
|
return [from, to];
|
|
};
|
|
|
|
const mergeRanges = (ranges) => {
|
|
const merged = [];
|
|
for (const [from, to] of ranges.sort((left, right) => left[0] - right[0])) {
|
|
const previous = merged.at(-1);
|
|
if (previous && from <= previous[1] + 1) previous[1] = Math.max(previous[1], to);
|
|
else merged.push([from, to]);
|
|
}
|
|
return merged;
|
|
};
|
|
|
|
const parseDerivedProperty = (source, property) =>
|
|
mergeRanges(
|
|
source
|
|
.split("\n")
|
|
.map((line) => line.match(/^([0-9A-F]+(?:\.\.[0-9A-F]+)?)\s*;\s*([A-Za-z_]+)/))
|
|
.filter((match) => match?.[2] === property)
|
|
.map((match) => parseRange(match[1])),
|
|
);
|
|
|
|
const parseCombiningMarkRanges = (source) => {
|
|
const ranges = [];
|
|
let pending;
|
|
for (const line of source.split("\n")) {
|
|
if (!line) continue;
|
|
const fields = line.split(";");
|
|
const codePoint = Number.parseInt(fields[0], 16);
|
|
const name = fields[1];
|
|
const category = fields[2];
|
|
if (name.endsWith(", First>")) {
|
|
pending = { codePoint, category };
|
|
} else if (name.endsWith(", Last>") && pending) {
|
|
if (pending.category === "Mn" || pending.category === "Mc") ranges.push([pending.codePoint, codePoint]);
|
|
pending = undefined;
|
|
} else if (category === "Mn" || category === "Mc") {
|
|
ranges.push([codePoint, codePoint]);
|
|
}
|
|
}
|
|
return mergeRanges(ranges);
|
|
};
|
|
|
|
const parseEmoji = (source) =>
|
|
source
|
|
.split("\n")
|
|
.map((line) => line.match(/^([0-9A-F ]+)\s*;\s*fully-qualified\b/))
|
|
.filter(Boolean)
|
|
.map((match) =>
|
|
String.fromCodePoint(
|
|
...match[1]
|
|
.trim()
|
|
.split(/\s+/)
|
|
.map((part) => Number.parseInt(part, 16)),
|
|
),
|
|
);
|
|
|
|
const formatRanges = (name, ranges) => {
|
|
const values = ranges.flatMap(([from, to]) => [`0x${from.toString(16)}`, `0x${to.toString(16)}`]);
|
|
const lines = [];
|
|
for (let index = 0; index < values.length; index += 12) lines.push(` ${values.slice(index, index + 12).join(", ")},`);
|
|
return `export const ${name}: readonly number[] = [\n${lines.join("\n")}\n];`;
|
|
};
|
|
|
|
const formatEmoji = (emoji) => {
|
|
const lines = [];
|
|
for (let index = 0; index < emoji.length; index += 12) {
|
|
lines.push(
|
|
` ${emoji
|
|
.slice(index, index + 12)
|
|
.map((value) => JSON.stringify(value))
|
|
.join(", ")},`,
|
|
);
|
|
}
|
|
return `export const FULLY_QUALIFIED_EMOJI: readonly string[] = [\n${lines.join("\n")}\n];`;
|
|
};
|
|
|
|
const unicodeDataDigest = (xidContinue, defaultIgnorable, combiningMarks, emoji) => {
|
|
const canonical = [];
|
|
const addRanges = (name, ranges) => {
|
|
for (const [from, to] of ranges) canonical.push(`${name}:${from.toString(16)}-${to.toString(16)}\n`);
|
|
};
|
|
addRanges("xid", xidContinue);
|
|
addRanges("ignorable", defaultIgnorable);
|
|
addRanges("combining", combiningMarks);
|
|
canonical.push(
|
|
...emoji.map((value) => `emoji:${[...value].map((character) => character.codePointAt(0).toString(16)).join(",")}\n`).sort(),
|
|
);
|
|
return createHash("sha256").update(canonical.join("")).digest("hex");
|
|
};
|
|
|
|
const [derived, emojiTest, unicodeData] = await Promise.all(Object.values(sources).map(fetchText));
|
|
const xidContinue = parseDerivedProperty(derived, "XID_Continue");
|
|
const defaultIgnorable = parseDerivedProperty(derived, "Default_Ignorable_Code_Point");
|
|
const combiningMarks = parseCombiningMarkRanges(unicodeData);
|
|
const emoji = parseEmoji(emojiTest);
|
|
const output = `// Generated by scripts/generate-tag-unicode-data.mjs. Do not edit.\n// Unicode and Emoji ${UNICODE_RELEASE} data used by Memos tag recognition.\n// Unicode data SHA-256: ${unicodeDataDigest(xidContinue, defaultIgnorable, combiningMarks, emoji)}\n// biome-ignore-all format: generated data is kept in compact source order\n\n${formatRanges(
|
|
"XID_CONTINUE_RANGES",
|
|
xidContinue,
|
|
)}\n\n${formatRanges("DEFAULT_IGNORABLE_RANGES", defaultIgnorable)}\n\n${formatRanges("COMBINING_MARK_RANGES", combiningMarks)}\n\n${formatEmoji(emoji)}\n`;
|
|
|
|
const outputURL = new URL("../src/utils/tag-unicode-data.ts", import.meta.url);
|
|
await writeFile(fileURLToPath(outputURL), output);
|