From d617f652b01086b9619500a46dc1eb9b13e84046 Mon Sep 17 00:00:00 2001 From: Johnny Date: Sun, 2 Aug 2026 21:08:57 +0800 Subject: [PATCH] fix(markdown): support word-internal apostrophes in tags (#6135) --- docs/adr/0001-tag-syntax-and-recognition.md | 74 +++++++++++++++++---- internal/markdown/markdown_test.go | 10 +++ internal/markdown/parser/tag.go | 25 +++++++ internal/markdown/parser/tag_test.go | 12 ++++ web/src/utils/tag-grammar.ts | 25 +++++++ web/tests/editor-decorations.test.ts | 11 +++ web/tests/editor-tag-autocomplete.test.ts | 10 +++ web/tests/remark-tag.test.tsx | 10 +++ web/tests/tag-grammar.test.ts | 8 +++ 9 files changed, 173 insertions(+), 12 deletions(-) diff --git a/docs/adr/0001-tag-syntax-and-recognition.md b/docs/adr/0001-tag-syntax-and-recognition.md index 41c51afe..792489a0 100644 --- a/docs/adr/0001-tag-syntax-and-recognition.md +++ b/docs/adr/0001-tag-syntax-and-recognition.md @@ -45,6 +45,7 @@ changing metadata or a separate tag record. Renaming `#Work` to `#work` across 1 ## Decision drivers - Work naturally for multilingual personal notes, including numeric tags and emoji. +- Preserve common word-internal apostrophes without absorbing surrounding quotation punctuation into tag values. - Recognize `hello#tag` intentionally without mistaking URL fragments for tags. - Give the backend, renderer, editor decoration, and completion the same values and source spans. - Use Markdown syntax rather than ad hoc URL or code regular expressions. @@ -100,6 +101,11 @@ changing metadata or a separate tag record. Renaming `#Work` to `#work` across 1 **Tag segment** : A non-empty component of a hierarchical tag identifier. Slashes separate segments and are not part of any segment. +**Apostrophe joiner** +: U+0027 APOSTROPHE (`'`) or U+2019 RIGHT SINGLE QUOTATION MARK (`’`) emitted inside a tag segment only when, after emoji-first tokenization, the + immediately preceding source code point emits as an `XID_Continue` code point and the immediately following source code point emits as a non-combining + `XID_Continue` code point. + **Display value** : The direct or implied value presented as a derived tag label. It preserves emitted code points exactly but does not contain ignored default-ignorable code points or ignored leading combining marks. @@ -143,7 +149,7 @@ Introducer := U+0023 NUMBER SIGN ("#") outside a FullyQualifiedEmoji TagSourceSpelling := TagSegmentSpelling ("/" TagSegmentSpelling)* TagSegmentSpelling := IgnoredPrefix* SegmentStarter SegmentContinuation* IgnoredPrefix := IgnoredDefaultIgnorable | IgnoredLeadingCombiningMark -SegmentContinuation := ValueUnit | IgnoredDefaultIgnorable +SegmentContinuation := ValueUnit | ApostropheJoiner | IgnoredDefaultIgnorable SegmentStarter := EmittedXIDContinueCodePointExceptCombiningMark | FullyQualifiedEmoji @@ -157,6 +163,9 @@ ValueUnit := EmittedXIDContinueCodePoint | "+" | "&" +ApostropheJoiner := U+0027 APOSTROPHE ("'") + | U+2019 RIGHT SINGLE QUOTATION MARK ("’") + IgnoredDefaultIgnorable := Default_Ignorable_Code_Point outside a FullyQualifiedEmoji IgnoredLeadingCombiningMark := XID_Continue with General_Category Mn or Mc, minus Default_Ignorable_Code_Point, before SegmentStarter @@ -164,13 +173,14 @@ IgnoredLeadingCombiningMark := XID_Continue with General_Category Mn or Mc, The grammar recognizes source spelling and emits a tag identifier as follows: -- Each `SegmentStarter` and `ValueUnit` emits its exact source code-point sequence. A `FullyQualifiedEmoji` therefore emits its complete matched sequence. +- Each `SegmentStarter`, `ValueUnit`, and contextually valid `ApostropheJoiner` emits its exact source code-point sequence. A `FullyQualifiedEmoji` + therefore emits its complete matched sequence. - Each consumed `/` emits one U+002F SOLIDUS into the identifier. - `Introducer`, `IgnoredDefaultIgnorable`, and `IgnoredLeadingCombiningMark` emit nothing. At every source position, the token priority is the longest `FullyQualifiedEmoji`, then `IgnoredDefaultIgnorable`, then `IgnoredLeadingCombiningMark`, then -an emitted code point or Memos extension unit. This preserves default-ignorable code points inside matched fully-qualified emoji sequences while omitting -them everywhere else. +a contextually valid `ApostropheJoiner`, then an emitted code point or Memos extension unit. This preserves default-ignorable code points inside matched +fully-qualified emoji sequences while omitting them everywhere else. The emitted tag identifier is the direct tag value. @@ -199,12 +209,20 @@ Rules: 12. A non-default-ignorable `XID_Continue` code point whose General Category is `Mn` (Nonspacing Mark) or `Mc` (Spacing Combining Mark) is consumed but omitted while it precedes the starter of its segment. The same non-default-ignorable code point is preserved as a value unit after that segment's starter. -13. There is no tag-specific identifier length limit. The enclosing memo-size limit provides the resource bound. +13. After applying the longest-`FullyQualifiedEmoji` token priority, U+0027 APOSTROPHE and U+2019 RIGHT SINGLE QUOTATION MARK are emitted as + `ApostropheJoiner` only when the immediately preceding source code point in the same segment was emitted as an `EmittedXIDContinueCodePoint` and the + immediately following source code point can emit as an + `EmittedXIDContinueCodePointExceptCombiningMark`. An apostrophe joiner therefore cannot start or end a segment, repeat without an intervening XID + code point, adjoin a fully-qualified emoji or Memos extension unit, or join across an ignored code point. +14. U+02BC MODIFIER LETTER APOSTROPHE (`ʼ`) is already an `XID_Continue` code point. It follows the ordinary XID rules rather than the contextual + apostrophe-joiner rule. +15. There is no tag-specific identifier length limit. The enclosing memo-size limit provides the resource bound. The grammar is intentionally broader than a programming-language identifier grammar. It does not require the first emitted unit to be `XID_Start`: digits, `_`, `-`, `+`, `&`, and fully-qualified emoji can start a segment. Default-ignorable code points and non-default-ignorable `Mn` or `Mc` code points may occur before each segment's starter but are omitted from the value. After the starter, those non-default-ignorable `Mn` and `Mc` code points are -preserved. A slash cannot begin a segment, and segments made only from `-`, `+`, and `&` are explicitly valid. +preserved. ASCII and right-curly apostrophes are emitted only as contextual joiners between XID code points. A slash cannot begin a segment, and segments +made only from `-`, `+`, and `&` are explicitly valid. ### Maximal-prefix scanning @@ -215,12 +233,15 @@ After finding an introducer, the lexer consumes the longest valid `TagSourceSpel 2. Emit a segment starter, trying the longest matching `FullyQualifiedEmoji` before any shorter unit. 3. After the starter, preserve non-default-ignorable `XID_Continue` combining marks as ordinary value units; continue consuming default-ignorable code points without emitting them. -4. Consume `/` only when the following source, after any ignored prefix, can emit a segment starter; then consume that segment. -5. Stop before a `/` that is leading, trailing, or followed only by an ignored prefix and then another `/`, a non-starter, or the end of the literal-source +4. Emit U+0027 or U+2019 as an apostrophe joiner only when, after emoji-first tokenization, the immediately preceding source code point emitted an XID + continuation unit and the immediately following source code point can emit a non-combining XID continuation unit. Otherwise stop before the + apostrophe. +5. Consume `/` only when the following source, after any ignored prefix, can emit a segment starter; then consume that segment. +6. Stop before a `/` that is leading, trailing, or followed only by an ignored prefix and then another `/`, a non-starter, or the end of the literal-source run, leaving that slash and the remaining source unconsumed. -6. Otherwise stop before the first code point that cannot begin a valid continuation unit. -7. Keep the valid prefix already consumed; a later invalid character does not invalidate it. -8. Produce no candidate if the first `TagSegmentSpelling` cannot match. +7. Otherwise stop before the first code point that cannot begin a valid continuation unit. +8. Keep the valid prefix already consumed; a later invalid character does not invalidate it. +9. Produce no candidate if the first `TagSegmentSpelling` cannot match. ### Candidate enumeration @@ -255,6 +276,13 @@ Examples: | `#price€` | `price` | `€` | | `#C++` | `C++` | empty | | `#R&D` | `R&D` | empty | +| `#tag's` | `tag's` | empty | +| `#сім'я` | `сім'я` | empty | +| `#O’Brien` | `O’Brien` | empty | +| `#café's` | `café's` | empty | +| `#users'` | `users` | `'` | +| `#foo'1️⃣` | `foo` | `'1️⃣` | +| `#'tag` | none | `'tag` | | `#-foo` | `-foo` | empty | | `#foo-` | `foo-` | empty | | `#---` | `---` | empty | @@ -380,6 +408,8 @@ deduplicating, counting, filtering, navigating, or performing exact metadata loo #café / #café #A / #A #straße / #STRASSE +#O'Brien / #O’Brien +#O’Brien / #OʼBrien ``` Multiple occurrences that emit exactly equal direct tag values in one memo produce one memo-tag membership. A tag metadata rule may deliberately match @@ -432,6 +462,16 @@ The following examples are normative for the lexical and context decisions alrea | `#2026` | `2026` | Numeric-only identifiers are valid | | `#C++` | `C++` | Explicit `+` extension | | `#R&D` | `R&D` | Explicit `&` extension | +| `#tag's` | `tag's` | ASCII apostrophe joins two XID code points | +| `#сім'я` | `сім'я` | ASCII apostrophe preserves a Ukrainian word | +| `#O’Brien` | `O’Brien` | Right single quotation mark joins two XID code points | +| `#café's` | `café's` | An emitted combining mark may precede an apostrophe joiner | +| `#users'` | `users` | A trailing apostrophe is not a joiner | +| `#foo'1️⃣` | `foo` | Emoji-first tokenization prevents an apostrophe from adjoining the keycap sequence | +| `#'tag` | none | An apostrophe cannot start a segment | +| `'#tag'` | `tag` | Surrounding quotation punctuation remains outside the occurrence | +| `#rock’n’roll` | `rock’n’roll` | Multiple apostrophe joiners are valid when each independently satisfies the context rule | +| `#OʼBrien` | `OʼBrien` | U+02BC is an ordinary `XID_Continue` code point | | `#` followed by 101 `a` code points | all 101 `a` code points | There is no tag-specific length limit | | `#-foo` | `-foo` | Visible connector extensions may begin a segment | | `#foo-` | `foo-` | Visible connector extensions may end a segment | @@ -490,7 +530,9 @@ The following examples are normative for the lexical and context decisions alrea - Common multilingual tags, numeric tags, hierarchy characters, and emoji remain expressive. - Hierarchical ancestors have one consistent meaning across API membership, exact filters, navigation, and counts. - `hello#tag` works consistently while actual Markdown links and URLs are excluded structurally. -- Markdown punctuation and delimiters such as comma, backtick, currency, and mathematical operators stop an identifier predictably. +- Common word-internal apostrophes support multilingual words and names while surrounding quotation punctuation remains outside tag values. +- Unsupported Markdown punctuation, delimiters, currency symbols, and operators stop an identifier predictably; apostrophes are the explicitly + constrained exception. - Exact equality preserves every emitted code-point distinction and avoids locale-dependent identity rules. - Ignoring default-ignorable code points outside emoji avoids invisible tag distinctions while preserving matched fully-qualified emoji sequences. - Ignoring leading non-default-ignorable `Mn` and `Mc` code points only before each segment's visible starter avoids invisible-leading segments without @@ -506,6 +548,8 @@ The following examples are normative for the lexical and context decisions alrea - Pinning Unicode data requires maintenance when Unicode and Emoji data are updated. - Visually indistinguishable or canonically equivalent source spellings whose emitted values differ remain separate tags unless the user edits their memo sources to make those values identical. +- Visually similar apostrophe spellings such as U+0027, U+2019, and U+02BC remain distinct under exact tag identity. +- English possessive-looking source such as `#tag's` denotes the complete tag value `tag's`, not `tag` followed by prose. - Source spellings that differ only by ignored default-ignorable code points intentionally collapse to the same emitted tag value. - Source spellings that differ only by ignored leading combining marks intentionally collapse to the same emitted tag value. @@ -554,6 +598,12 @@ tag. The slash and remaining source stay ordinary Markdown. Rejected. Requiring `-`, `+`, or `&` to be medial, non-repeating, or accompanied by a letter would add validation rules without resolving a structural ambiguity. They remain ordinary segment units; only `/` has structural meaning. +### Treat apostrophes as unrestricted value units + +Rejected. Allowing apostrophes as starters, trailing units, or repeatable ordinary units would absorb surrounding quotation punctuation into values such +as `tag'` and make quoted source such as `'#tag'` ambiguous. The contextual joiner rule supports words and names while leaving punctuation at tag +boundaries unconsumed. + ### Recognize compatibility number signs Rejected. U+FE5F SMALL NUMBER SIGN and U+FF03 FULLWIDTH NUMBER SIGN are visually similar to `#` but are not Markdown syntax. Recognizing only ASCII U+0023 diff --git a/internal/markdown/markdown_test.go b/internal/markdown/markdown_test.go index 11e45eb0..6b742d5e 100644 --- a/internal/markdown/markdown_test.go +++ b/internal/markdown/markdown_test.go @@ -576,6 +576,16 @@ func TestExtractTagsMemosTagV1(t *testing.T) { {name: "number sign keycap continuation", content: "#first#️⃣", expected: []string{"first#️⃣"}}, {name: "exact identity", content: "#Work #work", expected: []string{"Work", "work"}}, {name: "no normalization", content: "#café #cafe\u0301", expected: []string{"café", "cafe\u0301"}}, + { + name: "word internal apostrophes", + content: "#tag's #сім'я #O'Brien #O’Brien #OʼBrien #cafe\u0301's", + expected: []string{"tag's", "сім'я", "O'Brien", "O’Brien", "OʼBrien", "cafe\u0301's"}, + }, + { + name: "apostrophe boundaries", + content: "'#tag' #users' #'missing #rock''roll #O‘Brien #A\u200d'B", + expected: []string{"tag", "users", "rock", "O", "A"}, + }, {name: "ignored spellings deduplicate", content: "#AB #A\u200dB #\u0301AB", expected: []string{"AB"}}, {name: "hierarchy expansion", content: "#book/fiction/history", expected: []string{"book", "book/fiction", "book/fiction/history"}}, {name: "hierarchy exact dedupe", content: "#book/fiction #book", expected: []string{"book", "book/fiction"}}, diff --git a/internal/markdown/parser/tag.go b/internal/markdown/parser/tag.go index 485801c5..f25bbce9 100644 --- a/internal/markdown/parser/tag.go +++ b/internal/markdown/parser/tag.go @@ -60,6 +60,21 @@ func isSegmentContinuation(r rune) bool { return r == '-' || r == '+' || r == '&' || isXIDContinue(r) && !isDefaultIgnorable(r) } +func isApostropheJoiner(r rune) bool { + return r == '\'' || r == '\u2019' +} + +func startsNonCombiningXIDContinuation(source []byte) bool { + if len(source) == 0 || matchFullyQualifiedEmoji(source) > 0 { + return false + } + r, size := utf8.DecodeRune(source) + if r == utf8.RuneError && size == 1 { + return false + } + return isXIDContinue(r) && !isCombiningMark(r) && !isDefaultIgnorable(r) +} + func matchFullyQualifiedEmoji(source []byte) int { limit := min(len(source), maxEmoji17SequenceBytes) match := 0 @@ -82,6 +97,7 @@ func matchFullyQualifiedEmoji(source []byte) int { func scanTagSegment(source []byte) (int, []byte, bool) { pos := 0 var value []byte + previousWasXIDContinuation := false starter: for pos < len(source) { @@ -91,6 +107,7 @@ starter: if emojiLength := matchFullyQualifiedEmoji(source[pos:]); emojiLength > 0 { value = append(value, source[pos:pos+emojiLength]...) pos += emojiLength + previousWasXIDContinuation = false break } r, size := utf8.DecodeRune(source[pos:]) @@ -105,6 +122,7 @@ starter: case isSegmentStarter(r): value = append(value, source[pos:pos+size]...) pos += size + previousWasXIDContinuation = isXIDContinue(r) && !isDefaultIgnorable(r) break starter default: return 0, nil, false @@ -122,6 +140,7 @@ starter: if emojiLength := matchFullyQualifiedEmoji(source[pos:]); emojiLength > 0 { value = append(value, source[pos:pos+emojiLength]...) pos += emojiLength + previousWasXIDContinuation = false continue } r, size := utf8.DecodeRune(source[pos:]) @@ -131,9 +150,15 @@ starter: switch { case isDefaultIgnorable(r): pos += size + previousWasXIDContinuation = false + case isApostropheJoiner(r) && previousWasXIDContinuation && startsNonCombiningXIDContinuation(source[pos+size:]): + value = append(value, source[pos:pos+size]...) + pos += size + previousWasXIDContinuation = false case isSegmentContinuation(r): value = append(value, source[pos:pos+size]...) pos += size + previousWasXIDContinuation = isXIDContinue(r) && !isDefaultIgnorable(r) default: return pos, value, true } diff --git a/internal/markdown/parser/tag_test.go b/internal/markdown/parser/tag_test.go index 07709d68..0dd2e5f4 100644 --- a/internal/markdown/parser/tag_test.go +++ b/internal/markdown/parser/tag_test.go @@ -222,6 +222,18 @@ func TestFindTagMatchesMemosTagV1(t *testing.T) { {name: "Unicode 16 XID character under Unicode 17 profile", input: "#\u1c89", expectedTag: "\u1c89", expectedSource: "#\u1c89", shouldParse: true}, {name: "middle dot XID continue", input: "#l·l", expectedTag: "l·l", expectedSource: "#l·l", shouldParse: true}, {name: "connector punctuation XID continue", input: "#foo‿bar", expectedTag: "foo‿bar", expectedSource: "#foo‿bar", shouldParse: true}, + {name: "ASCII apostrophe joiner", input: "#tag's", expectedTag: "tag's", expectedSource: "#tag's", shouldParse: true}, + {name: "ASCII apostrophe in Ukrainian", input: "#сім'я", expectedTag: "сім'я", expectedSource: "#сім'я", shouldParse: true}, + {name: "right curly apostrophe joiner", input: "#O’Brien", expectedTag: "O’Brien", expectedSource: "#O’Brien", shouldParse: true}, + {name: "modifier letter apostrophe stays XID", input: "#OʼBrien", expectedTag: "OʼBrien", expectedSource: "#OʼBrien", shouldParse: true}, + {name: "combining mark before apostrophe", input: "#cafe\u0301's", expectedTag: "cafe\u0301's", expectedSource: "#cafe\u0301's", shouldParse: true}, + {name: "apostrophe cannot start", input: "#'tag", shouldParse: false}, + {name: "apostrophe cannot end", input: "#users'", expectedTag: "users", expectedSource: "#users", expectedRest: "'", shouldParse: true}, + {name: "apostrophe cannot repeat", input: "#rock''roll", expectedTag: "rock", expectedSource: "#rock", expectedRest: "''roll", shouldParse: true}, + {name: "left curly quote terminates", input: "#O‘Brien", expectedTag: "O", expectedSource: "#O", expectedRest: "‘Brien", shouldParse: true}, + {name: "apostrophe does not adjoin extension", input: "#foo-'bar", expectedTag: "foo-", expectedSource: "#foo-", expectedRest: "'bar", shouldParse: true}, + {name: "apostrophe does not adjoin fully qualified emoji", input: "#foo'1️⃣", expectedTag: "foo", expectedSource: "#foo", expectedRest: "'1️⃣", shouldParse: true}, + {name: "apostrophe does not join across ignored code point", input: "#A\u200d'B", expectedTag: "A", expectedSource: "#A\u200d", expectedRest: "'B", shouldParse: true}, {name: "currency terminates", input: "#price€", expectedTag: "price", expectedSource: "#price", expectedRest: "€", shouldParse: true}, {name: "currency cannot start", input: "#€budget", shouldParse: false}, {name: "other number terminates", input: "#v²", expectedTag: "v", expectedSource: "#v", expectedRest: "²", shouldParse: true}, diff --git a/web/src/utils/tag-grammar.ts b/web/src/utils/tag-grammar.ts index f7f287d1..65166396 100644 --- a/web/src/utils/tag-grammar.ts +++ b/web/src/utils/tag-grammar.ts @@ -89,6 +89,16 @@ function isExtensionUnit(value: string): boolean { return value === "-" || value === "+" || value === "&"; } +function isApostropheJoiner(value: string): boolean { + return value === "'" || value === "’"; +} + +function isNonCombiningXIDContinuationAt(source: string, index: number, limit: number): boolean { + if (index >= limit || emojiAt(source, index, limit)) return false; + const value = codePointAt(source, index); + return index + value.length <= limit && isXIDContinue(value) && !isCombiningMark(value); +} + interface SegmentMatch { to: number; value: string; @@ -109,6 +119,7 @@ function scanSegment(source: string, from: number, limit: number): SegmentMatch } let value = ""; + let previousWasXIDContinuation = false; const starterEmoji = emojiAt(source, index, limit); if (starterEmoji) { value = starterEmoji; @@ -118,6 +129,7 @@ function scanSegment(source: string, from: number, limit: number): SegmentMatch if (!starter || (!(isXIDContinue(starter) && !isCombiningMark(starter)) && !isExtensionUnit(starter))) return undefined; value = starter; index += starter.length; + previousWasXIDContinuation = isXIDContinue(starter); } while (index < limit) { @@ -125,17 +137,30 @@ function scanSegment(source: string, from: number, limit: number): SegmentMatch if (emoji) { value += emoji; index += emoji.length; + previousWasXIDContinuation = false; continue; } const codePoint = codePointAt(source, index); if (isDefaultIgnorable(codePoint)) { index += codePoint.length; + previousWasXIDContinuation = false; + continue; + } + if ( + isApostropheJoiner(codePoint) && + previousWasXIDContinuation && + isNonCombiningXIDContinuationAt(source, index + codePoint.length, limit) + ) { + value += codePoint; + index += codePoint.length; + previousWasXIDContinuation = false; continue; } if (isXIDContinue(codePoint) || isExtensionUnit(codePoint)) { value += codePoint; index += codePoint.length; + previousWasXIDContinuation = isXIDContinue(codePoint); continue; } break; diff --git a/web/tests/editor-decorations.test.ts b/web/tests/editor-decorations.test.ts index 60038231..fb8a30ad 100644 --- a/web/tests/editor-decorations.test.ts +++ b/web/tests/editor-decorations.test.ts @@ -21,6 +21,17 @@ function countClass(doc: string, cls: string): number { describe("tag/mention decorations", () => { it("decorates #tags", () => expect(countClass("a #todo and #work/sub b", "cm-memo-tag")).toBe(2)); it("does not require a left boundary", () => expect(countClass("hello#tag 中文#标签", "cm-memo-tag")).toBe(2)); + it("keeps apostrophes only inside XID words", () => { + const source = "#tag's #сім'я #O’Brien '#quoted' #users' #'missing"; + const state = EditorState.create({ doc: source, extensions: [markdown({ extensions: memoMarkdownExtensions })] }); + expect(findMarkdownTagMatches(state, 0, source.length).map(({ source, value }) => ({ source, value }))).toEqual([ + { source: "tag's", value: "tag's" }, + { source: "сім'я", value: "сім'я" }, + { source: "O’Brien", value: "O’Brien" }, + { source: "quoted", value: "quoted" }, + { source: "users", value: "users" }, + ]); + }); it("uses emitted spans for ignored characters and emoji", () => expect(countClass("#A‍B #́foo #foo/́bar #👩‍💻 #️⃣ ##️⃣", "cm-memo-tag")).toBe(5)); it("keeps maximal valid prefixes and adjacent tags", () => diff --git a/web/tests/editor-tag-autocomplete.test.ts b/web/tests/editor-tag-autocomplete.test.ts index 1f30707d..16e36ef3 100644 --- a/web/tests/editor-tag-autocomplete.test.ts +++ b/web/tests/editor-tag-autocomplete.test.ts @@ -38,6 +38,16 @@ describe("tag autocomplete", () => { expect(result?.options.map((option) => option.label)).toEqual(["AB"]); }); + it("completes tags containing word-internal apostrophes", () => { + const ascii = complete("#O'Br", 5, ["O'Brien", "O’Connor"]); + expect(ascii?.from).toBe(1); + expect(ascii?.options.map((option) => option.label)).toEqual(["O'Brien"]); + + const curly = complete("#O’Co", 5, ["O'Brien", "O’Connor"]); + expect(curly?.from).toBe(1); + expect(curly?.options.map((option) => option.label)).toEqual(["O’Connor"]); + }); + it("keeps unknown character-reference shapes in literal tag source", () => { const source = "#R&bogus;D"; const position = source.indexOf(";"); diff --git a/web/tests/remark-tag.test.tsx b/web/tests/remark-tag.test.tsx index 4a581715..0ba09a0d 100644 --- a/web/tests/remark-tag.test.tsx +++ b/web/tests/remark-tag.test.tsx @@ -534,6 +534,16 @@ describe("remarkMemoSyntax", () => { expect(html).toContain('data-tag="👩‍💻"'); }); + it("keeps word-internal apostrophes without absorbing quotation punctuation", () => { + const html = renderMarkdown("#tag's #сім'я #O’Brien '#quoted' #users'"); + + expect(html).toContain('#tag's'); + expect(html).toContain('#сім'я'); + expect(html).toContain('#O’Brien'); + expect(html).toContain(''#quoted''); + expect(html).toContain('#users''); + }); + it("uses maximal-prefix hierarchy and adjacent-introducer rules", () => { const html = renderMarkdown("#book/ #book//fiction #first#second ##tag #️⃣ ##️⃣"); diff --git a/web/tests/tag-grammar.test.ts b/web/tests/tag-grammar.test.ts index 8c63e50b..e355a673 100644 --- a/web/tests/tag-grammar.test.ts +++ b/web/tests/tag-grammar.test.ts @@ -18,6 +18,10 @@ describe("tag scanner", () => { ["#book//fiction", ["book"]], ["#book/fiction/", ["book/fiction"]], ["#l·l #foo‿bar", ["l·l", "foo‿bar"]], + ["#tag's #сім'я #O'Brien #O’Brien #OʼBrien", ["tag's", "сім'я", "O'Brien", "O’Brien", "OʼBrien"]], + ["#café's", ["café's"]], + ["'#tag' #users' #'missing #rock''roll", ["tag", "users", "rock"]], + ["#O‘Brien #foo-'bar #foo'1️⃣ #A‍'B", ["O", "foo-", "foo", "A"]], ["#foo,bar #price€ #€budget #v²", ["foo", "price", "v"]], ["#first#second", ["first", "second"]], ["##tag", ["tag"]], @@ -50,4 +54,8 @@ describe("tag scanner", () => { it("does not match an emoji across the requested source limit", () => { expect(findTagMatches("#😀", 0, 2)).toEqual([]); }); + + it("does not join an apostrophe across the requested source limit", () => { + expect(findTagMatches("#O'B", 0, 3).map((match) => match.value)).toEqual(["O"]); + }); });