fix(markdown): support word-internal apostrophes in tags (#6135)
This commit is contained in:
parent
8648b97b9e
commit
d617f652b0
9 changed files with 173 additions and 12 deletions
|
|
@ -45,6 +45,7 @@ changing metadata or a separate tag record. Renaming `#Work` to `#work` across 1
|
||||||
## Decision drivers
|
## Decision drivers
|
||||||
|
|
||||||
- Work naturally for multilingual personal notes, including numeric tags and emoji.
|
- Work naturally for multilingual personal notes, including numeric tags and emoji.
|
||||||
|
- Preserve common word-internal apostrophes without absorbing surrounding quotation punctuation into tag values.
|
||||||
- Recognize `hello#tag` intentionally without mistaking URL fragments for tags.
|
- Recognize `hello#tag` intentionally without mistaking URL fragments for tags.
|
||||||
- Give the backend, renderer, editor decoration, and completion the same values and source spans.
|
- Give the backend, renderer, editor decoration, and completion the same values and source spans.
|
||||||
- Use Markdown syntax rather than ad hoc URL or code regular expressions.
|
- Use Markdown syntax rather than ad hoc URL or code regular expressions.
|
||||||
|
|
@ -100,6 +101,11 @@ changing metadata or a separate tag record. Renaming `#Work` to `#work` across 1
|
||||||
**Tag segment**
|
**Tag segment**
|
||||||
: A non-empty component of a hierarchical tag identifier. Slashes separate segments and are not part of any segment.
|
: A non-empty component of a hierarchical tag identifier. Slashes separate segments and are not part of any segment.
|
||||||
|
|
||||||
|
**Apostrophe joiner**
|
||||||
|
: U+0027 APOSTROPHE (`'`) or U+2019 RIGHT SINGLE QUOTATION MARK (`’`) emitted inside a tag segment only when, after emoji-first tokenization, the
|
||||||
|
immediately preceding source code point emits as an `XID_Continue` code point and the immediately following source code point emits as a non-combining
|
||||||
|
`XID_Continue` code point.
|
||||||
|
|
||||||
**Display value**
|
**Display value**
|
||||||
: The direct or implied value presented as a derived tag label. It preserves emitted code points exactly but does not contain ignored default-ignorable code
|
: The direct or implied value presented as a derived tag label. It preserves emitted code points exactly but does not contain ignored default-ignorable code
|
||||||
points or ignored leading combining marks.
|
points or ignored leading combining marks.
|
||||||
|
|
@ -143,7 +149,7 @@ Introducer := U+0023 NUMBER SIGN ("#") outside a FullyQualifiedEmoji
|
||||||
TagSourceSpelling := TagSegmentSpelling ("/" TagSegmentSpelling)*
|
TagSourceSpelling := TagSegmentSpelling ("/" TagSegmentSpelling)*
|
||||||
TagSegmentSpelling := IgnoredPrefix* SegmentStarter SegmentContinuation*
|
TagSegmentSpelling := IgnoredPrefix* SegmentStarter SegmentContinuation*
|
||||||
IgnoredPrefix := IgnoredDefaultIgnorable | IgnoredLeadingCombiningMark
|
IgnoredPrefix := IgnoredDefaultIgnorable | IgnoredLeadingCombiningMark
|
||||||
SegmentContinuation := ValueUnit | IgnoredDefaultIgnorable
|
SegmentContinuation := ValueUnit | ApostropheJoiner | IgnoredDefaultIgnorable
|
||||||
|
|
||||||
SegmentStarter := EmittedXIDContinueCodePointExceptCombiningMark
|
SegmentStarter := EmittedXIDContinueCodePointExceptCombiningMark
|
||||||
| FullyQualifiedEmoji
|
| FullyQualifiedEmoji
|
||||||
|
|
@ -157,6 +163,9 @@ ValueUnit := EmittedXIDContinueCodePoint
|
||||||
| "+"
|
| "+"
|
||||||
| "&"
|
| "&"
|
||||||
|
|
||||||
|
ApostropheJoiner := U+0027 APOSTROPHE ("'")
|
||||||
|
| U+2019 RIGHT SINGLE QUOTATION MARK ("’")
|
||||||
|
|
||||||
IgnoredDefaultIgnorable := Default_Ignorable_Code_Point outside a FullyQualifiedEmoji
|
IgnoredDefaultIgnorable := Default_Ignorable_Code_Point outside a FullyQualifiedEmoji
|
||||||
IgnoredLeadingCombiningMark := XID_Continue with General_Category Mn or Mc,
|
IgnoredLeadingCombiningMark := XID_Continue with General_Category Mn or Mc,
|
||||||
minus Default_Ignorable_Code_Point, before SegmentStarter
|
minus Default_Ignorable_Code_Point, before SegmentStarter
|
||||||
|
|
@ -164,13 +173,14 @@ IgnoredLeadingCombiningMark := XID_Continue with General_Category Mn or Mc,
|
||||||
|
|
||||||
The grammar recognizes source spelling and emits a tag identifier as follows:
|
The grammar recognizes source spelling and emits a tag identifier as follows:
|
||||||
|
|
||||||
- Each `SegmentStarter` and `ValueUnit` emits its exact source code-point sequence. A `FullyQualifiedEmoji` therefore emits its complete matched sequence.
|
- Each `SegmentStarter`, `ValueUnit`, and contextually valid `ApostropheJoiner` emits its exact source code-point sequence. A `FullyQualifiedEmoji`
|
||||||
|
therefore emits its complete matched sequence.
|
||||||
- Each consumed `/` emits one U+002F SOLIDUS into the identifier.
|
- Each consumed `/` emits one U+002F SOLIDUS into the identifier.
|
||||||
- `Introducer`, `IgnoredDefaultIgnorable`, and `IgnoredLeadingCombiningMark` emit nothing.
|
- `Introducer`, `IgnoredDefaultIgnorable`, and `IgnoredLeadingCombiningMark` emit nothing.
|
||||||
|
|
||||||
At every source position, the token priority is the longest `FullyQualifiedEmoji`, then `IgnoredDefaultIgnorable`, then `IgnoredLeadingCombiningMark`, then
|
At every source position, the token priority is the longest `FullyQualifiedEmoji`, then `IgnoredDefaultIgnorable`, then `IgnoredLeadingCombiningMark`, then
|
||||||
an emitted code point or Memos extension unit. This preserves default-ignorable code points inside matched fully-qualified emoji sequences while omitting
|
a contextually valid `ApostropheJoiner`, then an emitted code point or Memos extension unit. This preserves default-ignorable code points inside matched
|
||||||
them everywhere else.
|
fully-qualified emoji sequences while omitting them everywhere else.
|
||||||
|
|
||||||
The emitted tag identifier is the direct tag value.
|
The emitted tag identifier is the direct tag value.
|
||||||
|
|
||||||
|
|
@ -199,12 +209,20 @@ Rules:
|
||||||
12. A non-default-ignorable `XID_Continue` code point whose General Category is `Mn` (Nonspacing Mark) or `Mc` (Spacing Combining Mark) is consumed but
|
12. A non-default-ignorable `XID_Continue` code point whose General Category is `Mn` (Nonspacing Mark) or `Mc` (Spacing Combining Mark) is consumed but
|
||||||
omitted while it precedes the starter of its segment. The same non-default-ignorable code point is preserved as a value unit after that segment's
|
omitted while it precedes the starter of its segment. The same non-default-ignorable code point is preserved as a value unit after that segment's
|
||||||
starter.
|
starter.
|
||||||
13. There is no tag-specific identifier length limit. The enclosing memo-size limit provides the resource bound.
|
13. After applying the longest-`FullyQualifiedEmoji` token priority, U+0027 APOSTROPHE and U+2019 RIGHT SINGLE QUOTATION MARK are emitted as
|
||||||
|
`ApostropheJoiner` only when the immediately preceding source code point in the same segment was emitted as an `EmittedXIDContinueCodePoint` and the
|
||||||
|
immediately following source code point can emit as an
|
||||||
|
`EmittedXIDContinueCodePointExceptCombiningMark`. An apostrophe joiner therefore cannot start or end a segment, repeat without an intervening XID
|
||||||
|
code point, adjoin a fully-qualified emoji or Memos extension unit, or join across an ignored code point.
|
||||||
|
14. U+02BC MODIFIER LETTER APOSTROPHE (`ʼ`) is already an `XID_Continue` code point. It follows the ordinary XID rules rather than the contextual
|
||||||
|
apostrophe-joiner rule.
|
||||||
|
15. There is no tag-specific identifier length limit. The enclosing memo-size limit provides the resource bound.
|
||||||
|
|
||||||
The grammar is intentionally broader than a programming-language identifier grammar. It does not require the first emitted unit to be `XID_Start`:
|
The grammar is intentionally broader than a programming-language identifier grammar. It does not require the first emitted unit to be `XID_Start`:
|
||||||
digits, `_`, `-`, `+`, `&`, and fully-qualified emoji can start a segment. Default-ignorable code points and non-default-ignorable `Mn` or `Mc` code points
|
digits, `_`, `-`, `+`, `&`, and fully-qualified emoji can start a segment. Default-ignorable code points and non-default-ignorable `Mn` or `Mc` code points
|
||||||
may occur before each segment's starter but are omitted from the value. After the starter, those non-default-ignorable `Mn` and `Mc` code points are
|
may occur before each segment's starter but are omitted from the value. After the starter, those non-default-ignorable `Mn` and `Mc` code points are
|
||||||
preserved. A slash cannot begin a segment, and segments made only from `-`, `+`, and `&` are explicitly valid.
|
preserved. ASCII and right-curly apostrophes are emitted only as contextual joiners between XID code points. A slash cannot begin a segment, and segments
|
||||||
|
made only from `-`, `+`, and `&` are explicitly valid.
|
||||||
|
|
||||||
### Maximal-prefix scanning
|
### Maximal-prefix scanning
|
||||||
|
|
||||||
|
|
@ -215,12 +233,15 @@ After finding an introducer, the lexer consumes the longest valid `TagSourceSpel
|
||||||
2. Emit a segment starter, trying the longest matching `FullyQualifiedEmoji` before any shorter unit.
|
2. Emit a segment starter, trying the longest matching `FullyQualifiedEmoji` before any shorter unit.
|
||||||
3. After the starter, preserve non-default-ignorable `XID_Continue` combining marks as ordinary value units; continue consuming default-ignorable code
|
3. After the starter, preserve non-default-ignorable `XID_Continue` combining marks as ordinary value units; continue consuming default-ignorable code
|
||||||
points without emitting them.
|
points without emitting them.
|
||||||
4. Consume `/` only when the following source, after any ignored prefix, can emit a segment starter; then consume that segment.
|
4. Emit U+0027 or U+2019 as an apostrophe joiner only when, after emoji-first tokenization, the immediately preceding source code point emitted an XID
|
||||||
5. Stop before a `/` that is leading, trailing, or followed only by an ignored prefix and then another `/`, a non-starter, or the end of the literal-source
|
continuation unit and the immediately following source code point can emit a non-combining XID continuation unit. Otherwise stop before the
|
||||||
|
apostrophe.
|
||||||
|
5. Consume `/` only when the following source, after any ignored prefix, can emit a segment starter; then consume that segment.
|
||||||
|
6. Stop before a `/` that is leading, trailing, or followed only by an ignored prefix and then another `/`, a non-starter, or the end of the literal-source
|
||||||
run, leaving that slash and the remaining source unconsumed.
|
run, leaving that slash and the remaining source unconsumed.
|
||||||
6. Otherwise stop before the first code point that cannot begin a valid continuation unit.
|
7. Otherwise stop before the first code point that cannot begin a valid continuation unit.
|
||||||
7. Keep the valid prefix already consumed; a later invalid character does not invalidate it.
|
8. Keep the valid prefix already consumed; a later invalid character does not invalidate it.
|
||||||
8. Produce no candidate if the first `TagSegmentSpelling` cannot match.
|
9. Produce no candidate if the first `TagSegmentSpelling` cannot match.
|
||||||
|
|
||||||
### Candidate enumeration
|
### Candidate enumeration
|
||||||
|
|
||||||
|
|
@ -255,6 +276,13 @@ Examples:
|
||||||
| `#price€` | `price` | `€` |
|
| `#price€` | `price` | `€` |
|
||||||
| `#C++` | `C++` | empty |
|
| `#C++` | `C++` | empty |
|
||||||
| `#R&D` | `R&D` | empty |
|
| `#R&D` | `R&D` | empty |
|
||||||
|
| `#tag's` | `tag's` | empty |
|
||||||
|
| `#сім'я` | `сім'я` | empty |
|
||||||
|
| `#O’Brien` | `O’Brien` | empty |
|
||||||
|
| `#café's` | `café's` | empty |
|
||||||
|
| `#users'` | `users` | `'` |
|
||||||
|
| `#foo'1️⃣` | `foo` | `'1️⃣` |
|
||||||
|
| `#'tag` | none | `'tag` |
|
||||||
| `#-foo` | `-foo` | empty |
|
| `#-foo` | `-foo` | empty |
|
||||||
| `#foo-` | `foo-` | empty |
|
| `#foo-` | `foo-` | empty |
|
||||||
| `#---` | `---` | empty |
|
| `#---` | `---` | empty |
|
||||||
|
|
@ -380,6 +408,8 @@ deduplicating, counting, filtering, navigating, or performing exact metadata loo
|
||||||
#café / #café
|
#café / #café
|
||||||
#A / #A
|
#A / #A
|
||||||
#straße / #STRASSE
|
#straße / #STRASSE
|
||||||
|
#O'Brien / #O’Brien
|
||||||
|
#O’Brien / #OʼBrien
|
||||||
```
|
```
|
||||||
|
|
||||||
Multiple occurrences that emit exactly equal direct tag values in one memo produce one memo-tag membership. A tag metadata rule may deliberately match
|
Multiple occurrences that emit exactly equal direct tag values in one memo produce one memo-tag membership. A tag metadata rule may deliberately match
|
||||||
|
|
@ -432,6 +462,16 @@ The following examples are normative for the lexical and context decisions alrea
|
||||||
| `#2026` | `2026` | Numeric-only identifiers are valid |
|
| `#2026` | `2026` | Numeric-only identifiers are valid |
|
||||||
| `#C++` | `C++` | Explicit `+` extension |
|
| `#C++` | `C++` | Explicit `+` extension |
|
||||||
| `#R&D` | `R&D` | Explicit `&` extension |
|
| `#R&D` | `R&D` | Explicit `&` extension |
|
||||||
|
| `#tag's` | `tag's` | ASCII apostrophe joins two XID code points |
|
||||||
|
| `#сім'я` | `сім'я` | ASCII apostrophe preserves a Ukrainian word |
|
||||||
|
| `#O’Brien` | `O’Brien` | Right single quotation mark joins two XID code points |
|
||||||
|
| `#café's` | `café's` | An emitted combining mark may precede an apostrophe joiner |
|
||||||
|
| `#users'` | `users` | A trailing apostrophe is not a joiner |
|
||||||
|
| `#foo'1️⃣` | `foo` | Emoji-first tokenization prevents an apostrophe from adjoining the keycap sequence |
|
||||||
|
| `#'tag` | none | An apostrophe cannot start a segment |
|
||||||
|
| `'#tag'` | `tag` | Surrounding quotation punctuation remains outside the occurrence |
|
||||||
|
| `#rock’n’roll` | `rock’n’roll` | Multiple apostrophe joiners are valid when each independently satisfies the context rule |
|
||||||
|
| `#OʼBrien` | `OʼBrien` | U+02BC is an ordinary `XID_Continue` code point |
|
||||||
| `#` followed by 101 `a` code points | all 101 `a` code points | There is no tag-specific length limit |
|
| `#` followed by 101 `a` code points | all 101 `a` code points | There is no tag-specific length limit |
|
||||||
| `#-foo` | `-foo` | Visible connector extensions may begin a segment |
|
| `#-foo` | `-foo` | Visible connector extensions may begin a segment |
|
||||||
| `#foo-` | `foo-` | Visible connector extensions may end a segment |
|
| `#foo-` | `foo-` | Visible connector extensions may end a segment |
|
||||||
|
|
@ -490,7 +530,9 @@ The following examples are normative for the lexical and context decisions alrea
|
||||||
- Common multilingual tags, numeric tags, hierarchy characters, and emoji remain expressive.
|
- Common multilingual tags, numeric tags, hierarchy characters, and emoji remain expressive.
|
||||||
- Hierarchical ancestors have one consistent meaning across API membership, exact filters, navigation, and counts.
|
- Hierarchical ancestors have one consistent meaning across API membership, exact filters, navigation, and counts.
|
||||||
- `hello#tag` works consistently while actual Markdown links and URLs are excluded structurally.
|
- `hello#tag` works consistently while actual Markdown links and URLs are excluded structurally.
|
||||||
- Markdown punctuation and delimiters such as comma, backtick, currency, and mathematical operators stop an identifier predictably.
|
- Common word-internal apostrophes support multilingual words and names while surrounding quotation punctuation remains outside tag values.
|
||||||
|
- Unsupported Markdown punctuation, delimiters, currency symbols, and operators stop an identifier predictably; apostrophes are the explicitly
|
||||||
|
constrained exception.
|
||||||
- Exact equality preserves every emitted code-point distinction and avoids locale-dependent identity rules.
|
- Exact equality preserves every emitted code-point distinction and avoids locale-dependent identity rules.
|
||||||
- Ignoring default-ignorable code points outside emoji avoids invisible tag distinctions while preserving matched fully-qualified emoji sequences.
|
- Ignoring default-ignorable code points outside emoji avoids invisible tag distinctions while preserving matched fully-qualified emoji sequences.
|
||||||
- Ignoring leading non-default-ignorable `Mn` and `Mc` code points only before each segment's visible starter avoids invisible-leading segments without
|
- Ignoring leading non-default-ignorable `Mn` and `Mc` code points only before each segment's visible starter avoids invisible-leading segments without
|
||||||
|
|
@ -506,6 +548,8 @@ The following examples are normative for the lexical and context decisions alrea
|
||||||
- Pinning Unicode data requires maintenance when Unicode and Emoji data are updated.
|
- Pinning Unicode data requires maintenance when Unicode and Emoji data are updated.
|
||||||
- Visually indistinguishable or canonically equivalent source spellings whose emitted values differ remain separate tags unless the user edits their memo
|
- Visually indistinguishable or canonically equivalent source spellings whose emitted values differ remain separate tags unless the user edits their memo
|
||||||
sources to make those values identical.
|
sources to make those values identical.
|
||||||
|
- Visually similar apostrophe spellings such as U+0027, U+2019, and U+02BC remain distinct under exact tag identity.
|
||||||
|
- English possessive-looking source such as `#tag's` denotes the complete tag value `tag's`, not `tag` followed by prose.
|
||||||
- Source spellings that differ only by ignored default-ignorable code points intentionally collapse to the same emitted tag value.
|
- Source spellings that differ only by ignored default-ignorable code points intentionally collapse to the same emitted tag value.
|
||||||
- Source spellings that differ only by ignored leading combining marks intentionally collapse to the same emitted tag value.
|
- Source spellings that differ only by ignored leading combining marks intentionally collapse to the same emitted tag value.
|
||||||
|
|
||||||
|
|
@ -554,6 +598,12 @@ tag. The slash and remaining source stay ordinary Markdown.
|
||||||
Rejected. Requiring `-`, `+`, or `&` to be medial, non-repeating, or accompanied by a letter would add validation rules without resolving a structural
|
Rejected. Requiring `-`, `+`, or `&` to be medial, non-repeating, or accompanied by a letter would add validation rules without resolving a structural
|
||||||
ambiguity. They remain ordinary segment units; only `/` has structural meaning.
|
ambiguity. They remain ordinary segment units; only `/` has structural meaning.
|
||||||
|
|
||||||
|
### Treat apostrophes as unrestricted value units
|
||||||
|
|
||||||
|
Rejected. Allowing apostrophes as starters, trailing units, or repeatable ordinary units would absorb surrounding quotation punctuation into values such
|
||||||
|
as `tag'` and make quoted source such as `'#tag'` ambiguous. The contextual joiner rule supports words and names while leaving punctuation at tag
|
||||||
|
boundaries unconsumed.
|
||||||
|
|
||||||
### Recognize compatibility number signs
|
### Recognize compatibility number signs
|
||||||
|
|
||||||
Rejected. U+FE5F SMALL NUMBER SIGN and U+FF03 FULLWIDTH NUMBER SIGN are visually similar to `#` but are not Markdown syntax. Recognizing only ASCII U+0023
|
Rejected. U+FE5F SMALL NUMBER SIGN and U+FF03 FULLWIDTH NUMBER SIGN are visually similar to `#` but are not Markdown syntax. Recognizing only ASCII U+0023
|
||||||
|
|
|
||||||
|
|
@ -576,6 +576,16 @@ func TestExtractTagsMemosTagV1(t *testing.T) {
|
||||||
{name: "number sign keycap continuation", content: "#first#️⃣", expected: []string{"first#️⃣"}},
|
{name: "number sign keycap continuation", content: "#first#️⃣", expected: []string{"first#️⃣"}},
|
||||||
{name: "exact identity", content: "#Work #work", expected: []string{"Work", "work"}},
|
{name: "exact identity", content: "#Work #work", expected: []string{"Work", "work"}},
|
||||||
{name: "no normalization", content: "#café #cafe\u0301", expected: []string{"café", "cafe\u0301"}},
|
{name: "no normalization", content: "#café #cafe\u0301", expected: []string{"café", "cafe\u0301"}},
|
||||||
|
{
|
||||||
|
name: "word internal apostrophes",
|
||||||
|
content: "#tag's #сім'я #O'Brien #O’Brien #OʼBrien #cafe\u0301's",
|
||||||
|
expected: []string{"tag's", "сім'я", "O'Brien", "O’Brien", "OʼBrien", "cafe\u0301's"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "apostrophe boundaries",
|
||||||
|
content: "'#tag' #users' #'missing #rock''roll #O‘Brien #A\u200d'B",
|
||||||
|
expected: []string{"tag", "users", "rock", "O", "A"},
|
||||||
|
},
|
||||||
{name: "ignored spellings deduplicate", content: "#AB #A\u200dB #\u0301AB", expected: []string{"AB"}},
|
{name: "ignored spellings deduplicate", content: "#AB #A\u200dB #\u0301AB", expected: []string{"AB"}},
|
||||||
{name: "hierarchy expansion", content: "#book/fiction/history", expected: []string{"book", "book/fiction", "book/fiction/history"}},
|
{name: "hierarchy expansion", content: "#book/fiction/history", expected: []string{"book", "book/fiction", "book/fiction/history"}},
|
||||||
{name: "hierarchy exact dedupe", content: "#book/fiction #book", expected: []string{"book", "book/fiction"}},
|
{name: "hierarchy exact dedupe", content: "#book/fiction #book", expected: []string{"book", "book/fiction"}},
|
||||||
|
|
|
||||||
|
|
@ -60,6 +60,21 @@ func isSegmentContinuation(r rune) bool {
|
||||||
return r == '-' || r == '+' || r == '&' || isXIDContinue(r) && !isDefaultIgnorable(r)
|
return r == '-' || r == '+' || r == '&' || isXIDContinue(r) && !isDefaultIgnorable(r)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func isApostropheJoiner(r rune) bool {
|
||||||
|
return r == '\'' || r == '\u2019'
|
||||||
|
}
|
||||||
|
|
||||||
|
func startsNonCombiningXIDContinuation(source []byte) bool {
|
||||||
|
if len(source) == 0 || matchFullyQualifiedEmoji(source) > 0 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
r, size := utf8.DecodeRune(source)
|
||||||
|
if r == utf8.RuneError && size == 1 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return isXIDContinue(r) && !isCombiningMark(r) && !isDefaultIgnorable(r)
|
||||||
|
}
|
||||||
|
|
||||||
func matchFullyQualifiedEmoji(source []byte) int {
|
func matchFullyQualifiedEmoji(source []byte) int {
|
||||||
limit := min(len(source), maxEmoji17SequenceBytes)
|
limit := min(len(source), maxEmoji17SequenceBytes)
|
||||||
match := 0
|
match := 0
|
||||||
|
|
@ -82,6 +97,7 @@ func matchFullyQualifiedEmoji(source []byte) int {
|
||||||
func scanTagSegment(source []byte) (int, []byte, bool) {
|
func scanTagSegment(source []byte) (int, []byte, bool) {
|
||||||
pos := 0
|
pos := 0
|
||||||
var value []byte
|
var value []byte
|
||||||
|
previousWasXIDContinuation := false
|
||||||
|
|
||||||
starter:
|
starter:
|
||||||
for pos < len(source) {
|
for pos < len(source) {
|
||||||
|
|
@ -91,6 +107,7 @@ starter:
|
||||||
if emojiLength := matchFullyQualifiedEmoji(source[pos:]); emojiLength > 0 {
|
if emojiLength := matchFullyQualifiedEmoji(source[pos:]); emojiLength > 0 {
|
||||||
value = append(value, source[pos:pos+emojiLength]...)
|
value = append(value, source[pos:pos+emojiLength]...)
|
||||||
pos += emojiLength
|
pos += emojiLength
|
||||||
|
previousWasXIDContinuation = false
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
r, size := utf8.DecodeRune(source[pos:])
|
r, size := utf8.DecodeRune(source[pos:])
|
||||||
|
|
@ -105,6 +122,7 @@ starter:
|
||||||
case isSegmentStarter(r):
|
case isSegmentStarter(r):
|
||||||
value = append(value, source[pos:pos+size]...)
|
value = append(value, source[pos:pos+size]...)
|
||||||
pos += size
|
pos += size
|
||||||
|
previousWasXIDContinuation = isXIDContinue(r) && !isDefaultIgnorable(r)
|
||||||
break starter
|
break starter
|
||||||
default:
|
default:
|
||||||
return 0, nil, false
|
return 0, nil, false
|
||||||
|
|
@ -122,6 +140,7 @@ starter:
|
||||||
if emojiLength := matchFullyQualifiedEmoji(source[pos:]); emojiLength > 0 {
|
if emojiLength := matchFullyQualifiedEmoji(source[pos:]); emojiLength > 0 {
|
||||||
value = append(value, source[pos:pos+emojiLength]...)
|
value = append(value, source[pos:pos+emojiLength]...)
|
||||||
pos += emojiLength
|
pos += emojiLength
|
||||||
|
previousWasXIDContinuation = false
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
r, size := utf8.DecodeRune(source[pos:])
|
r, size := utf8.DecodeRune(source[pos:])
|
||||||
|
|
@ -131,9 +150,15 @@ starter:
|
||||||
switch {
|
switch {
|
||||||
case isDefaultIgnorable(r):
|
case isDefaultIgnorable(r):
|
||||||
pos += size
|
pos += size
|
||||||
|
previousWasXIDContinuation = false
|
||||||
|
case isApostropheJoiner(r) && previousWasXIDContinuation && startsNonCombiningXIDContinuation(source[pos+size:]):
|
||||||
|
value = append(value, source[pos:pos+size]...)
|
||||||
|
pos += size
|
||||||
|
previousWasXIDContinuation = false
|
||||||
case isSegmentContinuation(r):
|
case isSegmentContinuation(r):
|
||||||
value = append(value, source[pos:pos+size]...)
|
value = append(value, source[pos:pos+size]...)
|
||||||
pos += size
|
pos += size
|
||||||
|
previousWasXIDContinuation = isXIDContinue(r) && !isDefaultIgnorable(r)
|
||||||
default:
|
default:
|
||||||
return pos, value, true
|
return pos, value, true
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -222,6 +222,18 @@ func TestFindTagMatchesMemosTagV1(t *testing.T) {
|
||||||
{name: "Unicode 16 XID character under Unicode 17 profile", input: "#\u1c89", expectedTag: "\u1c89", expectedSource: "#\u1c89", shouldParse: true},
|
{name: "Unicode 16 XID character under Unicode 17 profile", input: "#\u1c89", expectedTag: "\u1c89", expectedSource: "#\u1c89", shouldParse: true},
|
||||||
{name: "middle dot XID continue", input: "#l·l", expectedTag: "l·l", expectedSource: "#l·l", shouldParse: true},
|
{name: "middle dot XID continue", input: "#l·l", expectedTag: "l·l", expectedSource: "#l·l", shouldParse: true},
|
||||||
{name: "connector punctuation XID continue", input: "#foo‿bar", expectedTag: "foo‿bar", expectedSource: "#foo‿bar", shouldParse: true},
|
{name: "connector punctuation XID continue", input: "#foo‿bar", expectedTag: "foo‿bar", expectedSource: "#foo‿bar", shouldParse: true},
|
||||||
|
{name: "ASCII apostrophe joiner", input: "#tag's", expectedTag: "tag's", expectedSource: "#tag's", shouldParse: true},
|
||||||
|
{name: "ASCII apostrophe in Ukrainian", input: "#сім'я", expectedTag: "сім'я", expectedSource: "#сім'я", shouldParse: true},
|
||||||
|
{name: "right curly apostrophe joiner", input: "#O’Brien", expectedTag: "O’Brien", expectedSource: "#O’Brien", shouldParse: true},
|
||||||
|
{name: "modifier letter apostrophe stays XID", input: "#OʼBrien", expectedTag: "OʼBrien", expectedSource: "#OʼBrien", shouldParse: true},
|
||||||
|
{name: "combining mark before apostrophe", input: "#cafe\u0301's", expectedTag: "cafe\u0301's", expectedSource: "#cafe\u0301's", shouldParse: true},
|
||||||
|
{name: "apostrophe cannot start", input: "#'tag", shouldParse: false},
|
||||||
|
{name: "apostrophe cannot end", input: "#users'", expectedTag: "users", expectedSource: "#users", expectedRest: "'", shouldParse: true},
|
||||||
|
{name: "apostrophe cannot repeat", input: "#rock''roll", expectedTag: "rock", expectedSource: "#rock", expectedRest: "''roll", shouldParse: true},
|
||||||
|
{name: "left curly quote terminates", input: "#O‘Brien", expectedTag: "O", expectedSource: "#O", expectedRest: "‘Brien", shouldParse: true},
|
||||||
|
{name: "apostrophe does not adjoin extension", input: "#foo-'bar", expectedTag: "foo-", expectedSource: "#foo-", expectedRest: "'bar", shouldParse: true},
|
||||||
|
{name: "apostrophe does not adjoin fully qualified emoji", input: "#foo'1️⃣", expectedTag: "foo", expectedSource: "#foo", expectedRest: "'1️⃣", shouldParse: true},
|
||||||
|
{name: "apostrophe does not join across ignored code point", input: "#A\u200d'B", expectedTag: "A", expectedSource: "#A\u200d", expectedRest: "'B", shouldParse: true},
|
||||||
{name: "currency terminates", input: "#price€", expectedTag: "price", expectedSource: "#price", expectedRest: "€", shouldParse: true},
|
{name: "currency terminates", input: "#price€", expectedTag: "price", expectedSource: "#price", expectedRest: "€", shouldParse: true},
|
||||||
{name: "currency cannot start", input: "#€budget", shouldParse: false},
|
{name: "currency cannot start", input: "#€budget", shouldParse: false},
|
||||||
{name: "other number terminates", input: "#v²", expectedTag: "v", expectedSource: "#v", expectedRest: "²", shouldParse: true},
|
{name: "other number terminates", input: "#v²", expectedTag: "v", expectedSource: "#v", expectedRest: "²", shouldParse: true},
|
||||||
|
|
|
||||||
|
|
@ -89,6 +89,16 @@ function isExtensionUnit(value: string): boolean {
|
||||||
return value === "-" || value === "+" || value === "&";
|
return value === "-" || value === "+" || value === "&";
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function isApostropheJoiner(value: string): boolean {
|
||||||
|
return value === "'" || value === "’";
|
||||||
|
}
|
||||||
|
|
||||||
|
function isNonCombiningXIDContinuationAt(source: string, index: number, limit: number): boolean {
|
||||||
|
if (index >= limit || emojiAt(source, index, limit)) return false;
|
||||||
|
const value = codePointAt(source, index);
|
||||||
|
return index + value.length <= limit && isXIDContinue(value) && !isCombiningMark(value);
|
||||||
|
}
|
||||||
|
|
||||||
interface SegmentMatch {
|
interface SegmentMatch {
|
||||||
to: number;
|
to: number;
|
||||||
value: string;
|
value: string;
|
||||||
|
|
@ -109,6 +119,7 @@ function scanSegment(source: string, from: number, limit: number): SegmentMatch
|
||||||
}
|
}
|
||||||
|
|
||||||
let value = "";
|
let value = "";
|
||||||
|
let previousWasXIDContinuation = false;
|
||||||
const starterEmoji = emojiAt(source, index, limit);
|
const starterEmoji = emojiAt(source, index, limit);
|
||||||
if (starterEmoji) {
|
if (starterEmoji) {
|
||||||
value = starterEmoji;
|
value = starterEmoji;
|
||||||
|
|
@ -118,6 +129,7 @@ function scanSegment(source: string, from: number, limit: number): SegmentMatch
|
||||||
if (!starter || (!(isXIDContinue(starter) && !isCombiningMark(starter)) && !isExtensionUnit(starter))) return undefined;
|
if (!starter || (!(isXIDContinue(starter) && !isCombiningMark(starter)) && !isExtensionUnit(starter))) return undefined;
|
||||||
value = starter;
|
value = starter;
|
||||||
index += starter.length;
|
index += starter.length;
|
||||||
|
previousWasXIDContinuation = isXIDContinue(starter);
|
||||||
}
|
}
|
||||||
|
|
||||||
while (index < limit) {
|
while (index < limit) {
|
||||||
|
|
@ -125,17 +137,30 @@ function scanSegment(source: string, from: number, limit: number): SegmentMatch
|
||||||
if (emoji) {
|
if (emoji) {
|
||||||
value += emoji;
|
value += emoji;
|
||||||
index += emoji.length;
|
index += emoji.length;
|
||||||
|
previousWasXIDContinuation = false;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
const codePoint = codePointAt(source, index);
|
const codePoint = codePointAt(source, index);
|
||||||
if (isDefaultIgnorable(codePoint)) {
|
if (isDefaultIgnorable(codePoint)) {
|
||||||
index += codePoint.length;
|
index += codePoint.length;
|
||||||
|
previousWasXIDContinuation = false;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (
|
||||||
|
isApostropheJoiner(codePoint) &&
|
||||||
|
previousWasXIDContinuation &&
|
||||||
|
isNonCombiningXIDContinuationAt(source, index + codePoint.length, limit)
|
||||||
|
) {
|
||||||
|
value += codePoint;
|
||||||
|
index += codePoint.length;
|
||||||
|
previousWasXIDContinuation = false;
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if (isXIDContinue(codePoint) || isExtensionUnit(codePoint)) {
|
if (isXIDContinue(codePoint) || isExtensionUnit(codePoint)) {
|
||||||
value += codePoint;
|
value += codePoint;
|
||||||
index += codePoint.length;
|
index += codePoint.length;
|
||||||
|
previousWasXIDContinuation = isXIDContinue(codePoint);
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
|
|
|
||||||
|
|
@ -21,6 +21,17 @@ function countClass(doc: string, cls: string): number {
|
||||||
describe("tag/mention decorations", () => {
|
describe("tag/mention decorations", () => {
|
||||||
it("decorates #tags", () => expect(countClass("a #todo and #work/sub b", "cm-memo-tag")).toBe(2));
|
it("decorates #tags", () => expect(countClass("a #todo and #work/sub b", "cm-memo-tag")).toBe(2));
|
||||||
it("does not require a left boundary", () => expect(countClass("hello#tag 中文#标签", "cm-memo-tag")).toBe(2));
|
it("does not require a left boundary", () => expect(countClass("hello#tag 中文#标签", "cm-memo-tag")).toBe(2));
|
||||||
|
it("keeps apostrophes only inside XID words", () => {
|
||||||
|
const source = "#tag's #сім'я #O’Brien '#quoted' #users' #'missing";
|
||||||
|
const state = EditorState.create({ doc: source, extensions: [markdown({ extensions: memoMarkdownExtensions })] });
|
||||||
|
expect(findMarkdownTagMatches(state, 0, source.length).map(({ source, value }) => ({ source, value }))).toEqual([
|
||||||
|
{ source: "tag's", value: "tag's" },
|
||||||
|
{ source: "сім'я", value: "сім'я" },
|
||||||
|
{ source: "O’Brien", value: "O’Brien" },
|
||||||
|
{ source: "quoted", value: "quoted" },
|
||||||
|
{ source: "users", value: "users" },
|
||||||
|
]);
|
||||||
|
});
|
||||||
it("uses emitted spans for ignored characters and emoji", () =>
|
it("uses emitted spans for ignored characters and emoji", () =>
|
||||||
expect(countClass("#AB #́foo #foo/́bar #👩💻 #️⃣ ##️⃣", "cm-memo-tag")).toBe(5));
|
expect(countClass("#AB #́foo #foo/́bar #👩💻 #️⃣ ##️⃣", "cm-memo-tag")).toBe(5));
|
||||||
it("keeps maximal valid prefixes and adjacent tags", () =>
|
it("keeps maximal valid prefixes and adjacent tags", () =>
|
||||||
|
|
|
||||||
|
|
@ -38,6 +38,16 @@ describe("tag autocomplete", () => {
|
||||||
expect(result?.options.map((option) => option.label)).toEqual(["AB"]);
|
expect(result?.options.map((option) => option.label)).toEqual(["AB"]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("completes tags containing word-internal apostrophes", () => {
|
||||||
|
const ascii = complete("#O'Br", 5, ["O'Brien", "O’Connor"]);
|
||||||
|
expect(ascii?.from).toBe(1);
|
||||||
|
expect(ascii?.options.map((option) => option.label)).toEqual(["O'Brien"]);
|
||||||
|
|
||||||
|
const curly = complete("#O’Co", 5, ["O'Brien", "O’Connor"]);
|
||||||
|
expect(curly?.from).toBe(1);
|
||||||
|
expect(curly?.options.map((option) => option.label)).toEqual(["O’Connor"]);
|
||||||
|
});
|
||||||
|
|
||||||
it("keeps unknown character-reference shapes in literal tag source", () => {
|
it("keeps unknown character-reference shapes in literal tag source", () => {
|
||||||
const source = "#R&bogus;D";
|
const source = "#R&bogus;D";
|
||||||
const position = source.indexOf(";");
|
const position = source.indexOf(";");
|
||||||
|
|
|
||||||
|
|
@ -534,6 +534,16 @@ describe("remarkMemoSyntax", () => {
|
||||||
expect(html).toContain('data-tag="👩💻"');
|
expect(html).toContain('data-tag="👩💻"');
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("keeps word-internal apostrophes without absorbing quotation punctuation", () => {
|
||||||
|
const html = renderMarkdown("#tag's #сім'я #O’Brien '#quoted' #users'");
|
||||||
|
|
||||||
|
expect(html).toContain('<span class="tag" data-tag="tag's">#tag's</span>');
|
||||||
|
expect(html).toContain('<span class="tag" data-tag="сім'я">#сім'я</span>');
|
||||||
|
expect(html).toContain('<span class="tag" data-tag="O’Brien">#O’Brien</span>');
|
||||||
|
expect(html).toContain(''<span class="tag" data-tag="quoted">#quoted</span>'');
|
||||||
|
expect(html).toContain('<span class="tag" data-tag="users">#users</span>'');
|
||||||
|
});
|
||||||
|
|
||||||
it("uses maximal-prefix hierarchy and adjacent-introducer rules", () => {
|
it("uses maximal-prefix hierarchy and adjacent-introducer rules", () => {
|
||||||
const html = renderMarkdown("#book/ #book//fiction #first#second ##tag #️⃣ ##️⃣");
|
const html = renderMarkdown("#book/ #book//fiction #first#second ##tag #️⃣ ##️⃣");
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -18,6 +18,10 @@ describe("tag scanner", () => {
|
||||||
["#book//fiction", ["book"]],
|
["#book//fiction", ["book"]],
|
||||||
["#book/fiction/", ["book/fiction"]],
|
["#book/fiction/", ["book/fiction"]],
|
||||||
["#l·l #foo‿bar", ["l·l", "foo‿bar"]],
|
["#l·l #foo‿bar", ["l·l", "foo‿bar"]],
|
||||||
|
["#tag's #сім'я #O'Brien #O’Brien #OʼBrien", ["tag's", "сім'я", "O'Brien", "O’Brien", "OʼBrien"]],
|
||||||
|
["#café's", ["café's"]],
|
||||||
|
["'#tag' #users' #'missing #rock''roll", ["tag", "users", "rock"]],
|
||||||
|
["#O‘Brien #foo-'bar #foo'1️⃣ #A'B", ["O", "foo-", "foo", "A"]],
|
||||||
["#foo,bar #price€ #€budget #v²", ["foo", "price", "v"]],
|
["#foo,bar #price€ #€budget #v²", ["foo", "price", "v"]],
|
||||||
["#first#second", ["first", "second"]],
|
["#first#second", ["first", "second"]],
|
||||||
["##tag", ["tag"]],
|
["##tag", ["tag"]],
|
||||||
|
|
@ -50,4 +54,8 @@ describe("tag scanner", () => {
|
||||||
it("does not match an emoji across the requested source limit", () => {
|
it("does not match an emoji across the requested source limit", () => {
|
||||||
expect(findTagMatches("#😀", 0, 2)).toEqual([]);
|
expect(findTagMatches("#😀", 0, 2)).toEqual([]);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it("does not join an apostrophe across the requested source limit", () => {
|
||||||
|
expect(findTagMatches("#O'B", 0, 3).map((match) => match.value)).toEqual(["O"]);
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue