diff --git a/index.js b/index.js index aad1b05..42ec3a1 100644 --- a/index.js +++ b/index.js @@ -5,22 +5,49 @@ import {eastAsianWidth} from 'get-east-asian-width'; Logic: - Segment graphemes to match how terminals render clusters. - Width rules: - 1. Skip non-printing clusters (Default_Ignorable, Control, pure Mark, lone Surrogates). Tabs are ignored by design. + 1. Skip non-printing clusters (Default_Ignorable, Control, pure nonspacing/enclosing Mark, lone Surrogates). Tabs are ignored by design. 2. RGI emoji clusters (\p{RGI_Emoji}) are double-width. - 3. Otherwise use East Asian Width of the cluster’s first visible code point, and add widths for trailing Halfwidth/Fullwidth Forms within the same cluster (e.g., dakuten/handakuten/prolonged sound mark). + 3. Minimally-qualified/unqualified emoji clusters (ZWJ sequences with 2+ Extended_Pictographic, or keycap sequences) are double-width. + 4. Hangul jamo collapse each standard modern Hangul L+V or L+V+T syllable piece to width 2. + Unmatched repeated leading/vowel/trailing jamo stay additive because that matches how the terminals we target render them. + 5. Otherwise use East Asian Width of the cluster's first visible code point, and add widths for trailing spacing marks and Halfwidth/Fullwidth Forms within the same cluster (e.g., dakuten/handakuten/prolonged sound mark). */ const segmenter = new Intl.Segmenter(); // Whole-cluster zero-width -const zeroWidthClusterRegex = /^(?:\p{Default_Ignorable_Code_Point}|\p{Control}|\p{Mark}|\p{Surrogate})+$/v; +const zeroWidthClusterRegex = /^(?:\p{Default_Ignorable_Code_Point}|\p{Control}|\p{Format}|\p{Nonspacing_Mark}|\p{Enclosing_Mark}|\p{Surrogate})+$/v; // Pick the base scalar if the cluster starts with Prepend/Format/Marks -const leadingNonPrintingRegex = /^[\p{Default_Ignorable_Code_Point}\p{Control}\p{Format}\p{Mark}\p{Surrogate}]+/v; +const leadingNonPrintingRegex = /^[\p{Default_Ignorable_Code_Point}\p{Control}\p{Format}\p{Nonspacing_Mark}\p{Enclosing_Mark}\p{Surrogate}]+/v; +const spacingMarkRegex = /\p{Spacing_Mark}/v; // RGI emoji sequences const rgiEmojiRegex = /^\p{RGI_Emoji}$/v; +// Detect minimally-qualified/unqualified emoji sequences (missing VS16 but still render as double-width) +const unqualifiedKeycapRegex = /^[\d#*]\u20E3$/; +const extendedPictographicRegex = /\p{Extended_Pictographic}/gu; + +function isDoubleWidthNonRgiEmojiSequence(segment) { + // Real emoji clusters are < 30 chars; guard against pathological input + if (segment.length > 50) { + return false; + } + + if (unqualifiedKeycapRegex.test(segment)) { + return true; + } + + // ZWJ sequences with 2+ Extended_Pictographic + if (segment.includes('\u200D')) { + const pictographics = segment.match(extendedPictographicRegex); + return pictographics !== null && pictographics.length >= 2; + } + + return false; +} + function baseVisible(segment) { return segment.replace(leadingNonPrintingRegex, ''); } @@ -29,13 +56,91 @@ function isZeroWidthCluster(segment) { return zeroWidthClusterRegex.test(segment); } -function trailingHalfwidthWidth(segment, eastAsianWidthOptions) { - let extra = 0; - if (segment.length > 1) { - for (const char of segment.slice(1)) { - if (char >= '\uFF00' && char <= '\uFFEF') { - extra += eastAsianWidth(char.codePointAt(0), eastAsianWidthOptions); +function isHangulLeadingJamo(codePoint) { + return (codePoint >= 0x11_00 && codePoint <= 0x11_5F) + || (codePoint >= 0xA9_60 && codePoint <= 0xA9_7C); +} + +function isHangulVowelJamo(codePoint) { + return (codePoint >= 0x11_60 && codePoint <= 0x11_A7) + || (codePoint >= 0xD7_B0 && codePoint <= 0xD7_C6); +} + +function isHangulTrailingJamo(codePoint) { + return (codePoint >= 0x11_A8 && codePoint <= 0x11_FF) + || (codePoint >= 0xD7_CB && codePoint <= 0xD7_FB); +} + +function isHangulJamo(codePoint) { + return isHangulLeadingJamo(codePoint) + || isHangulVowelJamo(codePoint) + || isHangulTrailingJamo(codePoint); +} + +function hangulClusterWidth(visibleSegment, eastAsianWidthOptions) { + const codePoints = []; + + for (const character of visibleSegment) { + if (zeroWidthClusterRegex.test(character)) { + continue; + } + + codePoints.push(character.codePointAt(0)); + } + + if (codePoints.length === 0) { + return undefined; + } + + let width = 0; + + for (let index = 0; index < codePoints.length; index++) { + const codePoint = codePoints[index]; + if (!isHangulJamo(codePoint)) { + if (width === 0) { + return undefined; + } + + // Mixed cluster (e.g., L + precomposed syllable): use EAW for non-jamo remainder + for (let remaining = index; remaining < codePoints.length; remaining++) { + width += eastAsianWidth(codePoints[remaining], eastAsianWidthOptions); } + + return width; + } + + // Modern Hangul L+V(+T) shapes as one syllable block. Unmatched jamo stay additive: + // U+1100 U+1100 U+1161 => U+1100 + (U+1100 U+1161) => 2 + 2. + if ( + isHangulLeadingJamo(codePoint) + && isHangulVowelJamo(codePoints[index + 1]) + ) { + width += 2; + index += isHangulTrailingJamo(codePoints[index + 2]) ? 2 : 1; + continue; + } + + width += eastAsianWidth(codePoint, eastAsianWidthOptions); + } + + return width; +} + +function trailingWidth(visibleSegment, eastAsianWidthOptions) { + let extra = 0; + let first = true; + + for (const character of visibleSegment) { + if (first) { + first = false; + continue; + } + + if ( + spacingMarkRegex.test(character) + || (character >= '\uFF00' && character <= '\uFFEF') + ) { + extra += eastAsianWidth(character.codePointAt(0), eastAsianWidthOptions); } } @@ -54,7 +159,8 @@ export default function stringWidth(input, options = {}) { let string = input; - if (!countAnsiEscapeCodes) { + // Avoid calling stripAnsi when there are no ANSI escape sequences (ESC = 0x1B, CSI = 0x9B) + if (!countAnsiEscapeCodes && (string.includes('\u001B') || string.includes('\u009B'))) { string = stripAnsi(string); } @@ -62,6 +168,11 @@ export default function stringWidth(input, options = {}) { return 0; } + // Fast path: printable ASCII (0x20–0x7E) needs no segmenter, regex, or EAW lookup — width equals length. + if (/^[\u0020-\u007E]*$/.test(string)) { + return string.length; + } + let width = 0; const eastAsianWidthOptions = {ambiguousAsWide: !ambiguousIsNarrow}; @@ -72,17 +183,24 @@ export default function stringWidth(input, options = {}) { } // Emoji width logic - if (rgiEmojiRegex.test(segment)) { + if (rgiEmojiRegex.test(segment) || isDoubleWidthNonRgiEmojiSequence(segment)) { width += 2; continue; } + const visibleSegment = baseVisible(segment); + const hangulWidth = hangulClusterWidth(visibleSegment, eastAsianWidthOptions); + if (hangulWidth !== undefined) { + width += hangulWidth; + continue; + } + // Everything else: EAW of the cluster’s first visible scalar - const codePoint = baseVisible(segment).codePointAt(0); + const codePoint = visibleSegment.codePointAt(0); width += eastAsianWidth(codePoint, eastAsianWidthOptions); - // Add width for trailing Halfwidth and Fullwidth Forms (e.g., ゙, ゚, ー) - width += trailingHalfwidthWidth(segment, eastAsianWidthOptions); + // Add width for trailing spacing marks and Halfwidth/Fullwidth Forms (e.g., ゙, ゚, ー) + width += trailingWidth(visibleSegment, eastAsianWidthOptions); } return width; diff --git a/package.json b/package.json index 21c7241..53b7ed0 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "string-width", - "version": "8.1.0", + "version": "8.2.2", "description": "Get the visual width of a string - the number of columns required to display it", "license": "MIT", "repository": "sindresorhus/string-width", @@ -54,12 +54,12 @@ "east-asian-width" ], "dependencies": { - "get-east-asian-width": "^1.3.0", - "strip-ansi": "^7.1.0" + "get-east-asian-width": "^1.5.0", + "strip-ansi": "^7.1.2" }, "devDependencies": { "ava": "^6.4.1", "tsd": "^0.33.0", - "xo": "^1.2.2" + "xo": "^1.2.3" } } diff --git a/readme.md b/readme.md index 1cccd00..095f67a 100644 --- a/readme.md +++ b/readme.md @@ -2,7 +2,7 @@ > Get the visual width of a string - the number of columns required to display it -Some Unicode characters are [fullwidth](https://en.wikipedia.org/wiki/Halfwidth_and_fullwidth_forms) and use double the normal width. [ANSI escape codes](https://en.wikipedia.org/wiki/ANSI_escape_code) are stripped and doesn't affect the width. +Some Unicode characters are [fullwidth](https://en.wikipedia.org/wiki/Halfwidth_and_fullwidth_forms) and use double the normal width. [ANSI escape codes](https://en.wikipedia.org/wiki/ANSI_escape_code) are stripped and do not affect the width. Useful to be able to measure the actual width of command-line output. diff --git a/test.js b/test.js index 2b90be5..c34131f 100644 --- a/test.js +++ b/test.js @@ -25,6 +25,32 @@ test('non-string input (undefined)', t => { test('full-width characters', macro, '你好', 4); test('half-width characters', macro, 'hello', 5); test('mixed width', macro, 'hello世界', 9); +test('repeated leading Hangul jamo in one grapheme cluster stay additive', macro, 'ᄀᄀ', 4); +test('long leading Hangul jamo cluster stays additive', macro, 'ᄀᄀᄀᄀᄀᄀ', 12); +test('decomposed Hangul syllable cluster', macro, '가', 2); +test('decomposed Hangul syllable cluster with trailing jamo', macro, '각', 2); +test('decomposed Hangul syllable cluster with variation selector', macro, '가\uFE0F', 2); +test('repeated leading jamo before vowel composes last pair', macro, 'ᄀ가', 4); +test('repeated vowel jamo after leading composes first pair', macro, '가ᅡ', 3); +test('repeated trailing jamo after syllable composes first three', macro, '각ᆨ', 3); +test('repeated leading Hangul jamo with variation selector stay additive', macro, 'ᄀᄀ\uFE0F', 4); +test('repeated leading Hangul jamo with trailing ZWJ stay additive', macro, 'ᄀᄀ\u200D', 4); +test('vowel-only Hangul jamo cluster stays additive', macro, 'ᅡᅡᅡ', 3); +test('trailing-only Hangul jamo cluster stays additive', macro, 'ᆨᆨ', 2); +test('single leading jamo alone is width 2', macro, 'ᄀ', 2); +test('single vowel jamo alone is width 1', macro, 'ᅡ', 1); +test('single trailing jamo alone is width 1', macro, 'ᆨ', 1); +test('vowel jamo before leading jamo does not compose', macro, 'ᅡᄀ', 3); +test('leading jamo followed by trailing jamo without vowel does not compose', macro, 'ᄀᆨ', 3); +test('extended leading jamo U+A960 + standard vowel composes to width 2', macro, 'ꥠᅡ', 2); +test('standard leading jamo + extended vowel U+D7B0 composes to width 2', macro, 'ᄀힰ', 2); +test('extended leading + standard vowel + standard trailing composes to width 2', macro, 'ꥠᅡᆨ', 2); +test('Hangul Compatibility Jamo is full-width via EAW', macro, 'ㄱ', 2); +test('two Hangul Compatibility Jamo do not compose', macro, 'ㄱㄱ', 4); +test('precomposed syllable 가 U+AC00', macro, '가', 2); +test('precomposed syllable 한 U+D55C', macro, '한', 2); +test('three precomposed syllables', macro, '한국어', 6); +test('leading jamo + precomposed syllable in one cluster', macro, 'ᄀ가', 4); // Halfwidth Katakana with dakuten/handakuten (issue #55) test('halfwidth kana voiced sound mark (ba)', macro, 'バ', 2); @@ -75,6 +101,11 @@ test('Indic conjunct via ZWJ', macro, 'क्\u200Dष', 1); test('combining diacritical mark', macro, 'e\u0301', 1); test('multiple combining marks', macro, 'e\u0301\u0302', 1); test('combining marks only', macro, '\u0301\u0302', 0); +test('Tibetan combining mark', macro, 'ཟླ', 1); +test('enclosing mark', macro, 'a\u20DD', 1); +test('spacing mark alone', macro, '\u093E', 1); +test('spacing mark after base character', macro, '\u0915\u093E', 2); +test('pre-base spacing mark after base character', macro, '\u0915\u093F', 2); // Surrogate pairs and high code points test('emoji surrogate pair', macro, '😀', 2); @@ -249,3 +280,60 @@ test('digit zero as plain text (not emoji)', macro, '0', 1); test('digit one as plain text', macro, '1', 1); test('asterisk as plain text', macro, '*', 1); test('hash as plain text', macro, '#', 1); + +// Unicode Format characters (non-default-ignorable) +test('Arabic number sign U+0600', macro, '\u0600', 0); +test('Arabic end of ayah U+06DD', macro, '\u06DD', 0); +test('Syriac abbreviation mark U+070F', macro, '\u070F', 0); + +// Minimally-qualified/unqualified emoji sequences +// These are emoji sequences missing VS16 but should still be width 2 +test('heart on fire (MQ)', macro, '\u2764\u200D\u{1F525}', 2); // ❤‍🔥 +test('rainbow flag (MQ)', macro, '\u{1F3F3}\u200D\u{1F308}', 2); // 🏳‍🌈 +test('transgender flag (MQ)', macro, '\u{1F3F3}\u200D\u26A7', 2); // 🏳‍⚧ +test('broken chain (MQ)', macro, '\u26D3\u200D\u{1F4A5}', 2); // ⛓‍💥 +test('eye in speech bubble (MQ)', macro, '\u{1F441}\u200D\u{1F5E8}', 2); // 👁‍🗨 +test('man bouncing ball (MQ)', macro, '\u26F9\u200D\u2642', 2); // ⛹‍♂ +test('woman bouncing ball (MQ)', macro, '\u26F9\u200D\u2640', 2); // ⛹‍♀ +test('man detective (MQ)', macro, '\u{1F575}\u200D\u2642', 2); // 🕵‍♂ +test('woman detective (MQ)', macro, '\u{1F575}\u200D\u2640', 2); // 🕵‍♀ + +// Unqualified keycap sequences (missing VS16) +test('keycap # (UQ)', macro, '#\u20E3', 2); // #⃣ +test('keycap 0 (UQ)', macro, '0\u20E3', 2); // 0⃣ +test('keycap * (UQ)', macro, '*\u20E3', 2); // *⃣ + +// Ensure invalid keycap sequences don't match +test('phone + keycap (invalid)', macro, '\u260E\uFE0F\u20E3', 1); // Not a valid keycap base + +// Latin1 range (0xA0–0x2FF) — width 1, no segmenter or EAW lookup needed +test('non-breaking space U+00A0', macro, '\u00A0', 1); +test('Latin ñ U+00F1', macro, 'ñ', 1); +test('Latin ü U+00FC', macro, 'ü', 1); +test('Latin ÿ U+00FF', macro, 'ÿ', 1); +test('Latin1 in text', macro, 'café', 4); +test('Spacing Modifier U+02FF', macro, '\u02FF', 1); + +// Soft hyphen (0xAD) — zero-width +test('soft hyphen U+00AD', macro, '\u00AD', 0); +test('soft hyphen in text', macro, 'a\u00ADb', 2); + +// Combining diacritical boundary (0x300) — zero-width combining mark +test('combining grave U+0300 alone', macro, '\u0300', 0); +test('char + combining at 0x300 boundary', macro, 'a\u0300', 1); + +// ASCII boundary characters +test('space U+0020 (lowest printable ASCII)', macro, ' ', 1); +test('tilde U+007E (highest printable ASCII)', macro, '~', 1); +test('DEL U+007F (just above printable ASCII)', macro, '\u007F', 0); +test('unit separator U+001F (just below printable ASCII)', macro, '\u001F', 0); + +// Ambiguous characters +test('ambiguous in text (narrow default)', macro, '±×÷', 3); +test('ambiguous in text (wide)', macro, '±×÷', 6, {ambiguousIsNarrow: false}); +test('ambiguous mixed with CJK', macro, '±你', 3); +test('ambiguous mixed with CJK (wide)', macro, '±你', 4, {ambiguousIsNarrow: false}); + +// `stripAnsi` guard: non-ANSI strings should not call `stripAnsi` +test('non-ASCII without ANSI escapes', macro, '你好世界', 8); +test('Latin1 without ANSI escapes', macro, 'résumé', 6);