{
  "version": "v1",
  "kind": "combining",
  "description": "Combining marks, NFC vs NFD, stacked diacritics, Zalgo-style overflow, Indic and Thai",
  "count": 17,
  "seed": null,
  "items": [
    {
      "id": "combining-01",
      "text": "é é",
      "note": "Precomposed vs decomposed e-acute (NFC vs NFD look identical)",
      "lang": null,
      "length": {
        "codepoints": 4,
        "utf16_units": 4,
        "utf8_bytes": 6
      },
      "codepoints": "U+00E9 U+0020 U+0065 U+0301"
    },
    {
      "id": "combining-02",
      "text": "Amélie Amélie",
      "note": "Same word in NFC and NFD: different bytes, equal after normalization",
      "lang": "fr",
      "length": {
        "codepoints": 14,
        "utf16_units": 14,
        "utf8_bytes": 16
      },
      "codepoints": "U+0041 U+006D U+00E9 U+006C U+0069 U+0065 U+0020 U+0041 U+006D U+0065 U+0301 U+006C U+0069 U+0065"
    },
    {
      "id": "combining-03",
      "text": "Việt Nam",
      "note": "Vietnamese precomposed multi-diacritic letters (NFC)",
      "lang": "vi",
      "length": {
        "codepoints": 8,
        "utf16_units": 8,
        "utf8_bytes": 10
      },
      "codepoints": "U+0056 U+0069 U+1EC7 U+0074 U+0020 U+004E U+0061 U+006D"
    },
    {
      "id": "combining-04",
      "text": "Việt Nam",
      "note": "Vietnamese with decomposed dot-below + circumflex (NFD)",
      "lang": "vi",
      "length": {
        "codepoints": 10,
        "utf16_units": 10,
        "utf8_bytes": 12
      },
      "codepoints": "U+0056 U+0069 U+0065 U+0323 U+0302 U+0074 U+0020 U+004E U+0061 U+006D"
    },
    {
      "id": "combining-05",
      "text": "한글 한글",
      "note": "Korean decomposed jamo vs precomposed syllables",
      "lang": "ko",
      "length": {
        "codepoints": 9,
        "utf16_units": 9,
        "utf8_bytes": 25
      },
      "codepoints": "U+1112 U+1161 U+11AB U+1100 U+1173 U+11AF U+0020 U+D55C U+AE00"
    },
    {
      "id": "combining-06",
      "text": "à̛̖̗̘̙̜̝̞̟́̂̃̄̆̇̈̉̊̋̌̍̎̏̐̑̒̓̔̚",
      "note": "30 combining marks on one letter (line-height overflow)",
      "lang": null,
      "length": {
        "codepoints": 31,
        "utf16_units": 31,
        "utf8_bytes": 61
      },
      "codepoints": "U+0061 U+0300 U+0301 U+0302 U+0303 U+0304 U+0306 U+0307 U+0308 U+0309 U+030A U+030B U+030C U+030D U+030E U+030F U+0310 U+0311 U+0312 U+0313 U+0314 U+0316 U+0317 U+0318 U+0319 U+031A U+031B U+031C U+031D U+031E U+031F"
    },
    {
      "id": "combining-07",
      "text": "s̶t̶r̶i̶k̶e̶t̶h̶r̶o̶u̶g̶h̶",
      "note": "Combining long stroke overlay (U+0336) on every letter",
      "lang": "en",
      "length": {
        "codepoints": 26,
        "utf16_units": 26,
        "utf8_bytes": 39
      },
      "codepoints": "U+0073 U+0336 U+0074 U+0336 U+0072 U+0336 U+0069 U+0336 U+006B U+0336 U+0065 U+0336 U+0074 U+0336 U+0068 U+0336 U+0072 U+0336 U+006F U+0336 U+0075 U+0336 U+0067 U+0336 U+0068 U+0336"
    },
    {
      "id": "combining-08",
      "text": "u̲n̲d̲e̲r̲l̲i̲n̲e̲d̲",
      "note": "Combining low line (U+0332) on every letter",
      "lang": "en",
      "length": {
        "codepoints": 20,
        "utf16_units": 20,
        "utf8_bytes": 30
      },
      "codepoints": "U+0075 U+0332 U+006E U+0332 U+0064 U+0332 U+0065 U+0332 U+0072 U+0332 U+006C U+0332 U+0069 U+0332 U+006E U+0332 U+0065 U+0332 U+0064 U+0332"
    },
    {
      "id": "combining-09",
      "text": "A⃝ B⃞",
      "note": "Combining enclosing circle and square",
      "lang": null,
      "length": {
        "codepoints": 5,
        "utf16_units": 5,
        "utf8_bytes": 9
      },
      "codepoints": "U+0041 U+20DD U+0020 U+0042 U+20DE"
    },
    {
      "id": "combining-10",
      "text": "क्षत्रिय नमस्ते",
      "note": "Devanagari conjuncts and virama",
      "lang": "hi",
      "length": {
        "codepoints": 15,
        "utf16_units": 15,
        "utf8_bytes": 43
      },
      "codepoints": "U+0915 U+094D U+0937 U+0924 U+094D U+0930 U+093F U+092F U+0020 U+0928 U+092E U+0938 U+094D U+0924 U+0947"
    },
    {
      "id": "combining-11",
      "text": "தமிழ்நாடு",
      "note": "Tamil with vowel signs and pulli",
      "lang": "ta",
      "length": {
        "codepoints": 9,
        "utf16_units": 9,
        "utf8_bytes": 27
      },
      "codepoints": "U+0BA4 U+0BAE U+0BBF U+0BB4 U+0BCD U+0BA8 U+0BBE U+0B9F U+0BC1"
    },
    {
      "id": "combining-12",
      "text": "น้ำแข็ง ผู้ใหญ่",
      "note": "Thai stacked tone marks",
      "lang": "th",
      "length": {
        "codepoints": 15,
        "utf16_units": 15,
        "utf8_bytes": 43
      },
      "codepoints": "U+0E19 U+0E49 U+0E33 U+0E41 U+0E02 U+0E47 U+0E07 U+0020 U+0E1C U+0E39 U+0E49 U+0E43 U+0E2B U+0E0D U+0E48"
    },
    {
      "id": "combining-13",
      "text": "שָׁלוֹם",
      "note": "Hebrew with niqqud (vowel points)",
      "lang": "he",
      "length": {
        "codepoints": 7,
        "utf16_units": 7,
        "utf8_bytes": 14
      },
      "codepoints": "U+05E9 U+05C1 U+05B8 U+05DC U+05D5 U+05B9 U+05DD"
    },
    {
      "id": "combining-14",
      "text": "كِتَابٌ",
      "note": "Arabic with harakat (short-vowel marks)",
      "lang": "ar",
      "length": {
        "codepoints": 7,
        "utf16_units": 7,
        "utf8_bytes": 14
      },
      "codepoints": "U+0643 U+0650 U+062A U+064E U+0627 U+0628 U+064C"
    },
    {
      "id": "combining-15",
      "text": "ﬁle ﬀ Å Ω",
      "note": "Compatibility characters (fi/ff ligatures, Angstrom and Ohm signs): NFKC changes them",
      "lang": null,
      "length": {
        "codepoints": 9,
        "utf16_units": 9,
        "utf8_bytes": 17
      },
      "codepoints": "U+FB01 U+006C U+0065 U+0020 U+FB00 U+0020 U+212B U+0020 U+2126"
    },
    {
      "id": "combining-16",
      "text": "H̆ȩ̶̏l͕ͩ͟l̚͜o̓͒ w̢̗ͅòr̶͟l̮d̷̀͟",
      "note": "Zalgo-lite: 1-3 random combining marks per letter",
      "lang": "en",
      "length": {
        "codepoints": 32,
        "utf16_units": 32,
        "utf8_bytes": 53
      },
      "codepoints": "U+0048 U+0306 U+0065 U+0336 U+030F U+0327 U+006C U+0355 U+0369 U+035F U+006C U+035C U+031A U+006F U+0313 U+0352 U+0020 U+0077 U+0345 U+0322 U+0317 U+006F U+0340 U+0072 U+0336 U+035F U+006C U+032E U+0064 U+0337 U+0300 U+035F"
    },
    {
      "id": "combining-17",
      "text": "P̵̶̨ͬ́ͬĺ̬͉͙̌̓̉̍ą̥̝ͤͭc̛͕̩̀e͚̞͑̏ḩ̤̥̹ͥͤ̒̌ȯ̶̶͖̝̈͆͟l̮̜̉̂̃̃͐̚d̯̬̿ͤ͋͊̿̕ḛ͙̣͝r̵̲̯̈́̓",
      "note": "Zalgo: 4-8 random combining marks per letter (overflows line height)",
      "lang": "en",
      "length": {
        "codepoints": 79,
        "utf16_units": 79,
        "utf8_bytes": 147
      },
      "codepoints": null
    }
  ]
}