{
  "version": "v1",
  "count": null,
  "seed": null,
  "kinds": {
    "long-words": {
      "description": "Long compound words, unbroken strings, long URLs, e-mails and paths",
      "items": [
        {
          "id": "long-words-01",
          "text": "Donaudampfschifffahrtsgesellschaftskapitän",
          "note": "German compound noun",
          "lang": "de",
          "length": {
            "codepoints": 42,
            "utf16_units": 42,
            "utf8_bytes": 43
          },
          "codepoints": null
        },
        {
          "id": "long-words-02",
          "text": "Rindfleischetikettierungsüberwachungsaufgabenübertragungsgesetz",
          "note": "German compound noun (former law name)",
          "lang": "de",
          "length": {
            "codepoints": 63,
            "utf16_units": 63,
            "utf8_bytes": 65
          },
          "codepoints": null
        },
        {
          "id": "long-words-03",
          "text": "Grundstücksverkehrsgenehmigungszuständigkeitsübertragungsverordnung",
          "note": "German compound noun",
          "lang": "de",
          "length": {
            "codepoints": 67,
            "utf16_units": 67,
            "utf8_bytes": 70
          },
          "codepoints": null
        },
        {
          "id": "long-words-04",
          "text": "Kraftfahrzeug-Haftpflichtversicherung",
          "note": "German compound with hyphen (break opportunity)",
          "lang": "de",
          "length": {
            "codepoints": 37,
            "utf16_units": 37,
            "utf8_bytes": 37
          },
          "codepoints": "U+004B U+0072 U+0061 U+0066 U+0074 U+0066 U+0061 U+0068 U+0072 U+007A U+0065 U+0075 U+0067 U+002D U+0048 U+0061 U+0066 U+0074 U+0070 U+0066 U+006C U+0069 U+0063 U+0068 U+0074 U+0076 U+0065 U+0072 U+0073 U+0069 U+0063 U+0068 U+0065 U+0072 U+0075 U+006E U+0067"
        },
        {
          "id": "long-words-05",
          "text": "Donau­dampf­schiff­fahrts­gesell­schafts­kapitän",
          "note": "German compound with soft hyphens (U+00AD): should hyphenate, not overflow",
          "lang": "de",
          "length": {
            "codepoints": 48,
            "utf16_units": 48,
            "utf8_bytes": 55
          },
          "codepoints": null
        },
        {
          "id": "long-words-06",
          "text": "lentokonesuihkuturbiinimoottoriapumekaanikkoaliupseerioppilas",
          "note": "Finnish compound noun",
          "lang": "fi",
          "length": {
            "codepoints": 61,
            "utf16_units": 61,
            "utf8_bytes": 61
          },
          "codepoints": null
        },
        {
          "id": "long-words-07",
          "text": "epäjärjestelmällistyttämättömyydellänsäkäänköhän",
          "note": "Finnish heavily inflected word",
          "lang": "fi",
          "length": {
            "codepoints": 48,
            "utf16_units": 48,
            "utf8_bytes": 60
          },
          "codepoints": null
        },
        {
          "id": "long-words-08",
          "text": "kindercarnavalsoptochtvoorbereidingswerkzaamhedenplan",
          "note": "Dutch compound noun",
          "lang": "nl",
          "length": {
            "codepoints": 53,
            "utf16_units": 53,
            "utf8_bytes": 53
          },
          "codepoints": null
        },
        {
          "id": "long-words-09",
          "text": "arbeidsongeschiktheidsverzekering",
          "note": "Dutch compound noun",
          "lang": "nl",
          "length": {
            "codepoints": 33,
            "utf16_units": 33,
            "utf8_bytes": 33
          },
          "codepoints": "U+0061 U+0072 U+0062 U+0065 U+0069 U+0064 U+0073 U+006F U+006E U+0067 U+0065 U+0073 U+0063 U+0068 U+0069 U+006B U+0074 U+0068 U+0065 U+0069 U+0064 U+0073 U+0076 U+0065 U+0072 U+007A U+0065 U+006B U+0065 U+0072 U+0069 U+006E U+0067"
        },
        {
          "id": "long-words-10",
          "text": "Nordöstersjökustartilleriflygspaningssimulatoranläggningsmaterielunderhållsuppföljningssystemdiskussionsinläggsförberedelsearbeten",
          "note": "Swedish compound noun (constructed)",
          "lang": "sv",
          "length": {
            "codepoints": 130,
            "utf16_units": 130,
            "utf8_bytes": 137
          },
          "codepoints": null
        },
        {
          "id": "long-words-11",
          "text": "realisationsvinstbeskattning",
          "note": "Swedish compound noun",
          "lang": "sv",
          "length": {
            "codepoints": 28,
            "utf16_units": 28,
            "utf8_bytes": 28
          },
          "codepoints": "U+0072 U+0065 U+0061 U+006C U+0069 U+0073 U+0061 U+0074 U+0069 U+006F U+006E U+0073 U+0076 U+0069 U+006E U+0073 U+0074 U+0062 U+0065 U+0073 U+006B U+0061 U+0074 U+0074 U+006E U+0069 U+006E U+0067"
        },
        {
          "id": "long-words-12",
          "text": "menneskerettighetsorganisasjonene",
          "note": "Norwegian compound noun",
          "lang": "nb",
          "length": {
            "codepoints": 33,
            "utf16_units": 33,
            "utf8_bytes": 33
          },
          "codepoints": "U+006D U+0065 U+006E U+006E U+0065 U+0073 U+006B U+0065 U+0072 U+0065 U+0074 U+0074 U+0069 U+0067 U+0068 U+0065 U+0074 U+0073 U+006F U+0072 U+0067 U+0061 U+006E U+0069 U+0073 U+0061 U+0073 U+006A U+006F U+006E U+0065 U+006E U+0065"
        },
        {
          "id": "long-words-13",
          "text": "speciallægepraksisplanlægningsstabiliseringsperiode",
          "note": "Danish compound noun",
          "lang": "da",
          "length": {
            "codepoints": 51,
            "utf16_units": 51,
            "utf8_bytes": 53
          },
          "codepoints": null
        },
        {
          "id": "long-words-14",
          "text": "Llanfairpwllgwyngyllgogerychwyndrobwllllantysiliogogogoch",
          "note": "Welsh place name",
          "lang": "cy",
          "length": {
            "codepoints": 57,
            "utf16_units": 57,
            "utf8_bytes": 57
          },
          "codepoints": null
        },
        {
          "id": "long-words-15",
          "text": "Konstantynopolitańczykowianeczka",
          "note": "Polish long word",
          "lang": "pl",
          "length": {
            "codepoints": 32,
            "utf16_units": 32,
            "utf8_bytes": 33
          },
          "codepoints": "U+004B U+006F U+006E U+0073 U+0074 U+0061 U+006E U+0074 U+0079 U+006E U+006F U+0070 U+006F U+006C U+0069 U+0074 U+0061 U+0144 U+0063 U+007A U+0079 U+006B U+006F U+0077 U+0069 U+0061 U+006E U+0065 U+0063 U+007A U+006B U+0061"
        },
        {
          "id": "long-words-16",
          "text": "превысокомногорассмотрительствующий",
          "note": "Russian long word (Cyrillic)",
          "lang": "ru",
          "length": {
            "codepoints": 35,
            "utf16_units": 35,
            "utf8_bytes": 70
          },
          "codepoints": "U+043F U+0440 U+0435 U+0432 U+044B U+0441 U+043E U+043A U+043E U+043C U+043D U+043E U+0433 U+043E U+0440 U+0430 U+0441 U+0441 U+043C U+043E U+0442 U+0440 U+0438 U+0442 U+0435 U+043B U+044C U+0441 U+0442 U+0432 U+0443 U+044E U+0449 U+0438 U+0439"
        },
        {
          "id": "long-words-17",
          "text": "pneumonoultramicroscopicsilicovolcanoconiosis",
          "note": "English long word",
          "lang": "en",
          "length": {
            "codepoints": 45,
            "utf16_units": 45,
            "utf8_bytes": 45
          },
          "codepoints": null
        },
        {
          "id": "long-words-18",
          "text": "การทดสอบการตัดคำภาษาไทยซึ่งไม่มีการเว้นวรรคระหว่างคำ",
          "note": "Thai sentence with no spaces between words: needs dictionary line breaking",
          "lang": "th",
          "length": {
            "codepoints": 52,
            "utf16_units": 52,
            "utf8_bytes": 156
          },
          "codepoints": null
        },
        {
          "id": "long-words-19",
          "text": "WWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWWW",
          "note": "64 x W, the widest Latin glyph in most fonts",
          "lang": null,
          "length": {
            "codepoints": 64,
            "utf16_units": 64,
            "utf8_bytes": 64
          },
          "codepoints": null
        },
        {
          "id": "long-words-20",
          "text": "xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx",
          "note": "255 characters, a common VARCHAR limit",
          "lang": null,
          "length": {
            "codepoints": 255,
            "utf16_units": 255,
            "utf8_bytes": 255
          },
          "codepoints": null
        },
        {
          "id": "long-words-21",
          "text": "xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx",
          "note": "256 characters, one over a common VARCHAR limit",
          "lang": null,
          "length": {
            "codepoints": 256,
            "utf16_units": 256,
            "utf8_bytes": 256
          },
          "codepoints": null
        },
        {
          "id": "long-words-22",
          "text": "a373ff095476d699923556a5def23f09b3963569b34117aad41ad37a06f2a3d23665911f7ef25b78e315336f36d170a5435f037634b966143bd725f4590a8667",
          "note": "128-character hex string (no break opportunities)",
          "lang": null,
          "length": {
            "codepoints": 128,
            "utf16_units": 128,
            "utf8_bytes": 128
          },
          "codepoints": null
        },
        {
          "id": "long-words-23",
          "text": "LNka48WXhrEOv8IVJ1IHYxO5ulQ9ES+fFDwvYVA8Vip2pZ3KkTEkudapKeyO30CqnsT4alKPSEPToCzATZC0vyzZGuPFl4axDr/CFSdSB2MTubpUPREvnxQ8L2FQPFYqdqWdypExJLnWqSnsjt9Aqp7E+GpSj0hD06AswE2QtL8=",
          "note": "Long base64 token (contains + and /)",
          "lang": null,
          "length": {
            "codepoints": 172,
            "utf16_units": 172,
            "utf8_bytes": 172
          },
          "codepoints": null
        },
        {
          "id": "long-words-24",
          "text": "https://www.example.com/products/category/subcategory/a-very-long-product-name-that-keeps-going-and-going?utm_source=newsletter&utm_medium=email&utm_campaign=autumn-sale-2026&redirect=https%3A%2F%2Fexample.org%2Fcheckout%3Fitem%3D12345%26qty%3D2#section-with-a-long-anchor-name",
          "note": "Long URL with query string and encoded redirect",
          "lang": null,
          "length": {
            "codepoints": 277,
            "utf16_units": 277,
            "utf8_bytes": 277
          },
          "codepoints": null
        },
        {
          "id": "long-words-25",
          "text": "firstname.middlename.lastname.department.team@subdomain.mail.example.com",
          "note": "Long e-mail address",
          "lang": null,
          "length": {
            "codepoints": 72,
            "utf16_units": 72,
            "utf8_bytes": 72
          },
          "codepoints": null
        },
        {
          "id": "long-words-26",
          "text": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa@bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb.example.com",
          "note": "E-mail with a 64-character local part and a 63-character label (RFC 5321 maximums)",
          "lang": null,
          "length": {
            "codepoints": 140,
            "utf16_units": 140,
            "utf8_bytes": 140
          },
          "codepoints": null
        },
        {
          "id": "long-words-27",
          "text": "Quarterly_Report_FINAL_v2_reviewed_by_legal_and_finance_team_2026-10-08 (copy 3).pdf",
          "note": "Long file name with spaces and parentheses",
          "lang": null,
          "length": {
            "codepoints": 84,
            "utf16_units": 84,
            "utf8_bytes": 84
          },
          "codepoints": null
        },
        {
          "id": "long-words-28",
          "text": "/var/lib/application/data/users/00000000-0000-0000-0000-000000000000/uploads/2026/10/08/original/image.jpeg",
          "note": "Long path with UUID",
          "lang": null,
          "length": {
            "codepoints": 107,
            "utf16_units": 107,
            "utf8_bytes": 107
          },
          "codepoints": null
        }
      ]
    },
    "rtl": {
      "description": "Right-to-left scripts, mixed bidirectional text and bidi controls",
      "items": [
        {
          "id": "rtl-01",
          "text": "مرحبًا بك في موقعنا الجديد.",
          "note": "Arabic sentence",
          "lang": "ar",
          "length": {
            "codepoints": 27,
            "utf16_units": 27,
            "utf8_bytes": 49
          },
          "codepoints": "U+0645 U+0631 U+062D U+0628 U+064B U+0627 U+0020 U+0628 U+0643 U+0020 U+0641 U+064A U+0020 U+0645 U+0648 U+0642 U+0639 U+0646 U+0627 U+0020 U+0627 U+0644 U+062C U+062F U+064A U+062F U+002E"
        },
        {
          "id": "rtl-02",
          "text": "نعمل كل يوم لنقدم لك أفضل المنتجات وأسرع خدمة توصيل.",
          "note": "Arabic sentence",
          "lang": "ar",
          "length": {
            "codepoints": 52,
            "utf16_units": 52,
            "utf8_bytes": 94
          },
          "codepoints": null
        },
        {
          "id": "rtl-03",
          "text": "ברוכים הבאים לאתר החדש שלנו.",
          "note": "Hebrew sentence",
          "lang": "he",
          "length": {
            "codepoints": 28,
            "utf16_units": 28,
            "utf8_bytes": 51
          },
          "codepoints": "U+05D1 U+05E8 U+05D5 U+05DB U+05D9 U+05DD U+0020 U+05D4 U+05D1 U+05D0 U+05D9 U+05DD U+0020 U+05DC U+05D0 U+05EA U+05E8 U+0020 U+05D4 U+05D7 U+05D3 U+05E9 U+0020 U+05E9 U+05DC U+05E0 U+05D5 U+002E"
        },
        {
          "id": "rtl-04",
          "text": "به وب‌سایت جدید ما خوش آمدید.",
          "note": "Persian sentence with ZWNJ (U+200C) inside a word",
          "lang": "fa",
          "length": {
            "codepoints": 29,
            "utf16_units": 29,
            "utf8_bytes": 53
          },
          "codepoints": "U+0628 U+0647 U+0020 U+0648 U+0628 U+200C U+0633 U+0627 U+06CC U+062A U+0020 U+062C U+062F U+06CC U+062F U+0020 U+0645 U+0627 U+0020 U+062E U+0648 U+0634 U+0020 U+0622 U+0645 U+062F U+06CC U+062F U+002E"
        },
        {
          "id": "rtl-05",
          "text": "ہماری نئی ویب سائٹ میں خوش آمدید۔",
          "note": "Urdu sentence (Arabic script, Urdu full stop U+06D4)",
          "lang": "ur",
          "length": {
            "codepoints": 33,
            "utf16_units": 33,
            "utf8_bytes": 60
          },
          "codepoints": "U+06C1 U+0645 U+0627 U+0631 U+06CC U+0020 U+0646 U+0626 U+06CC U+0020 U+0648 U+06CC U+0628 U+0020 U+0633 U+0627 U+0626 U+0679 U+0020 U+0645 U+06CC U+06BA U+0020 U+062E U+0648 U+0634 U+0020 U+0622 U+0645 U+062F U+06CC U+062F U+06D4"
        },
        {
          "id": "rtl-06",
          "text": "رقم الطلب 12345 جاهز للشحن.",
          "note": "Arabic with Western digits",
          "lang": "ar",
          "length": {
            "codepoints": 27,
            "utf16_units": 27,
            "utf8_bytes": 44
          },
          "codepoints": "U+0631 U+0642 U+0645 U+0020 U+0627 U+0644 U+0637 U+0644 U+0628 U+0020 U+0031 U+0032 U+0033 U+0034 U+0035 U+0020 U+062C U+0627 U+0647 U+0632 U+0020 U+0644 U+0644 U+0634 U+062D U+0646 U+002E"
        },
        {
          "id": "rtl-07",
          "text": "הזמנה מס׳ 4711 נשלחה אל example.com",
          "note": "Hebrew with digits and a Latin domain",
          "lang": "he",
          "length": {
            "codepoints": 35,
            "utf16_units": 35,
            "utf8_bytes": 50
          },
          "codepoints": "U+05D4 U+05D6 U+05DE U+05E0 U+05D4 U+0020 U+05DE U+05E1 U+05F3 U+0020 U+0034 U+0037 U+0031 U+0031 U+0020 U+05E0 U+05E9 U+05DC U+05D7 U+05D4 U+0020 U+05D0 U+05DC U+0020 U+0065 U+0078 U+0061 U+006D U+0070 U+006C U+0065 U+002E U+0063 U+006F U+006D"
        },
        {
          "id": "rtl-08",
          "text": "Version 2.0 של המוצר זמינה עכשיו",
          "note": "Hebrew sentence that starts with Latin text (first-strong detection)",
          "lang": "he",
          "length": {
            "codepoints": 32,
            "utf16_units": 32,
            "utf8_bytes": 49
          },
          "codepoints": "U+0056 U+0065 U+0072 U+0073 U+0069 U+006F U+006E U+0020 U+0032 U+002E U+0030 U+0020 U+05E9 U+05DC U+0020 U+05D4 U+05DE U+05D5 U+05E6 U+05E8 U+0020 U+05D6 U+05DE U+05D9 U+05E0 U+05D4 U+0020 U+05E2 U+05DB U+05E9 U+05D9 U+05D5"
        },
        {
          "id": "rtl-09",
          "text": "استخدم الرمز ABC-123 لتسجيل الدخول.",
          "note": "Arabic with an embedded Latin code",
          "lang": "ar",
          "length": {
            "codepoints": 35,
            "utf16_units": 35,
            "utf8_bytes": 58
          },
          "codepoints": "U+0627 U+0633 U+062A U+062E U+062F U+0645 U+0020 U+0627 U+0644 U+0631 U+0645 U+0632 U+0020 U+0041 U+0042 U+0043 U+002D U+0031 U+0032 U+0033 U+0020 U+0644 U+062A U+0633 U+062C U+064A U+0644 U+0020 U+0627 U+0644 U+062F U+062E U+0648 U+0644 U+002E"
        },
        {
          "id": "rtl-10",
          "text": "البريد الإلكتروني: support@example.com",
          "note": "Arabic label followed by an e-mail address",
          "lang": "ar",
          "length": {
            "codepoints": 38,
            "utf16_units": 38,
            "utf8_bytes": 54
          },
          "codepoints": "U+0627 U+0644 U+0628 U+0631 U+064A U+062F U+0020 U+0627 U+0644 U+0625 U+0644 U+0643 U+062A U+0631 U+0648 U+0646 U+064A U+003A U+0020 U+0073 U+0075 U+0070 U+0070 U+006F U+0072 U+0074 U+0040 U+0065 U+0078 U+0061 U+006D U+0070 U+006C U+0065 U+002E U+0063 U+006F U+006D"
        },
        {
          "id": "rtl-11",
          "text": "התקשרו אלינו: +1 555 0100",
          "note": "Hebrew with a phone number",
          "lang": "he",
          "length": {
            "codepoints": 25,
            "utf16_units": 25,
            "utf8_bytes": 36
          },
          "codepoints": "U+05D4 U+05EA U+05E7 U+05E9 U+05E8 U+05D5 U+0020 U+05D0 U+05DC U+05D9 U+05E0 U+05D5 U+003A U+0020 U+002B U+0031 U+0020 U+0035 U+0035 U+0035 U+0020 U+0030 U+0031 U+0030 U+0030"
        },
        {
          "id": "rtl-12",
          "text": "السعر: 1,234.50 ر.س",
          "note": "Arabic price with Western digits and a currency abbreviation",
          "lang": "ar",
          "length": {
            "codepoints": 19,
            "utf16_units": 19,
            "utf8_bytes": 26
          },
          "codepoints": "U+0627 U+0644 U+0633 U+0639 U+0631 U+003A U+0020 U+0031 U+002C U+0032 U+0033 U+0034 U+002E U+0035 U+0030 U+0020 U+0631 U+002E U+0633"
        },
        {
          "id": "rtl-13",
          "text": "שלום, עולם!",
          "note": "Hebrew with trailing punctuation (should render at the left end)",
          "lang": "he",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 19
          },
          "codepoints": "U+05E9 U+05DC U+05D5 U+05DD U+002C U+0020 U+05E2 U+05D5 U+05DC U+05DD U+0021"
        },
        {
          "id": "rtl-14",
          "text": "(مرحبا) [عالم] {نص} <وسم>",
          "note": "Arabic in brackets: brackets must mirror",
          "lang": "ar",
          "length": {
            "codepoints": 25,
            "utf16_units": 25,
            "utf8_bytes": 39
          },
          "codepoints": "U+0028 U+0645 U+0631 U+062D U+0628 U+0627 U+0029 U+0020 U+005B U+0639 U+0627 U+0644 U+0645 U+005D U+0020 U+007B U+0646 U+0635 U+007D U+0020 U+003C U+0648 U+0633 U+0645 U+003E"
        },
        {
          "id": "rtl-15",
          "text": "٠١٢٣٤٥٦٧٨٩",
          "note": "Arabic-Indic digits U+0660-U+0669",
          "lang": "ar",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 20
          },
          "codepoints": "U+0660 U+0661 U+0662 U+0663 U+0664 U+0665 U+0666 U+0667 U+0668 U+0669"
        },
        {
          "id": "rtl-16",
          "text": "۰۱۲۳۴۵۶۷۸۹",
          "note": "Extended Arabic-Indic (Persian) digits U+06F0-U+06F9",
          "lang": "fa",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 20
          },
          "codepoints": "U+06F0 U+06F1 U+06F2 U+06F3 U+06F4 U+06F5 U+06F6 U+06F7 U+06F8 U+06F9"
        },
        {
          "id": "rtl-17",
          "text": "مـــرحـــبـــا",
          "note": "Arabic with tatweel / kashida (U+0640) justification",
          "lang": "ar",
          "length": {
            "codepoints": 14,
            "utf16_units": 14,
            "utf8_bytes": 28
          },
          "codepoints": "U+0645 U+0640 U+0640 U+0640 U+0631 U+062D U+0640 U+0640 U+0640 U+0628 U+0640 U+0640 U+0640 U+0627"
        },
        {
          "id": "rtl-18",
          "text": "C++‎ היא שפת תכנות",
          "note": "Hebrew with LRM (U+200E) after 'C++' so the plus signs stay put",
          "lang": "he",
          "length": {
            "codepoints": 18,
            "utf16_units": 18,
            "utf8_bytes": 31
          },
          "codepoints": "U+0043 U+002B U+002B U+200E U+0020 U+05D4 U+05D9 U+05D0 U+0020 U+05E9 U+05E4 U+05EA U+0020 U+05EA U+05DB U+05E0 U+05D5 U+05EA"
        },
        {
          "id": "rtl-19",
          "text": "C++ היא שפת תכנות",
          "note": "Same without LRM: '++' usually jumps to the wrong side",
          "lang": "he",
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 28
          },
          "codepoints": "U+0043 U+002B U+002B U+0020 U+05D4 U+05D9 U+05D0 U+0020 U+05E9 U+05E4 U+05EA U+0020 U+05EA U+05DB U+05E0 U+05D5 U+05EA"
        },
        {
          "id": "rtl-20",
          "text": "Name: ⁧עברית⁩ (verified)",
          "note": "RLI...PDI isolate (U+2067/U+2069) inside English text",
          "lang": "en",
          "length": {
            "codepoints": 24,
            "utf16_units": 24,
            "utf8_bytes": 33
          },
          "codepoints": "U+004E U+0061 U+006D U+0065 U+003A U+0020 U+2067 U+05E2 U+05D1 U+05E8 U+05D9 U+05EA U+2069 U+0020 U+0028 U+0076 U+0065 U+0072 U+0069 U+0066 U+0069 U+0065 U+0064 U+0029"
        },
        {
          "id": "rtl-21",
          "text": "‫embedding that is never closed",
          "note": "RLE (U+202B) without PDF: direction leaks into what follows",
          "lang": null,
          "length": {
            "codepoints": 31,
            "utf16_units": 31,
            "utf8_bytes": 33
          },
          "codepoints": "U+202B U+0065 U+006D U+0062 U+0065 U+0064 U+0064 U+0069 U+006E U+0067 U+0020 U+0074 U+0068 U+0061 U+0074 U+0020 U+0069 U+0073 U+0020 U+006E U+0065 U+0076 U+0065 U+0072 U+0020 U+0063 U+006C U+006F U+0073 U+0065 U+0064"
        },
        {
          "id": "rtl-22",
          "text": "invoice‮txt.exe",
          "note": "RLO (U+202E) spoof: displays as 'invoiceexe.txt' (file-name spoofing)",
          "lang": null,
          "length": {
            "codepoints": 15,
            "utf16_units": 15,
            "utf8_bytes": 17
          },
          "codepoints": "U+0069 U+006E U+0076 U+006F U+0069 U+0063 U+0065 U+202E U+0074 U+0078 U+0074 U+002E U+0065 U+0078 U+0065"
        },
        {
          "id": "rtl-23",
          "text": "עברית English العربية 123 עוד",
          "note": "Mixed Hebrew, English, Arabic and digits in one line",
          "lang": "he",
          "length": {
            "codepoints": 29,
            "utf16_units": 29,
            "utf8_bytes": 44
          },
          "codepoints": "U+05E2 U+05D1 U+05E8 U+05D9 U+05EA U+0020 U+0045 U+006E U+0067 U+006C U+0069 U+0073 U+0068 U+0020 U+0627 U+0644 U+0639 U+0631 U+0628 U+064A U+0629 U+0020 U+0031 U+0032 U+0033 U+0020 U+05E2 U+05D5 U+05D3"
        }
      ]
    },
    "cjk": {
      "description": "Chinese, Japanese (kanji, kana, half-width) and Korean, full-width forms, astral ideographs",
      "items": [
        {
          "id": "cjk-01",
          "text": "欢迎访问我们的新网站。我们为您提供优质的产品和服务。",
          "note": "Chinese (Simplified)",
          "lang": "zh-Hans",
          "length": {
            "codepoints": 26,
            "utf16_units": 26,
            "utf8_bytes": 78
          },
          "codepoints": "U+6B22 U+8FCE U+8BBF U+95EE U+6211 U+4EEC U+7684 U+65B0 U+7F51 U+7AD9 U+3002 U+6211 U+4EEC U+4E3A U+60A8 U+63D0 U+4F9B U+4F18 U+8D28 U+7684 U+4EA7 U+54C1 U+548C U+670D U+52A1 U+3002"
        },
        {
          "id": "cjk-02",
          "text": "歡迎光臨我們的新網站。我們為您提供優質的產品與服務。",
          "note": "Chinese (Traditional)",
          "lang": "zh-Hant",
          "length": {
            "codepoints": 26,
            "utf16_units": 26,
            "utf8_bytes": 78
          },
          "codepoints": "U+6B61 U+8FCE U+5149 U+81E8 U+6211 U+5011 U+7684 U+65B0 U+7DB2 U+7AD9 U+3002 U+6211 U+5011 U+70BA U+60A8 U+63D0 U+4F9B U+512A U+8CEA U+7684 U+7522 U+54C1 U+8207 U+670D U+52D9 U+3002"
        },
        {
          "id": "cjk-03",
          "text": "新しいウェブサイトへようこそ。ご不明な点がございましたら、お気軽にお問い合わせください。",
          "note": "Japanese mixing kanji, hiragana and katakana",
          "lang": "ja",
          "length": {
            "codepoints": 44,
            "utf16_units": 44,
            "utf8_bytes": 132
          },
          "codepoints": null
        },
        {
          "id": "cjk-04",
          "text": "日本語の文章には単語の間にスペースがないため、改行の位置はブラウザやフォントによって異なることがあります。",
          "note": "Long Japanese sentence without spaces (line-breaking rules)",
          "lang": "ja",
          "length": {
            "codepoints": 53,
            "utf16_units": 53,
            "utf8_bytes": 159
          },
          "codepoints": null
        },
        {
          "id": "cjk-05",
          "text": "ひらがなだけでかいたぶんしょうです。",
          "note": "Hiragana only",
          "lang": "ja",
          "length": {
            "codepoints": 18,
            "utf16_units": 18,
            "utf8_bytes": 54
          },
          "codepoints": "U+3072 U+3089 U+304C U+306A U+3060 U+3051 U+3067 U+304B U+3044 U+305F U+3076 U+3093 U+3057 U+3087 U+3046 U+3067 U+3059 U+3002"
        },
        {
          "id": "cjk-06",
          "text": "コンピューター・ソフトウェア・エンジニアリング",
          "note": "Katakana with middle dots (U+30FB) and long vowel marks",
          "lang": "ja",
          "length": {
            "codepoints": 23,
            "utf16_units": 23,
            "utf8_bytes": 69
          },
          "codepoints": "U+30B3 U+30F3 U+30D4 U+30E5 U+30FC U+30BF U+30FC U+30FB U+30BD U+30D5 U+30C8 U+30A6 U+30A7 U+30A2 U+30FB U+30A8 U+30F3 U+30B8 U+30CB U+30A2 U+30EA U+30F3 U+30B0"
        },
        {
          "id": "cjk-07",
          "text": "ﾃﾞｰﾀﾍﾞｰｽ ｶﾀｶﾅ",
          "note": "Half-width katakana with separate dakuten (U+FF9E)",
          "lang": "ja",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 37
          },
          "codepoints": "U+FF83 U+FF9E U+FF70 U+FF80 U+FF8D U+FF9E U+FF70 U+FF7D U+0020 U+FF76 U+FF80 U+FF76 U+FF85"
        },
        {
          "id": "cjk-08",
          "text": "ＡＢＣ１２３　ｆｕｌｌ－ｗｉｄｔｈ",
          "note": "Full-width Latin and digits with ideographic space (U+3000)",
          "lang": "ja",
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 51
          },
          "codepoints": "U+FF21 U+FF22 U+FF23 U+FF11 U+FF12 U+FF13 U+3000 U+FF46 U+FF55 U+FF4C U+FF4C U+FF0D U+FF57 U+FF49 U+FF44 U+FF54 U+FF48"
        },
        {
          "id": "cjk-09",
          "text": "「引用」、『二重引用』。！？（括弧）【見出し】〜",
          "note": "Full-width punctuation",
          "lang": "ja",
          "length": {
            "codepoints": 24,
            "utf16_units": 24,
            "utf8_bytes": 72
          },
          "codepoints": "U+300C U+5F15 U+7528 U+300D U+3001 U+300E U+4E8C U+91CD U+5F15 U+7528 U+300F U+3002 U+FF01 U+FF1F U+FF08 U+62EC U+5F27 U+FF09 U+3010 U+898B U+51FA U+3057 U+3011 U+301C"
        },
        {
          "id": "cjk-10",
          "text": "２０２６年１０月８日（木）",
          "note": "Date with full-width digits",
          "lang": "ja",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 39
          },
          "codepoints": "U+FF12 U+FF10 U+FF12 U+FF16 U+5E74 U+FF11 U+FF10 U+6708 U+FF18 U+65E5 U+FF08 U+6728 U+FF09"
        },
        {
          "id": "cjk-11",
          "text": "价格：¥1,280（含税）",
          "note": "Chinese price with full-width colon and parentheses",
          "lang": "zh-Hans",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 28
          },
          "codepoints": "U+4EF7 U+683C U+FF1A U+00A5 U+0031 U+002C U+0032 U+0038 U+0030 U+FF08 U+542B U+7A0E U+FF09"
        },
        {
          "id": "cjk-12",
          "text": "支持Wi-Fi和USB-C的充电器",
          "note": "Chinese with embedded Latin words (no spaces)",
          "lang": "zh-Hans",
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 31
          },
          "codepoints": "U+652F U+6301 U+0057 U+0069 U+002D U+0046 U+0069 U+548C U+0055 U+0053 U+0042 U+002D U+0043 U+7684 U+5145 U+7535 U+5668"
        },
        {
          "id": "cjk-13",
          "text": "새로운 웹사이트에 오신 것을 환영합니다. 궁금한 점이 있으시면 언제든지 문의해 주세요.",
          "note": "Korean (Hangul, spaces between words)",
          "lang": "ko",
          "length": {
            "codepoints": 48,
            "utf16_units": 48,
            "utf8_bytes": 120
          },
          "codepoints": null
        },
        {
          "id": "cjk-14",
          "text": "𠮷田さん",
          "note": "Astral CJK ideograph U+20BB7: 2 UTF-16 units, 4 UTF-8 bytes (breaks MySQL utf8mb3)",
          "lang": "ja",
          "length": {
            "codepoints": 4,
            "utf16_units": 5,
            "utf8_bytes": 13
          },
          "codepoints": "U+20BB7 U+7530 U+3055 U+3093"
        },
        {
          "id": "cjk-15",
          "text": "𠀀𪛖𫝀",
          "note": "CJK Extension B, B and D ideographs (often missing from fonts)",
          "lang": "zh-Hant",
          "length": {
            "codepoints": 3,
            "utf16_units": 6,
            "utf8_bytes": 12
          },
          "codepoints": "U+20000 U+2A6D6 U+2B740"
        },
        {
          "id": "cjk-16",
          "text": "葛󠄀城",
          "note": "Ideographic variation selector (U+E0100) after a kanji",
          "lang": "ja",
          "length": {
            "codepoints": 3,
            "utf16_units": 4,
            "utf8_bytes": 10
          },
          "codepoints": "U+845B U+E0100 U+57CE"
        },
        {
          "id": "cjk-17",
          "text": "直 骨 写 角 底",
          "note": "Han-unified characters whose glyph shape depends on lang (zh vs ja)",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 19
          },
          "codepoints": "U+76F4 U+0020 U+9AA8 U+0020 U+5199 U+0020 U+89D2 U+0020 U+5E95"
        },
        {
          "id": "cjk-18",
          "text": "令和8年",
          "note": "Japanese era year",
          "lang": "ja",
          "length": {
            "codepoints": 4,
            "utf16_units": 4,
            "utf8_bytes": 10
          },
          "codepoints": "U+4EE4 U+548C U+0038 U+5E74"
        }
      ]
    },
    "emoji": {
      "description": "ZWJ sequences, skin tones, flags, keycaps, presentation selectors, recent emoji",
      "items": [
        {
          "id": "emoji-01",
          "text": "👍",
          "note": "Single emoji (1 code point, 2 UTF-16 units, 4 UTF-8 bytes)",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 2,
            "utf8_bytes": 4
          },
          "codepoints": "U+1F44D"
        },
        {
          "id": "emoji-02",
          "text": "👍🏽",
          "note": "Emoji with skin-tone modifier",
          "lang": null,
          "length": {
            "codepoints": 2,
            "utf16_units": 4,
            "utf8_bytes": 8
          },
          "codepoints": "U+1F44D U+1F3FD"
        },
        {
          "id": "emoji-03",
          "text": "👨‍👩‍👧‍👦",
          "note": "ZWJ family sequence (7 code points, 1 grapheme)",
          "lang": null,
          "length": {
            "codepoints": 7,
            "utf16_units": 11,
            "utf8_bytes": 25
          },
          "codepoints": "U+1F468 U+200D U+1F469 U+200D U+1F467 U+200D U+1F466"
        },
        {
          "id": "emoji-04",
          "text": "🧑🏼‍💻",
          "note": "ZWJ profession with skin tone",
          "lang": null,
          "length": {
            "codepoints": 4,
            "utf16_units": 7,
            "utf8_bytes": 15
          },
          "codepoints": "U+1F9D1 U+1F3FC U+200D U+1F4BB"
        },
        {
          "id": "emoji-05",
          "text": "👩🏾‍🤝‍👩🏻",
          "note": "ZWJ sequence with two different skin tones",
          "lang": null,
          "length": {
            "codepoints": 7,
            "utf16_units": 12,
            "utf8_bytes": 26
          },
          "codepoints": "U+1F469 U+1F3FE U+200D U+1F91D U+200D U+1F469 U+1F3FB"
        },
        {
          "id": "emoji-06",
          "text": "🧑🏻‍❤️‍💋‍🧑🏿",
          "note": "Kiss with two skin tones (10 code points)",
          "lang": null,
          "length": {
            "codepoints": 10,
            "utf16_units": 15,
            "utf8_bytes": 35
          },
          "codepoints": "U+1F9D1 U+1F3FB U+200D U+2764 U+FE0F U+200D U+1F48B U+200D U+1F9D1 U+1F3FF"
        },
        {
          "id": "emoji-07",
          "text": "🏳️‍🌈",
          "note": "Rainbow flag (VS16 + ZWJ)",
          "lang": null,
          "length": {
            "codepoints": 4,
            "utf16_units": 6,
            "utf8_bytes": 14
          },
          "codepoints": "U+1F3F3 U+FE0F U+200D U+1F308"
        },
        {
          "id": "emoji-08",
          "text": "🏴‍☠️",
          "note": "Pirate flag",
          "lang": null,
          "length": {
            "codepoints": 4,
            "utf16_units": 5,
            "utf8_bytes": 13
          },
          "codepoints": "U+1F3F4 U+200D U+2620 U+FE0F"
        },
        {
          "id": "emoji-09",
          "text": "🇸🇪 🇯🇵 🇧🇷 🇺🇳",
          "note": "Regional-indicator flags",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 19,
            "utf8_bytes": 35
          },
          "codepoints": "U+1F1F8 U+1F1EA U+0020 U+1F1EF U+1F1F5 U+0020 U+1F1E7 U+1F1F7 U+0020 U+1F1FA U+1F1F3"
        },
        {
          "id": "emoji-10",
          "text": "🏴󠁧󠁢󠁳󠁣󠁴󠁿",
          "note": "Subdivision flag via tag sequence (England/Scotland style)",
          "lang": null,
          "length": {
            "codepoints": 7,
            "utf16_units": 14,
            "utf8_bytes": 28
          },
          "codepoints": "U+1F3F4 U+E0067 U+E0062 U+E0073 U+E0063 U+E0074 U+E007F"
        },
        {
          "id": "emoji-11",
          "text": "1️⃣ #️⃣ *️⃣ 🔟",
          "note": "Keycap sequences",
          "lang": null,
          "length": {
            "codepoints": 13,
            "utf16_units": 14,
            "utf8_bytes": 28
          },
          "codepoints": "U+0031 U+FE0F U+20E3 U+0020 U+0023 U+FE0F U+20E3 U+0020 U+002A U+FE0F U+20E3 U+0020 U+1F51F"
        },
        {
          "id": "emoji-12",
          "text": "❤ ❤️",
          "note": "Text-style vs emoji-style heart (VS16)",
          "lang": null,
          "length": {
            "codepoints": 4,
            "utf16_units": 4,
            "utf8_bytes": 10
          },
          "codepoints": "U+2764 U+0020 U+2764 U+FE0F"
        },
        {
          "id": "emoji-13",
          "text": "☺︎ ☺️",
          "note": "Explicit text (VS15) vs emoji (VS16) presentation",
          "lang": null,
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 13
          },
          "codepoints": "U+263A U+FE0E U+0020 U+263A U+FE0F"
        },
        {
          "id": "emoji-14",
          "text": "🐻‍❄️",
          "note": "Polar bear (Emoji 13.0 ZWJ sequence)",
          "lang": null,
          "length": {
            "codepoints": 4,
            "utf16_units": 5,
            "utf8_bytes": 13
          },
          "codepoints": "U+1F43B U+200D U+2744 U+FE0F"
        },
        {
          "id": "emoji-15",
          "text": "🫨 🙂‍↔️ 🫩",
          "note": "Recent emoji (15.0, 15.1, 16.0): font fallback shows tofu or pieces",
          "lang": null,
          "length": {
            "codepoints": 8,
            "utf16_units": 11,
            "utf8_bytes": 23
          },
          "codepoints": "U+1FAE8 U+0020 U+1F642 U+200D U+2194 U+FE0F U+0020 U+1FAE9"
        },
        {
          "id": "emoji-16",
          "text": "🇦",
          "note": "Lone regional indicator",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 2,
            "utf8_bytes": 4
          },
          "codepoints": "U+1F1E6"
        },
        {
          "id": "emoji-17",
          "text": "🏻",
          "note": "Lone skin-tone modifier",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 2,
            "utf8_bytes": 4
          },
          "codepoints": "U+1F3FB"
        },
        {
          "id": "emoji-18",
          "text": "Great job! 🎉 See you soon 👋🏿",
          "note": "Emoji inside a sentence",
          "lang": "en",
          "length": {
            "codepoints": 28,
            "utf16_units": 31,
            "utf8_bytes": 37
          },
          "codepoints": "U+0047 U+0072 U+0065 U+0061 U+0074 U+0020 U+006A U+006F U+0062 U+0021 U+0020 U+1F389 U+0020 U+0053 U+0065 U+0065 U+0020 U+0079 U+006F U+0075 U+0020 U+0073 U+006F U+006F U+006E U+0020 U+1F44B U+1F3FF"
        },
        {
          "id": "emoji-19",
          "text": "co🔥de",
          "note": "Emoji inside a word",
          "lang": "en",
          "length": {
            "codepoints": 5,
            "utf16_units": 6,
            "utf8_bytes": 8
          },
          "codepoints": "U+0063 U+006F U+1F525 U+0064 U+0065"
        },
        {
          "id": "emoji-20",
          "text": "😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀😀",
          "note": "20 emoji in a row (80 UTF-8 bytes, 40 UTF-16 units)",
          "lang": null,
          "length": {
            "codepoints": 20,
            "utf16_units": 40,
            "utf8_bytes": 80
          },
          "codepoints": "U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600 U+1F600"
        }
      ]
    },
    "names": {
      "description": "Person names from 1 to 100+ characters: apostrophes, hyphens, particles, diacritics, scripts",
      "items": [
        {
          "id": "names-01",
          "text": "Ng",
          "note": "2-character surname",
          "lang": null,
          "length": {
            "codepoints": 2,
            "utf16_units": 2,
            "utf8_bytes": 2
          },
          "codepoints": "U+004E U+0067"
        },
        {
          "id": "names-02",
          "text": "Li Na",
          "note": "Short two-part name",
          "lang": null,
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 5
          },
          "codepoints": "U+004C U+0069 U+0020 U+004E U+0061"
        },
        {
          "id": "names-03",
          "text": "O",
          "note": "1-character name (exists)",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 1
          },
          "codepoints": "U+004F"
        },
        {
          "id": "names-04",
          "text": "O'Connor",
          "note": "ASCII apostrophe (SQL quoting)",
          "lang": "en",
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 8
          },
          "codepoints": "U+004F U+0027 U+0043 U+006F U+006E U+006E U+006F U+0072"
        },
        {
          "id": "names-05",
          "text": "Mary-Jane O’Neill",
          "note": "Hyphen and typographic apostrophe (U+2019)",
          "lang": "en",
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 19
          },
          "codepoints": "U+004D U+0061 U+0072 U+0079 U+002D U+004A U+0061 U+006E U+0065 U+0020 U+004F U+2019 U+004E U+0065 U+0069 U+006C U+006C"
        },
        {
          "id": "names-06",
          "text": "D'Angelo-Rossi",
          "note": "Apostrophe and hyphen",
          "lang": "it",
          "length": {
            "codepoints": 14,
            "utf16_units": 14,
            "utf8_bytes": 14
          },
          "codepoints": "U+0044 U+0027 U+0041 U+006E U+0067 U+0065 U+006C U+006F U+002D U+0052 U+006F U+0073 U+0073 U+0069"
        },
        {
          "id": "names-07",
          "text": "Siobhán Ní Bhriain",
          "note": "Irish name with particle",
          "lang": "ga",
          "length": {
            "codepoints": 18,
            "utf16_units": 18,
            "utf8_bytes": 20
          },
          "codepoints": "U+0053 U+0069 U+006F U+0062 U+0068 U+00E1 U+006E U+0020 U+004E U+00ED U+0020 U+0042 U+0068 U+0072 U+0069 U+0061 U+0069 U+006E"
        },
        {
          "id": "names-08",
          "text": "José María Fernández de la Cruz y Sánchez",
          "note": "Spanish multi-part name with particles",
          "lang": "es",
          "length": {
            "codepoints": 41,
            "utf16_units": 41,
            "utf8_bytes": 45
          },
          "codepoints": null
        },
        {
          "id": "names-09",
          "text": "Zoë Ångström-Øverli",
          "note": "Diaeresis, ring and slashed O",
          "lang": null,
          "length": {
            "codepoints": 19,
            "utf16_units": 19,
            "utf8_bytes": 23
          },
          "codepoints": "U+005A U+006F U+00EB U+0020 U+00C5 U+006E U+0067 U+0073 U+0074 U+0072 U+00F6 U+006D U+002D U+00D8 U+0076 U+0065 U+0072 U+006C U+0069"
        },
        {
          "id": "names-10",
          "text": "Zoë Möller",
          "note": "Decomposed diacritics (NFD): compare with NFC before matching",
          "lang": null,
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 14
          },
          "codepoints": "U+005A U+006F U+0065 U+0308 U+0020 U+004D U+006F U+0308 U+006C U+006C U+0065 U+0072"
        },
        {
          "id": "names-11",
          "text": "Þórdís Guðmundsdóttir",
          "note": "Icelandic thorn and eth",
          "lang": "is",
          "length": {
            "codepoints": 21,
            "utf16_units": 21,
            "utf8_bytes": 26
          },
          "codepoints": "U+00DE U+00F3 U+0072 U+0064 U+00ED U+0073 U+0020 U+0047 U+0075 U+00F0 U+006D U+0075 U+006E U+0064 U+0073 U+0064 U+00F3 U+0074 U+0074 U+0069 U+0072"
        },
        {
          "id": "names-12",
          "text": "İlkay Işıkoğlu",
          "note": "Turkish dotted/dotless I (case-mapping trap)",
          "lang": "tr",
          "length": {
            "codepoints": 14,
            "utf16_units": 14,
            "utf8_bytes": 18
          },
          "codepoints": "U+0130 U+006C U+006B U+0061 U+0079 U+0020 U+0049 U+015F U+0131 U+006B U+006F U+011F U+006C U+0075"
        },
        {
          "id": "names-13",
          "text": "Jürgen Weiß",
          "note": "Sharp s: upper-cases to WEISS",
          "lang": "de",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+004A U+00FC U+0072 U+0067 U+0065 U+006E U+0020 U+0057 U+0065 U+0069 U+00DF"
        },
        {
          "id": "names-14",
          "text": "Łucja Żółkiewska-Szczęsna",
          "note": "Polish diacritics",
          "lang": "pl",
          "length": {
            "codepoints": 25,
            "utf16_units": 25,
            "utf8_bytes": 30
          },
          "codepoints": "U+0141 U+0075 U+0063 U+006A U+0061 U+0020 U+017B U+00F3 U+0142 U+006B U+0069 U+0065 U+0077 U+0073 U+006B U+0061 U+002D U+0053 U+007A U+0063 U+007A U+0119 U+0073 U+006E U+0061"
        },
        {
          "id": "names-15",
          "text": "Nguyễn Văn An",
          "note": "Vietnamese stacked diacritics",
          "lang": "vi",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 16
          },
          "codepoints": "U+004E U+0067 U+0075 U+0079 U+1EC5 U+006E U+0020 U+0056 U+0103 U+006E U+0020 U+0041 U+006E"
        },
        {
          "id": "names-16",
          "text": "Trần Thị Bích Ngọc",
          "note": "Vietnamese four-part name",
          "lang": "vi",
          "length": {
            "codepoints": 18,
            "utf16_units": 18,
            "utf8_bytes": 25
          },
          "codepoints": "U+0054 U+0072 U+1EA7 U+006E U+0020 U+0054 U+0068 U+1ECB U+0020 U+0042 U+00ED U+0063 U+0068 U+0020 U+004E U+0067 U+1ECD U+0063"
        },
        {
          "id": "names-17",
          "text": "Szabó Éva",
          "note": "Hungarian family-name-first order",
          "lang": "hu",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 11
          },
          "codepoints": "U+0053 U+007A U+0061 U+0062 U+00F3 U+0020 U+00C9 U+0076 U+0061"
        },
        {
          "id": "names-18",
          "text": "Hans-Jürgen van der Meer",
          "note": "Lower-case particle in surname",
          "lang": "nl",
          "length": {
            "codepoints": 24,
            "utf16_units": 24,
            "utf8_bytes": 25
          },
          "codepoints": "U+0048 U+0061 U+006E U+0073 U+002D U+004A U+00FC U+0072 U+0067 U+0065 U+006E U+0020 U+0076 U+0061 U+006E U+0020 U+0064 U+0065 U+0072 U+0020 U+004D U+0065 U+0065 U+0072"
        },
        {
          "id": "names-19",
          "text": "Fiona MacLeod-McAllister",
          "note": "Internal capitals (naive title-casing breaks them)",
          "lang": "en",
          "length": {
            "codepoints": 24,
            "utf16_units": 24,
            "utf8_bytes": 24
          },
          "codepoints": "U+0046 U+0069 U+006F U+006E U+0061 U+0020 U+004D U+0061 U+0063 U+004C U+0065 U+006F U+0064 U+002D U+004D U+0063 U+0041 U+006C U+006C U+0069 U+0073 U+0074 U+0065 U+0072"
        },
        {
          "id": "names-20",
          "text": "Kealoha Nāone-Kaʻula",
          "note": "Hawaiian ʻokina (U+02BB) and macron",
          "lang": "haw",
          "length": {
            "codepoints": 20,
            "utf16_units": 20,
            "utf8_bytes": 22
          },
          "codepoints": "U+004B U+0065 U+0061 U+006C U+006F U+0068 U+0061 U+0020 U+004E U+0101 U+006F U+006E U+0065 U+002D U+004B U+0061 U+02BB U+0075 U+006C U+0061"
        },
        {
          "id": "names-21",
          "text": "Aroha Tāwhiao",
          "note": "Māori macron",
          "lang": "mi",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 14
          },
          "codepoints": "U+0041 U+0072 U+006F U+0068 U+0061 U+0020 U+0054 U+0101 U+0077 U+0068 U+0069 U+0061 U+006F"
        },
        {
          "id": "names-22",
          "text": "Wulandari",
          "note": "Mononym (single name, common in Indonesia)",
          "lang": "id",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0057 U+0075 U+006C U+0061 U+006E U+0064 U+0061 U+0072 U+0069"
        },
        {
          "id": "names-23",
          "text": "Ngozi Chukwuemeka-Okafor",
          "note": "Igbo name, hyphenated",
          "lang": "ig",
          "length": {
            "codepoints": 24,
            "utf16_units": 24,
            "utf8_bytes": 24
          },
          "codepoints": "U+004E U+0067 U+006F U+007A U+0069 U+0020 U+0043 U+0068 U+0075 U+006B U+0077 U+0075 U+0065 U+006D U+0065 U+006B U+0061 U+002D U+004F U+006B U+0061 U+0066 U+006F U+0072"
        },
        {
          "id": "names-24",
          "text": "Venkataraman Subramaniam Ramachandran",
          "note": "Long South Indian name",
          "lang": "ta",
          "length": {
            "codepoints": 37,
            "utf16_units": 37,
            "utf8_bytes": 37
          },
          "codepoints": "U+0056 U+0065 U+006E U+006B U+0061 U+0074 U+0061 U+0072 U+0061 U+006D U+0061 U+006E U+0020 U+0053 U+0075 U+0062 U+0072 U+0061 U+006D U+0061 U+006E U+0069 U+0061 U+006D U+0020 U+0052 U+0061 U+006D U+0061 U+0063 U+0068 U+0061 U+006E U+0064 U+0072 U+0061 U+006E"
        },
        {
          "id": "names-25",
          "text": "K. S. Ramanathan",
          "note": "Initials with periods",
          "lang": "ta",
          "length": {
            "codepoints": 16,
            "utf16_units": 16,
            "utf8_bytes": 16
          },
          "codepoints": "U+004B U+002E U+0020 U+0053 U+002E U+0020 U+0052 U+0061 U+006D U+0061 U+006E U+0061 U+0074 U+0068 U+0061 U+006E"
        },
        {
          "id": "names-26",
          "text": "Abdullah bin Khalid al-Rashid",
          "note": "Arabic name in Latin script with particles",
          "lang": "en",
          "length": {
            "codepoints": 29,
            "utf16_units": 29,
            "utf8_bytes": 29
          },
          "codepoints": "U+0041 U+0062 U+0064 U+0075 U+006C U+006C U+0061 U+0068 U+0020 U+0062 U+0069 U+006E U+0020 U+004B U+0068 U+0061 U+006C U+0069 U+0064 U+0020 U+0061 U+006C U+002D U+0052 U+0061 U+0073 U+0068 U+0069 U+0064"
        },
        {
          "id": "names-27",
          "text": "عبد الله بن خالد الراشد",
          "note": "Arabic name (RTL)",
          "lang": "ar",
          "length": {
            "codepoints": 23,
            "utf16_units": 23,
            "utf8_bytes": 42
          },
          "codepoints": "U+0639 U+0628 U+062F U+0020 U+0627 U+0644 U+0644 U+0647 U+0020 U+0628 U+0646 U+0020 U+062E U+0627 U+0644 U+062F U+0020 U+0627 U+0644 U+0631 U+0627 U+0634 U+062F"
        },
        {
          "id": "names-28",
          "text": "אברהם בן־דוד",
          "note": "Hebrew name with maqaf (U+05BE)",
          "lang": "he",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 23
          },
          "codepoints": "U+05D0 U+05D1 U+05E8 U+05D4 U+05DD U+0020 U+05D1 U+05DF U+05BE U+05D3 U+05D5 U+05D3"
        },
        {
          "id": "names-29",
          "text": "Анастасия Владимировна Кузнецова",
          "note": "Russian name with patronymic",
          "lang": "ru",
          "length": {
            "codepoints": 32,
            "utf16_units": 32,
            "utf8_bytes": 62
          },
          "codepoints": "U+0410 U+043D U+0430 U+0441 U+0442 U+0430 U+0441 U+0438 U+044F U+0020 U+0412 U+043B U+0430 U+0434 U+0438 U+043C U+0438 U+0440 U+043E U+0432 U+043D U+0430 U+0020 U+041A U+0443 U+0437 U+043D U+0435 U+0446 U+043E U+0432 U+0430"
        },
        {
          "id": "names-30",
          "text": "Αικατερίνη Παπαδοπούλου",
          "note": "Greek name",
          "lang": "el",
          "length": {
            "codepoints": 23,
            "utf16_units": 23,
            "utf8_bytes": 45
          },
          "codepoints": "U+0391 U+03B9 U+03BA U+03B1 U+03C4 U+03B5 U+03C1 U+03AF U+03BD U+03B7 U+0020 U+03A0 U+03B1 U+03C0 U+03B1 U+03B4 U+03BF U+03C0 U+03BF U+03CD U+03BB U+03BF U+03C5"
        },
        {
          "id": "names-31",
          "text": "王小明",
          "note": "Chinese placeholder name (3 characters, no space)",
          "lang": "zh-Hans",
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 9
          },
          "codepoints": "U+738B U+5C0F U+660E"
        },
        {
          "id": "names-32",
          "text": "山田太郎",
          "note": "Japanese placeholder name (no space)",
          "lang": "ja",
          "length": {
            "codepoints": 4,
            "utf16_units": 4,
            "utf8_bytes": 12
          },
          "codepoints": "U+5C71 U+7530 U+592A U+90CE"
        },
        {
          "id": "names-33",
          "text": "山田　花子",
          "note": "Japanese name with ideographic space (U+3000)",
          "lang": "ja",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+5C71 U+7530 U+3000 U+82B1 U+5B50"
        },
        {
          "id": "names-34",
          "text": "김민준",
          "note": "Korean name",
          "lang": "ko",
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 9
          },
          "codepoints": "U+AE40 U+BBFC U+C900"
        },
        {
          "id": "names-35",
          "text": "สมชาย ใจดี",
          "note": "Thai placeholder name",
          "lang": "th",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 28
          },
          "codepoints": "U+0E2A U+0E21 U+0E0A U+0E32 U+0E22 U+0020 U+0E43 U+0E08 U+0E14 U+0E35"
        },
        {
          "id": "names-36",
          "text": "Alex Null",
          "note": "Surname equal to a reserved word",
          "lang": "en",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0041 U+006C U+0065 U+0078 U+0020 U+004E U+0075 U+006C U+006C"
        },
        {
          "id": "names-37",
          "text": "Anna 🌸",
          "note": "Name containing an emoji",
          "lang": null,
          "length": {
            "codepoints": 6,
            "utf16_units": 7,
            "utf8_bytes": 9
          },
          "codepoints": "U+0041 U+006E U+006E U+0061 U+0020 U+1F338"
        },
        {
          "id": "names-38",
          "text": "Robert J. Smith-Jones III",
          "note": "Middle initial and generational suffix",
          "lang": "en",
          "length": {
            "codepoints": 25,
            "utf16_units": 25,
            "utf8_bytes": 25
          },
          "codepoints": "U+0052 U+006F U+0062 U+0065 U+0072 U+0074 U+0020 U+004A U+002E U+0020 U+0053 U+006D U+0069 U+0074 U+0068 U+002D U+004A U+006F U+006E U+0065 U+0073 U+0020 U+0049 U+0049 U+0049"
        },
        {
          "id": "names-39",
          "text": "Wolfeschlegelsteinhausenbergerdorff",
          "note": "Very long single surname",
          "lang": "de",
          "length": {
            "codepoints": 35,
            "utf16_units": 35,
            "utf8_bytes": 35
          },
          "codepoints": "U+0057 U+006F U+006C U+0066 U+0065 U+0073 U+0063 U+0068 U+006C U+0065 U+0067 U+0065 U+006C U+0073 U+0074 U+0065 U+0069 U+006E U+0068 U+0061 U+0075 U+0073 U+0065 U+006E U+0062 U+0065 U+0072 U+0067 U+0065 U+0072 U+0064 U+006F U+0072 U+0066 U+0066"
        },
        {
          "id": "names-40",
          "text": "Maximiliana Josefina Alexandra Charlotte Viktoria von und zu Hohenberg-Lichtenfels-Sankt Gallen",
          "note": "Name over 80 characters",
          "lang": "de",
          "length": {
            "codepoints": 95,
            "utf16_units": 95,
            "utf8_bytes": 95
          },
          "codepoints": null
        },
        {
          "id": "names-41",
          "text": "Johanna Marie Wolfeschlegelsteinhausenbergerdorff Senior-Ramírez de la Fuente y Montenegro-Castellanos",
          "note": "Name over 100 characters",
          "lang": null,
          "length": {
            "codepoints": 102,
            "utf16_units": 102,
            "utf8_bytes": 103
          },
          "codepoints": null
        }
      ]
    },
    "combining": {
      "description": "Combining marks, NFC vs NFD, stacked diacritics, Zalgo-style overflow, Indic and Thai",
      "items": [
        {
          "id": "combining-01",
          "text": "é é",
          "note": "Precomposed vs decomposed e-acute (NFC vs NFD look identical)",
          "lang": null,
          "length": {
            "codepoints": 4,
            "utf16_units": 4,
            "utf8_bytes": 6
          },
          "codepoints": "U+00E9 U+0020 U+0065 U+0301"
        },
        {
          "id": "combining-02",
          "text": "Amélie Amélie",
          "note": "Same word in NFC and NFD: different bytes, equal after normalization",
          "lang": "fr",
          "length": {
            "codepoints": 14,
            "utf16_units": 14,
            "utf8_bytes": 16
          },
          "codepoints": "U+0041 U+006D U+00E9 U+006C U+0069 U+0065 U+0020 U+0041 U+006D U+0065 U+0301 U+006C U+0069 U+0065"
        },
        {
          "id": "combining-03",
          "text": "Việt Nam",
          "note": "Vietnamese precomposed multi-diacritic letters (NFC)",
          "lang": "vi",
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 10
          },
          "codepoints": "U+0056 U+0069 U+1EC7 U+0074 U+0020 U+004E U+0061 U+006D"
        },
        {
          "id": "combining-04",
          "text": "Việt Nam",
          "note": "Vietnamese with decomposed dot-below + circumflex (NFD)",
          "lang": "vi",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 12
          },
          "codepoints": "U+0056 U+0069 U+0065 U+0323 U+0302 U+0074 U+0020 U+004E U+0061 U+006D"
        },
        {
          "id": "combining-05",
          "text": "한글 한글",
          "note": "Korean decomposed jamo vs precomposed syllables",
          "lang": "ko",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 25
          },
          "codepoints": "U+1112 U+1161 U+11AB U+1100 U+1173 U+11AF U+0020 U+D55C U+AE00"
        },
        {
          "id": "combining-06",
          "text": "à̛̖̗̘̙̜̝̞̟́̂̃̄̆̇̈̉̊̋̌̍̎̏̐̑̒̓̔̚",
          "note": "30 combining marks on one letter (line-height overflow)",
          "lang": null,
          "length": {
            "codepoints": 31,
            "utf16_units": 31,
            "utf8_bytes": 61
          },
          "codepoints": "U+0061 U+0300 U+0301 U+0302 U+0303 U+0304 U+0306 U+0307 U+0308 U+0309 U+030A U+030B U+030C U+030D U+030E U+030F U+0310 U+0311 U+0312 U+0313 U+0314 U+0316 U+0317 U+0318 U+0319 U+031A U+031B U+031C U+031D U+031E U+031F"
        },
        {
          "id": "combining-07",
          "text": "s̶t̶r̶i̶k̶e̶t̶h̶r̶o̶u̶g̶h̶",
          "note": "Combining long stroke overlay (U+0336) on every letter",
          "lang": "en",
          "length": {
            "codepoints": 26,
            "utf16_units": 26,
            "utf8_bytes": 39
          },
          "codepoints": "U+0073 U+0336 U+0074 U+0336 U+0072 U+0336 U+0069 U+0336 U+006B U+0336 U+0065 U+0336 U+0074 U+0336 U+0068 U+0336 U+0072 U+0336 U+006F U+0336 U+0075 U+0336 U+0067 U+0336 U+0068 U+0336"
        },
        {
          "id": "combining-08",
          "text": "u̲n̲d̲e̲r̲l̲i̲n̲e̲d̲",
          "note": "Combining low line (U+0332) on every letter",
          "lang": "en",
          "length": {
            "codepoints": 20,
            "utf16_units": 20,
            "utf8_bytes": 30
          },
          "codepoints": "U+0075 U+0332 U+006E U+0332 U+0064 U+0332 U+0065 U+0332 U+0072 U+0332 U+006C U+0332 U+0069 U+0332 U+006E U+0332 U+0065 U+0332 U+0064 U+0332"
        },
        {
          "id": "combining-09",
          "text": "A⃝ B⃞",
          "note": "Combining enclosing circle and square",
          "lang": null,
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 9
          },
          "codepoints": "U+0041 U+20DD U+0020 U+0042 U+20DE"
        },
        {
          "id": "combining-10",
          "text": "क्षत्रिय नमस्ते",
          "note": "Devanagari conjuncts and virama",
          "lang": "hi",
          "length": {
            "codepoints": 15,
            "utf16_units": 15,
            "utf8_bytes": 43
          },
          "codepoints": "U+0915 U+094D U+0937 U+0924 U+094D U+0930 U+093F U+092F U+0020 U+0928 U+092E U+0938 U+094D U+0924 U+0947"
        },
        {
          "id": "combining-11",
          "text": "தமிழ்நாடு",
          "note": "Tamil with vowel signs and pulli",
          "lang": "ta",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 27
          },
          "codepoints": "U+0BA4 U+0BAE U+0BBF U+0BB4 U+0BCD U+0BA8 U+0BBE U+0B9F U+0BC1"
        },
        {
          "id": "combining-12",
          "text": "น้ำแข็ง ผู้ใหญ่",
          "note": "Thai stacked tone marks",
          "lang": "th",
          "length": {
            "codepoints": 15,
            "utf16_units": 15,
            "utf8_bytes": 43
          },
          "codepoints": "U+0E19 U+0E49 U+0E33 U+0E41 U+0E02 U+0E47 U+0E07 U+0020 U+0E1C U+0E39 U+0E49 U+0E43 U+0E2B U+0E0D U+0E48"
        },
        {
          "id": "combining-13",
          "text": "שָׁלוֹם",
          "note": "Hebrew with niqqud (vowel points)",
          "lang": "he",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 14
          },
          "codepoints": "U+05E9 U+05C1 U+05B8 U+05DC U+05D5 U+05B9 U+05DD"
        },
        {
          "id": "combining-14",
          "text": "كِتَابٌ",
          "note": "Arabic with harakat (short-vowel marks)",
          "lang": "ar",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 14
          },
          "codepoints": "U+0643 U+0650 U+062A U+064E U+0627 U+0628 U+064C"
        },
        {
          "id": "combining-15",
          "text": "ﬁle ﬀ Å Ω",
          "note": "Compatibility characters (fi/ff ligatures, Angstrom and Ohm signs): NFKC changes them",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 17
          },
          "codepoints": "U+FB01 U+006C U+0065 U+0020 U+FB00 U+0020 U+212B U+0020 U+2126"
        },
        {
          "id": "combining-16",
          "text": "H̆ȩ̶̏l͕ͩ͟l̚͜o̓͒ w̢̗ͅòr̶͟l̮d̷̀͟",
          "note": "Zalgo-lite: 1-3 random combining marks per letter",
          "lang": "en",
          "length": {
            "codepoints": 32,
            "utf16_units": 32,
            "utf8_bytes": 53
          },
          "codepoints": "U+0048 U+0306 U+0065 U+0336 U+030F U+0327 U+006C U+0355 U+0369 U+035F U+006C U+035C U+031A U+006F U+0313 U+0352 U+0020 U+0077 U+0345 U+0322 U+0317 U+006F U+0340 U+0072 U+0336 U+035F U+006C U+032E U+0064 U+0337 U+0300 U+035F"
        },
        {
          "id": "combining-17",
          "text": "P̵̶̨ͬ́ͬĺ̬͉͙̌̓̉̍ą̥̝ͤͭc̛͕̩̀e͚̞͑̏ḩ̤̥̹ͥͤ̒̌ȯ̶̶͖̝̈͆͟l̮̜̉̂̃̃͐̚d̯̬̿ͤ͋͊̿̕ḛ͙̣͝r̵̲̯̈́̓",
          "note": "Zalgo: 4-8 random combining marks per letter (overflows line height)",
          "lang": "en",
          "length": {
            "codepoints": 79,
            "utf16_units": 79,
            "utf8_bytes": 147
          },
          "codepoints": null
        }
      ]
    },
    "whitespace": {
      "description": "Zero-width characters, no-break and fixed-width spaces, line separators, control characters",
      "items": [
        {
          "id": "whitespace-01",
          "text": "Hello​World",
          "note": "Zero-width space U+200B",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0048 U+0065 U+006C U+006C U+006F U+200B U+0057 U+006F U+0072 U+006C U+0064"
        },
        {
          "id": "whitespace-02",
          "text": "Hello‌World",
          "note": "Zero-width non-joiner U+200C",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0048 U+0065 U+006C U+006C U+006F U+200C U+0057 U+006F U+0072 U+006C U+0064"
        },
        {
          "id": "whitespace-03",
          "text": "Hello‍World",
          "note": "Zero-width joiner U+200D",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0048 U+0065 U+006C U+006C U+006F U+200D U+0057 U+006F U+0072 U+006C U+0064"
        },
        {
          "id": "whitespace-04",
          "text": "Hello⁠World",
          "note": "Word joiner U+2060",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0048 U+0065 U+006C U+006C U+006F U+2060 U+0057 U+006F U+0072 U+006C U+0064"
        },
        {
          "id": "whitespace-05",
          "text": "Hello﻿World",
          "note": "Zero-width no-break space / BOM U+FEFF inside text",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0048 U+0065 U+006C U+006C U+006F U+FEFF U+0057 U+006F U+0072 U+006C U+0064"
        },
        {
          "id": "whitespace-06",
          "text": "﻿Starts with a BOM",
          "note": "Leading BOM U+FEFF (often stripped or shown as junk)",
          "lang": null,
          "length": {
            "codepoints": 18,
            "utf16_units": 18,
            "utf8_bytes": 20
          },
          "codepoints": "U+FEFF U+0053 U+0074 U+0061 U+0072 U+0074 U+0073 U+0020 U+0077 U+0069 U+0074 U+0068 U+0020 U+0061 U+0020 U+0042 U+004F U+004D"
        },
        {
          "id": "whitespace-07",
          "text": "10 000",
          "note": "No-break space U+00A0",
          "lang": null,
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 7
          },
          "codepoints": "U+0031 U+0030 U+00A0 U+0030 U+0030 U+0030"
        },
        {
          "id": "whitespace-08",
          "text": "10 000 €",
          "note": "Narrow no-break space U+202F (French number formatting)",
          "lang": "fr",
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 14
          },
          "codepoints": "U+0031 U+0030 U+202F U+0030 U+0030 U+0030 U+202F U+20AC"
        },
        {
          "id": "whitespace-09",
          "text": "Thin space Hair space",
          "note": "Thin space U+2009 and hair space U+200A",
          "lang": null,
          "length": {
            "codepoints": 21,
            "utf16_units": 21,
            "utf8_bytes": 25
          },
          "codepoints": "U+0054 U+0068 U+0069 U+006E U+2009 U+0073 U+0070 U+0061 U+0063 U+0065 U+0020 U+0048 U+0061 U+0069 U+0072 U+200A U+0073 U+0070 U+0061 U+0063 U+0065"
        },
        {
          "id": "whitespace-10",
          "text": "En space Em space Figure space Punctuation space",
          "note": "Fixed-width spaces U+2002 U+2003 U+2007 U+2008",
          "lang": null,
          "length": {
            "codepoints": 48,
            "utf16_units": 48,
            "utf8_bytes": 56
          },
          "codepoints": null
        },
        {
          "id": "whitespace-11",
          "text": "Ogham space",
          "note": "Ogham space mark U+1680",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+004F U+0067 U+0068 U+0061 U+006D U+1680 U+0073 U+0070 U+0061 U+0063 U+0065"
        },
        {
          "id": "whitespace-12",
          "text": "Ideographic　space",
          "note": "Ideographic space U+3000",
          "lang": null,
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 19
          },
          "codepoints": "U+0049 U+0064 U+0065 U+006F U+0067 U+0072 U+0061 U+0070 U+0068 U+0069 U+0063 U+3000 U+0073 U+0070 U+0061 U+0063 U+0065"
        },
        {
          "id": "whitespace-13",
          "text": "Tab\there",
          "note": "Horizontal tab",
          "lang": null,
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 8
          },
          "codepoints": "U+0054 U+0061 U+0062 U+0009 U+0068 U+0065 U+0072 U+0065"
        },
        {
          "id": "whitespace-14",
          "text": "Line\r\nbreak",
          "note": "CRLF line break",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 11
          },
          "codepoints": "U+004C U+0069 U+006E U+0065 U+000D U+000A U+0062 U+0072 U+0065 U+0061 U+006B"
        },
        {
          "id": "whitespace-15",
          "text": "Line\rbreak",
          "note": "Lone carriage return",
          "lang": null,
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 10
          },
          "codepoints": "U+004C U+0069 U+006E U+0065 U+000D U+0062 U+0072 U+0065 U+0061 U+006B"
        },
        {
          "id": "whitespace-16",
          "text": "Line separator",
          "note": "Line separator U+2028 (broke JavaScript string literals before ES2019)",
          "lang": null,
          "length": {
            "codepoints": 14,
            "utf16_units": 14,
            "utf8_bytes": 16
          },
          "codepoints": "U+004C U+0069 U+006E U+0065 U+2028 U+0073 U+0065 U+0070 U+0061 U+0072 U+0061 U+0074 U+006F U+0072"
        },
        {
          "id": "whitespace-17",
          "text": "Paragraph separator",
          "note": "Paragraph separator U+2029",
          "lang": null,
          "length": {
            "codepoints": 19,
            "utf16_units": 19,
            "utf8_bytes": 21
          },
          "codepoints": "U+0050 U+0061 U+0072 U+0061 U+0067 U+0072 U+0061 U+0070 U+0068 U+2029 U+0073 U+0065 U+0070 U+0061 U+0072 U+0061 U+0074 U+006F U+0072"
        },
        {
          "id": "whitespace-18",
          "text": "Nextline",
          "note": "Next line (NEL) U+0085",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 10
          },
          "codepoints": "U+004E U+0065 U+0078 U+0074 U+0085 U+006C U+0069 U+006E U+0065"
        },
        {
          "id": "whitespace-19",
          "text": "Vertical\u000btab",
          "note": "Vertical tab U+000B",
          "lang": null,
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 12
          },
          "codepoints": "U+0056 U+0065 U+0072 U+0074 U+0069 U+0063 U+0061 U+006C U+000B U+0074 U+0061 U+0062"
        },
        {
          "id": "whitespace-20",
          "text": "Form\ffeed",
          "note": "Form feed U+000C",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0046 U+006F U+0072 U+006D U+000C U+0066 U+0065 U+0065 U+0064"
        },
        {
          "id": "whitespace-21",
          "text": "Null\u0000byte",
          "note": "NUL U+0000 (truncates C strings, rejected by PostgreSQL text)",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+004E U+0075 U+006C U+006C U+0000 U+0062 U+0079 U+0074 U+0065"
        },
        {
          "id": "whitespace-22",
          "text": "Soft­hyphen",
          "note": "Soft hyphen U+00AD (invisible unless the line breaks there)",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 12
          },
          "codepoints": "U+0053 U+006F U+0066 U+0074 U+00AD U+0068 U+0079 U+0070 U+0068 U+0065 U+006E"
        },
        {
          "id": "whitespace-23",
          "text": "  leading and trailing spaces  ",
          "note": "Leading and trailing spaces",
          "lang": null,
          "length": {
            "codepoints": 31,
            "utf16_units": 31,
            "utf8_bytes": 31
          },
          "codepoints": "U+0020 U+0020 U+006C U+0065 U+0061 U+0064 U+0069 U+006E U+0067 U+0020 U+0061 U+006E U+0064 U+0020 U+0074 U+0072 U+0061 U+0069 U+006C U+0069 U+006E U+0067 U+0020 U+0073 U+0070 U+0061 U+0063 U+0065 U+0073 U+0020 U+0020"
        },
        {
          "id": "whitespace-24",
          "text": "Multiple     spaces",
          "note": "Five consecutive spaces (collapsed in HTML)",
          "lang": null,
          "length": {
            "codepoints": 19,
            "utf16_units": 19,
            "utf8_bytes": 19
          },
          "codepoints": "U+004D U+0075 U+006C U+0074 U+0069 U+0070 U+006C U+0065 U+0020 U+0020 U+0020 U+0020 U+0020 U+0073 U+0070 U+0061 U+0063 U+0065 U+0073"
        },
        {
          "id": "whitespace-25",
          "text": "trailing newline\n",
          "note": "Trailing newline",
          "lang": null,
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 17
          },
          "codepoints": "U+0074 U+0072 U+0061 U+0069 U+006C U+0069 U+006E U+0067 U+0020 U+006E U+0065 U+0077 U+006C U+0069 U+006E U+0065 U+000A"
        },
        {
          "id": "whitespace-26",
          "text": "   ",
          "note": "Only spaces",
          "lang": null,
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 3
          },
          "codepoints": "U+0020 U+0020 U+0020"
        },
        {
          "id": "whitespace-27",
          "text": "   ",
          "note": "Only no-break spaces (not stripped by many trim() implementations)",
          "lang": null,
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 6
          },
          "codepoints": "U+00A0 U+00A0 U+00A0"
        },
        {
          "id": "whitespace-28",
          "text": "​",
          "note": "Only a zero-width space (looks empty, is not)",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 3
          },
          "codepoints": "U+200B"
        },
        {
          "id": "whitespace-29",
          "text": "",
          "note": "Empty string",
          "lang": null,
          "length": {
            "codepoints": 0,
            "utf16_units": 0,
            "utf8_bytes": 0
          },
          "codepoints": ""
        },
        {
          "id": "whitespace-30",
          "text": "ㅤ",
          "note": "Hangul filler U+3164 (renders blank, used to fake empty names)",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 3
          },
          "codepoints": "U+3164"
        },
        {
          "id": "whitespace-31",
          "text": "⠀",
          "note": "Braille pattern blank U+2800 (renders blank)",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 3
          },
          "codepoints": "U+2800"
        },
        {
          "id": "whitespace-32",
          "text": "᠎",
          "note": "Mongolian vowel separator U+180E (was whitespace before Unicode 6.3)",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 3
          },
          "codepoints": "U+180E"
        },
        {
          "id": "whitespace-33",
          "text": "ᅟᅠ",
          "note": "Hangul choseong and jungseong fillers",
          "lang": null,
          "length": {
            "codepoints": 2,
            "utf16_units": 2,
            "utf8_bytes": 6
          },
          "codepoints": "U+115F U+1160"
        }
      ]
    },
    "numbers": {
      "description": "Locale number, currency, date and time formats, plus floating-point traps",
      "items": [
        {
          "id": "numbers-01",
          "text": "1,234,567.89",
          "note": "Number, US English",
          "lang": "en-US",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 12
          },
          "codepoints": "U+0031 U+002C U+0032 U+0033 U+0034 U+002C U+0035 U+0036 U+0037 U+002E U+0038 U+0039"
        },
        {
          "id": "numbers-02",
          "text": "1.234.567,89",
          "note": "Number, German",
          "lang": "de-DE",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 12
          },
          "codepoints": "U+0031 U+002E U+0032 U+0033 U+0034 U+002E U+0035 U+0036 U+0037 U+002C U+0038 U+0039"
        },
        {
          "id": "numbers-03",
          "text": "1 234 567,89",
          "note": "Number, French (narrow no-break space U+202F)",
          "lang": "fr-FR",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 16
          },
          "codepoints": "U+0031 U+202F U+0032 U+0033 U+0034 U+202F U+0035 U+0036 U+0037 U+002C U+0038 U+0039"
        },
        {
          "id": "numbers-04",
          "text": "1 234 567,89",
          "note": "Number, Swedish (no-break space U+00A0)",
          "lang": "sv-SE",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 14
          },
          "codepoints": "U+0031 U+00A0 U+0032 U+0033 U+0034 U+00A0 U+0035 U+0036 U+0037 U+002C U+0038 U+0039"
        },
        {
          "id": "numbers-05",
          "text": "1’234’567.89",
          "note": "Number, Swiss German (apostrophe U+2019)",
          "lang": "de-CH",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 16
          },
          "codepoints": "U+0031 U+2019 U+0032 U+0033 U+0034 U+2019 U+0035 U+0036 U+0037 U+002E U+0038 U+0039"
        },
        {
          "id": "numbers-06",
          "text": "12,34,567.89",
          "note": "Number, Indian grouping (lakh)",
          "lang": "en-IN",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 12
          },
          "codepoints": "U+0031 U+0032 U+002C U+0033 U+0034 U+002C U+0035 U+0036 U+0037 U+002E U+0038 U+0039"
        },
        {
          "id": "numbers-07",
          "text": "١٬٢٣٤٬٥٦٧٫٨٩",
          "note": "Number, Arabic-Indic digits and separators",
          "lang": "ar-EG",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 24
          },
          "codepoints": "U+0661 U+066C U+0662 U+0663 U+0664 U+066C U+0665 U+0666 U+0667 U+066B U+0668 U+0669"
        },
        {
          "id": "numbers-08",
          "text": "۱٬۲۳۴٬۵۶۷٫۸۹",
          "note": "Number, Persian digits",
          "lang": "fa-IR",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 24
          },
          "codepoints": "U+06F1 U+066C U+06F2 U+06F3 U+06F4 U+066C U+06F5 U+06F6 U+06F7 U+066B U+06F8 U+06F9"
        },
        {
          "id": "numbers-09",
          "text": "१२,३४,५६७.८९",
          "note": "Number, Devanagari digits with lakh grouping",
          "lang": "hi-IN",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 30
          },
          "codepoints": "U+0967 U+0968 U+002C U+0969 U+096A U+002C U+096B U+096C U+096D U+002E U+096E U+096F"
        },
        {
          "id": "numbers-10",
          "text": "１２３，４５６",
          "note": "Number, full-width digits",
          "lang": "ja-JP",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 21
          },
          "codepoints": "U+FF11 U+FF12 U+FF13 U+FF0C U+FF14 U+FF15 U+FF16"
        },
        {
          "id": "numbers-11",
          "text": "$1,234.56",
          "note": "Currency, US dollars",
          "lang": "en-US",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0024 U+0031 U+002C U+0032 U+0033 U+0034 U+002E U+0035 U+0036"
        },
        {
          "id": "numbers-12",
          "text": "1.234,56 €",
          "note": "Currency, euro in Germany",
          "lang": "de-DE",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 13
          },
          "codepoints": "U+0031 U+002E U+0032 U+0033 U+0034 U+002C U+0035 U+0036 U+00A0 U+20AC"
        },
        {
          "id": "numbers-13",
          "text": "€ 1.234,56",
          "note": "Currency, euro in the Netherlands (symbol first)",
          "lang": "nl-NL",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 13
          },
          "codepoints": "U+20AC U+00A0 U+0031 U+002E U+0032 U+0033 U+0034 U+002C U+0035 U+0036"
        },
        {
          "id": "numbers-14",
          "text": "1 234,56 kr",
          "note": "Currency, Swedish krona",
          "lang": "sv-SE",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0031 U+00A0 U+0032 U+0033 U+0034 U+002C U+0035 U+0036 U+00A0 U+006B U+0072"
        },
        {
          "id": "numbers-15",
          "text": "kr 1 234,56",
          "note": "Currency, Norwegian krone (symbol first)",
          "lang": "nb-NO",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+006B U+0072 U+00A0 U+0031 U+00A0 U+0032 U+0033 U+0034 U+002C U+0035 U+0036"
        },
        {
          "id": "numbers-16",
          "text": "¥1,235",
          "note": "Currency, Japanese yen (no decimals)",
          "lang": "ja-JP",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 7
          },
          "codepoints": "U+00A5 U+0031 U+002C U+0032 U+0033 U+0035"
        },
        {
          "id": "numbers-17",
          "text": "₹12,34,567.89",
          "note": "Currency, Indian rupee",
          "lang": "en-IN",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 15
          },
          "codepoints": "U+20B9 U+0031 U+0032 U+002C U+0033 U+0034 U+002C U+0035 U+0036 U+0037 U+002E U+0038 U+0039"
        },
        {
          "id": "numbers-18",
          "text": "CHF 1’234.50",
          "note": "Currency, Swiss franc",
          "lang": "de-CH",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 15
          },
          "codepoints": "U+0043 U+0048 U+0046 U+00A0 U+0031 U+2019 U+0032 U+0033 U+0034 U+002E U+0035 U+0030"
        },
        {
          "id": "numbers-19",
          "text": "R$ 1.234,56",
          "note": "Currency, Brazilian real",
          "lang": "pt-BR",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 12
          },
          "codepoints": "U+0052 U+0024 U+00A0 U+0031 U+002E U+0032 U+0033 U+0034 U+002C U+0035 U+0036"
        },
        {
          "id": "numbers-20",
          "text": "−1 234,56 kr",
          "note": "Negative with real minus sign U+2212 (not a hyphen)",
          "lang": "sv-SE",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 16
          },
          "codepoints": "U+2212 U+0031 U+00A0 U+0032 U+0033 U+0034 U+002C U+0035 U+0036 U+00A0 U+006B U+0072"
        },
        {
          "id": "numbers-21",
          "text": "(1,234.56)",
          "note": "Accounting-style negative",
          "lang": "en-US",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 10
          },
          "codepoints": "U+0028 U+0031 U+002C U+0032 U+0033 U+0034 U+002E U+0035 U+0036 U+0029"
        },
        {
          "id": "numbers-22",
          "text": "12 %",
          "note": "Percent, French (narrow no-break space)",
          "lang": "fr-FR",
          "length": {
            "codepoints": 4,
            "utf16_units": 4,
            "utf8_bytes": 6
          },
          "codepoints": "U+0031 U+0032 U+202F U+0025"
        },
        {
          "id": "numbers-23",
          "text": "١٢٪",
          "note": "Percent, Arabic-Indic",
          "lang": "ar-EG",
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 6
          },
          "codepoints": "U+0661 U+0662 U+066A"
        },
        {
          "id": "numbers-24",
          "text": "2026-10-08",
          "note": "Date, ISO 8601",
          "lang": null,
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 10
          },
          "codepoints": "U+0032 U+0030 U+0032 U+0036 U+002D U+0031 U+0030 U+002D U+0030 U+0038"
        },
        {
          "id": "numbers-25",
          "text": "10/8/2026",
          "note": "Date, US (month first)",
          "lang": "en-US",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0031 U+0030 U+002F U+0038 U+002F U+0032 U+0030 U+0032 U+0036"
        },
        {
          "id": "numbers-26",
          "text": "08/10/2026",
          "note": "Date, UK (day first): same digits, different date",
          "lang": "en-GB",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 10
          },
          "codepoints": "U+0030 U+0038 U+002F U+0031 U+0030 U+002F U+0032 U+0030 U+0032 U+0036"
        },
        {
          "id": "numbers-27",
          "text": "08.10.2026",
          "note": "Date, German",
          "lang": "de-DE",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 10
          },
          "codepoints": "U+0030 U+0038 U+002E U+0031 U+0030 U+002E U+0032 U+0030 U+0032 U+0036"
        },
        {
          "id": "numbers-28",
          "text": "8.10.2026",
          "note": "Date, Finnish",
          "lang": "fi-FI",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0038 U+002E U+0031 U+0030 U+002E U+0032 U+0030 U+0032 U+0036"
        },
        {
          "id": "numbers-29",
          "text": "8. okt. 2026",
          "note": "Date, Norwegian",
          "lang": "nb-NO",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 12
          },
          "codepoints": "U+0038 U+002E U+0020 U+006F U+006B U+0074 U+002E U+0020 U+0032 U+0030 U+0032 U+0036"
        },
        {
          "id": "numbers-30",
          "text": "jeudi 8 octobre 2026",
          "note": "Date, French long form",
          "lang": "fr-FR",
          "length": {
            "codepoints": 20,
            "utf16_units": 20,
            "utf8_bytes": 20
          },
          "codepoints": "U+006A U+0065 U+0075 U+0064 U+0069 U+0020 U+0038 U+0020 U+006F U+0063 U+0074 U+006F U+0062 U+0072 U+0065 U+0020 U+0032 U+0030 U+0032 U+0036"
        },
        {
          "id": "numbers-31",
          "text": "2026年10月8日",
          "note": "Date, Japanese / Chinese",
          "lang": "ja-JP",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 16
          },
          "codepoints": "U+0032 U+0030 U+0032 U+0036 U+5E74 U+0031 U+0030 U+6708 U+0038 U+65E5"
        },
        {
          "id": "numbers-32",
          "text": "2026. 10. 8.",
          "note": "Date, Korean",
          "lang": "ko-KR",
          "length": {
            "codepoints": 12,
            "utf16_units": 12,
            "utf8_bytes": 12
          },
          "codepoints": "U+0032 U+0030 U+0032 U+0036 U+002E U+0020 U+0031 U+0030 U+002E U+0020 U+0038 U+002E"
        },
        {
          "id": "numbers-33",
          "text": "٨‏/١٠‏/٢٠٢٦",
          "note": "Date, Arabic with RLM marks (U+200F)",
          "lang": "ar-EG",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 22
          },
          "codepoints": "U+0668 U+200F U+002F U+0661 U+0660 U+200F U+002F U+0662 U+0660 U+0662 U+0666"
        },
        {
          "id": "numbers-34",
          "text": "8 ต.ค. 2569",
          "note": "Date, Thai Buddhist calendar (year 2569)",
          "lang": "th-TH",
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 15
          },
          "codepoints": "U+0038 U+0020 U+0E15 U+002E U+0E04 U+002E U+0020 U+0032 U+0035 U+0036 U+0039"
        },
        {
          "id": "numbers-35",
          "text": "14:05",
          "note": "Time, 24-hour",
          "lang": null,
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 5
          },
          "codepoints": "U+0031 U+0034 U+003A U+0030 U+0035"
        },
        {
          "id": "numbers-36",
          "text": "2:05 PM",
          "note": "Time, 12-hour with space",
          "lang": "en-US",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 7
          },
          "codepoints": "U+0032 U+003A U+0030 U+0035 U+0020 U+0050 U+004D"
        },
        {
          "id": "numbers-37",
          "text": "2:05 PM",
          "note": "Time, 12-hour with narrow no-break space (ICU 72+ output, breaks naive parsers)",
          "lang": "en-US",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 9
          },
          "codepoints": "U+0032 U+003A U+0030 U+0035 U+202F U+0050 U+004D"
        },
        {
          "id": "numbers-38",
          "text": "14.05",
          "note": "Time, Finnish (dot separator)",
          "lang": "fi-FI",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 5
          },
          "codepoints": "U+0031 U+0034 U+002E U+0030 U+0035"
        },
        {
          "id": "numbers-39",
          "text": "午後2時05分",
          "note": "Time, Japanese",
          "lang": "ja-JP",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 15
          },
          "codepoints": "U+5348 U+5F8C U+0032 U+6642 U+0030 U+0035 U+5206"
        },
        {
          "id": "numbers-40",
          "text": "9007199254740993",
          "note": "2^53 + 1: not exactly representable as a double",
          "lang": null,
          "length": {
            "codepoints": 16,
            "utf16_units": 16,
            "utf8_bytes": 16
          },
          "codepoints": "U+0039 U+0030 U+0030 U+0037 U+0031 U+0039 U+0039 U+0032 U+0035 U+0034 U+0037 U+0034 U+0030 U+0039 U+0039 U+0033"
        },
        {
          "id": "numbers-41",
          "text": "0.30000000000000004",
          "note": "0.1 + 0.2 in binary floating point",
          "lang": null,
          "length": {
            "codepoints": 19,
            "utf16_units": 19,
            "utf8_bytes": 19
          },
          "codepoints": "U+0030 U+002E U+0033 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0030 U+0034"
        },
        {
          "id": "numbers-42",
          "text": "1e309",
          "note": "Overflows to Infinity as a double",
          "lang": null,
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 5
          },
          "codepoints": "U+0031 U+0065 U+0033 U+0030 U+0039"
        },
        {
          "id": "numbers-43",
          "text": "-0",
          "note": "Negative zero",
          "lang": null,
          "length": {
            "codepoints": 2,
            "utf16_units": 2,
            "utf8_bytes": 2
          },
          "codepoints": "U+002D U+0030"
        },
        {
          "id": "numbers-44",
          "text": "NaN",
          "note": "Not a number (as text)",
          "lang": null,
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 3
          },
          "codepoints": "U+004E U+0061 U+004E"
        },
        {
          "id": "numbers-45",
          "text": "6.02214076e23",
          "note": "Scientific notation",
          "lang": null,
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 13
          },
          "codepoints": "U+0036 U+002E U+0030 U+0032 U+0032 U+0031 U+0034 U+0030 U+0037 U+0036 U+0065 U+0032 U+0033"
        },
        {
          "id": "numbers-46",
          "text": "007",
          "note": "Leading zeros (lost when parsed as a number)",
          "lang": null,
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 3
          },
          "codepoints": "U+0030 U+0030 U+0037"
        },
        {
          "id": "numbers-47",
          "text": "1_000_000",
          "note": "Underscore digit separators",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 9
          },
          "codepoints": "U+0031 U+005F U+0030 U+0030 U+0030 U+005F U+0030 U+0030 U+0030"
        },
        {
          "id": "numbers-48",
          "text": "Ⅻ ½ ⅞ ² ₃",
          "note": "Roman numeral, fractions, superscript and subscript digits",
          "lang": null,
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 17
          },
          "codepoints": "U+216B U+0020 U+00BD U+0020 U+215E U+0020 U+00B2 U+0020 U+2083"
        }
      ]
    },
    "casing": {
      "description": "Case-mapping traps: Turkish I, sharp s, final sigma, digraphs, ligatures",
      "items": [
        {
          "id": "casing-01",
          "text": "İstanbul ılık",
          "note": "Turkish dotted capital I and dotless small i: locale-sensitive case mapping",
          "lang": "tr",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 16
          },
          "codepoints": "U+0130 U+0073 U+0074 U+0061 U+006E U+0062 U+0075 U+006C U+0020 U+0131 U+006C U+0131 U+006B"
        },
        {
          "id": "casing-02",
          "text": "straße STRASSE ẞ",
          "note": "Sharp s upper-cases to SS; capital sharp s U+1E9E exists",
          "lang": "de",
          "length": {
            "codepoints": 16,
            "utf16_units": 16,
            "utf8_bytes": 19
          },
          "codepoints": "U+0073 U+0074 U+0072 U+0061 U+00DF U+0065 U+0020 U+0053 U+0054 U+0052 U+0041 U+0053 U+0053 U+0045 U+0020 U+1E9E"
        },
        {
          "id": "casing-03",
          "text": "ΟΔΥΣΣΕΥΣ οδυσσευς",
          "note": "Greek capital sigma lower-cases to ς at word end",
          "lang": "el",
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 33
          },
          "codepoints": "U+039F U+0394 U+03A5 U+03A3 U+03A3 U+0395 U+03A5 U+03A3 U+0020 U+03BF U+03B4 U+03C5 U+03C3 U+03C3 U+03B5 U+03C5 U+03C2"
        },
        {
          "id": "casing-04",
          "text": "ǅemal ǆ Ǆ",
          "note": "Titlecase digraph U+01C5 (three case forms)",
          "lang": "hr",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 12
          },
          "codepoints": "U+01C5 U+0065 U+006D U+0061 U+006C U+0020 U+01C6 U+0020 U+01C4"
        },
        {
          "id": "casing-05",
          "text": "ﬃce",
          "note": "ffi ligature U+FB03: upper-cases to three letters",
          "lang": null,
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 5
          },
          "codepoints": "U+FB03 U+0063 U+0065"
        },
        {
          "id": "casing-06",
          "text": "IJsbeer",
          "note": "Dutch IJ digraph: both letters capitalized",
          "lang": "nl",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 7
          },
          "codepoints": "U+0049 U+004A U+0073 U+0062 U+0065 U+0065 U+0072"
        },
        {
          "id": "casing-07",
          "text": "ŉ",
          "note": "U+0149 upper-cases to two code points",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 2
          },
          "codepoints": "U+0149"
        },
        {
          "id": "casing-08",
          "text": "K ſ",
          "note": "Kelvin sign U+212A and long s U+017F: case-fold to k and s",
          "lang": null,
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 6
          },
          "codepoints": "U+212A U+0020 U+017F"
        },
        {
          "id": "casing-09",
          "text": "և",
          "note": "Armenian ligature ech-yiwn: upper-cases to two letters",
          "lang": "hy",
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 2
          },
          "codepoints": "U+0587"
        },
        {
          "id": "casing-10",
          "text": "ǰ",
          "note": "U+01F0: upper-case form needs a combining mark",
          "lang": null,
          "length": {
            "codepoints": 1,
            "utf16_units": 1,
            "utf8_bytes": 2
          },
          "codepoints": "U+01F0"
        }
      ]
    },
    "confusables": {
      "description": "Look-alike characters: homograph domains, styled letters, dashes and quotes",
      "items": [
        {
          "id": "confusables-01",
          "text": "exаmple.com",
          "note": "Cyrillic a (U+0430) in a Latin domain (IDN homograph)",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 12
          },
          "codepoints": "U+0065 U+0078 U+0430 U+006D U+0070 U+006C U+0065 U+002E U+0063 U+006F U+006D"
        },
        {
          "id": "confusables-02",
          "text": "еxаmplе.com",
          "note": "Cyrillic e and a mixed into a Latin domain",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 14
          },
          "codepoints": "U+0435 U+0078 U+0430 U+006D U+0070 U+006C U+0435 U+002E U+0063 U+006F U+006D"
        },
        {
          "id": "confusables-03",
          "text": "admin admіn",
          "note": "Second word uses Cyrillic i (U+0456)",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 12
          },
          "codepoints": "U+0061 U+0064 U+006D U+0069 U+006E U+0020 U+0061 U+0064 U+006D U+0456 U+006E"
        },
        {
          "id": "confusables-04",
          "text": "0O o0 l1I |",
          "note": "ASCII look-alikes: zero/O, one/l/I, pipe",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 11
          },
          "codepoints": "U+0030 U+004F U+0020 U+006F U+0030 U+0020 U+006C U+0031 U+0049 U+0020 U+007C"
        },
        {
          "id": "confusables-05",
          "text": "𝐁𝐨𝐥𝐝 𝘐𝘵𝘢𝘭𝘪𝘤",
          "note": "Mathematical bold and italic letters used as fake styled text",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 21,
            "utf8_bytes": 41
          },
          "codepoints": "U+1D401 U+1D428 U+1D425 U+1D41D U+0020 U+1D610 U+1D635 U+1D622 U+1D62D U+1D62A U+1D624"
        },
        {
          "id": "confusables-06",
          "text": "ＡＤＭＩＮ",
          "note": "Full-width Latin (NFKC folds it to ADMIN)",
          "lang": null,
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+FF21 U+FF24 U+FF2D U+FF29 U+FF2E"
        },
        {
          "id": "confusables-07",
          "text": "example․com",
          "note": "One-dot leader U+2024 posing as a period",
          "lang": null,
          "length": {
            "codepoints": 11,
            "utf16_units": 11,
            "utf8_bytes": 13
          },
          "codepoints": "U+0065 U+0078 U+0061 U+006D U+0070 U+006C U+0065 U+2024 U+0063 U+006F U+006D"
        },
        {
          "id": "confusables-08",
          "text": "user＠example.com",
          "note": "Full-width @ (U+FF20)",
          "lang": null,
          "length": {
            "codepoints": 16,
            "utf16_units": 16,
            "utf8_bytes": 18
          },
          "codepoints": "U+0075 U+0073 U+0065 U+0072 U+FF20 U+0065 U+0078 U+0061 U+006D U+0070 U+006C U+0065 U+002E U+0063 U+006F U+006D"
        },
        {
          "id": "confusables-09",
          "text": "‐‑‒–—―−-",
          "note": "Eight different dashes and hyphens",
          "lang": null,
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 22
          },
          "codepoints": "U+2010 U+2011 U+2012 U+2013 U+2014 U+2015 U+2212 U+002D"
        },
        {
          "id": "confusables-10",
          "text": "‘’‚‛“”„«»‹›'\"",
          "note": "Quote marks: curly, low, guillemets and ASCII",
          "lang": null,
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 33
          },
          "codepoints": "U+2018 U+2019 U+201A U+201B U+201C U+201D U+201E U+00AB U+00BB U+2039 U+203A U+0027 U+0022"
        }
      ]
    },
    "scripts": {
      "description": "Language names in 30 scripts for font-fallback and line-height checks",
      "items": [
        {
          "id": "scripts-01",
          "text": "Ελληνικά",
          "note": "Greek",
          "lang": "el",
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 16
          },
          "codepoints": "U+0395 U+03BB U+03BB U+03B7 U+03BD U+03B9 U+03BA U+03AC"
        },
        {
          "id": "scripts-02",
          "text": "Русский",
          "note": "Russian (Cyrillic)",
          "lang": "ru",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 14
          },
          "codepoints": "U+0420 U+0443 U+0441 U+0441 U+043A U+0438 U+0439"
        },
        {
          "id": "scripts-03",
          "text": "Українська",
          "note": "Ukrainian",
          "lang": "uk",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 20
          },
          "codepoints": "U+0423 U+043A U+0440 U+0430 U+0457 U+043D U+0441 U+044C U+043A U+0430"
        },
        {
          "id": "scripts-04",
          "text": "Հայերեն",
          "note": "Armenian",
          "lang": "hy",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 14
          },
          "codepoints": "U+0540 U+0561 U+0575 U+0565 U+0580 U+0565 U+0576"
        },
        {
          "id": "scripts-05",
          "text": "ქართული",
          "note": "Georgian",
          "lang": "ka",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 21
          },
          "codepoints": "U+10E5 U+10D0 U+10E0 U+10D7 U+10E3 U+10DA U+10D8"
        },
        {
          "id": "scripts-06",
          "text": "עברית",
          "note": "Hebrew",
          "lang": "he",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 10
          },
          "codepoints": "U+05E2 U+05D1 U+05E8 U+05D9 U+05EA"
        },
        {
          "id": "scripts-07",
          "text": "العربية",
          "note": "Arabic",
          "lang": "ar",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 14
          },
          "codepoints": "U+0627 U+0644 U+0639 U+0631 U+0628 U+064A U+0629"
        },
        {
          "id": "scripts-08",
          "text": "ދިވެހި",
          "note": "Dhivehi (Thaana, RTL)",
          "lang": "dv",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 12
          },
          "codepoints": "U+078B U+07A8 U+0788 U+07AC U+0780 U+07A8"
        },
        {
          "id": "scripts-09",
          "text": "हिन्दी",
          "note": "Hindi (Devanagari)",
          "lang": "hi",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 18
          },
          "codepoints": "U+0939 U+093F U+0928 U+094D U+0926 U+0940"
        },
        {
          "id": "scripts-10",
          "text": "বাংলা",
          "note": "Bengali",
          "lang": "bn",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+09AC U+09BE U+0982 U+09B2 U+09BE"
        },
        {
          "id": "scripts-11",
          "text": "ਪੰਜਾਬੀ",
          "note": "Punjabi (Gurmukhi)",
          "lang": "pa",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 18
          },
          "codepoints": "U+0A2A U+0A70 U+0A1C U+0A3E U+0A2C U+0A40"
        },
        {
          "id": "scripts-12",
          "text": "ગુજરાતી",
          "note": "Gujarati",
          "lang": "gu",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 21
          },
          "codepoints": "U+0A97 U+0AC1 U+0A9C U+0AB0 U+0ABE U+0AA4 U+0AC0"
        },
        {
          "id": "scripts-13",
          "text": "ଓଡ଼ିଆ",
          "note": "Odia",
          "lang": "or",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+0B13 U+0B21 U+0B3C U+0B3F U+0B06"
        },
        {
          "id": "scripts-14",
          "text": "தமிழ்",
          "note": "Tamil",
          "lang": "ta",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+0BA4 U+0BAE U+0BBF U+0BB4 U+0BCD"
        },
        {
          "id": "scripts-15",
          "text": "తెలుగు",
          "note": "Telugu",
          "lang": "te",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 18
          },
          "codepoints": "U+0C24 U+0C46 U+0C32 U+0C41 U+0C17 U+0C41"
        },
        {
          "id": "scripts-16",
          "text": "ಕನ್ನಡ",
          "note": "Kannada",
          "lang": "kn",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+0C95 U+0CA8 U+0CCD U+0CA8 U+0CA1"
        },
        {
          "id": "scripts-17",
          "text": "മലയാളം",
          "note": "Malayalam",
          "lang": "ml",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 18
          },
          "codepoints": "U+0D2E U+0D32 U+0D2F U+0D3E U+0D33 U+0D02"
        },
        {
          "id": "scripts-18",
          "text": "සිංහල",
          "note": "Sinhala",
          "lang": "si",
          "length": {
            "codepoints": 5,
            "utf16_units": 5,
            "utf8_bytes": 15
          },
          "codepoints": "U+0DC3 U+0DD2 U+0D82 U+0DC4 U+0DBD"
        },
        {
          "id": "scripts-19",
          "text": "ภาษาไทย",
          "note": "Thai",
          "lang": "th",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 21
          },
          "codepoints": "U+0E20 U+0E32 U+0E29 U+0E32 U+0E44 U+0E17 U+0E22"
        },
        {
          "id": "scripts-20",
          "text": "ພາສາລາວ",
          "note": "Lao",
          "lang": "lo",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 21
          },
          "codepoints": "U+0E9E U+0EB2 U+0EAA U+0EB2 U+0EA5 U+0EB2 U+0EA7"
        },
        {
          "id": "scripts-21",
          "text": "ភាសាខ្មែរ",
          "note": "Khmer",
          "lang": "km",
          "length": {
            "codepoints": 9,
            "utf16_units": 9,
            "utf8_bytes": 27
          },
          "codepoints": "U+1797 U+17B6 U+179F U+17B6 U+1781 U+17D2 U+1798 U+17C2 U+179A"
        },
        {
          "id": "scripts-22",
          "text": "မြန်မာဘာသာ",
          "note": "Burmese (tall glyphs)",
          "lang": "my",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 30
          },
          "codepoints": "U+1019 U+103C U+1014 U+103A U+1019 U+102C U+1018 U+102C U+101E U+102C"
        },
        {
          "id": "scripts-23",
          "text": "བོད་ཡིག",
          "note": "Tibetan (tall stacks)",
          "lang": "bo",
          "length": {
            "codepoints": 7,
            "utf16_units": 7,
            "utf8_bytes": 21
          },
          "codepoints": "U+0F56 U+0F7C U+0F51 U+0F0B U+0F61 U+0F72 U+0F42"
        },
        {
          "id": "scripts-24",
          "text": "አማርኛ",
          "note": "Amharic (Ethiopic)",
          "lang": "am",
          "length": {
            "codepoints": 4,
            "utf16_units": 4,
            "utf8_bytes": 12
          },
          "codepoints": "U+12A0 U+121B U+122D U+129B"
        },
        {
          "id": "scripts-25",
          "text": "ᐃᓄᒃᑎᑐᑦ",
          "note": "Inuktitut (Canadian syllabics)",
          "lang": "iu",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 18
          },
          "codepoints": "U+1403 U+14C4 U+1483 U+144E U+1450 U+1466"
        },
        {
          "id": "scripts-26",
          "text": "ᏣᎳᎩ",
          "note": "Cherokee",
          "lang": "chr",
          "length": {
            "codepoints": 3,
            "utf16_units": 3,
            "utf8_bytes": 9
          },
          "codepoints": "U+13E3 U+13B3 U+13A9"
        },
        {
          "id": "scripts-27",
          "text": "ⵜⴰⵎⴰⵣⵉⵖⵜ",
          "note": "Tamazight (Tifinagh)",
          "lang": "zgh",
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 24
          },
          "codepoints": "U+2D5C U+2D30 U+2D4E U+2D30 U+2D63 U+2D49 U+2D56 U+2D5C"
        },
        {
          "id": "scripts-28",
          "text": "ᠮᠣᠩᠭᠣᠯ",
          "note": "Mongolian script (traditionally vertical)",
          "lang": "mn",
          "length": {
            "codepoints": 6,
            "utf16_units": 6,
            "utf8_bytes": 18
          },
          "codepoints": "U+182E U+1823 U+1829 U+182D U+1823 U+182F"
        },
        {
          "id": "scripts-29",
          "text": "Tiếng Việt",
          "note": "Vietnamese",
          "lang": "vi",
          "length": {
            "codepoints": 10,
            "utf16_units": 10,
            "utf8_bytes": 14
          },
          "codepoints": "U+0054 U+0069 U+1EBF U+006E U+0067 U+0020 U+0056 U+0069 U+1EC7 U+0074"
        },
        {
          "id": "scripts-30",
          "text": "𐌰𐌱𐌲",
          "note": "Gothic (astral plane, rarely in fonts)",
          "lang": "got",
          "length": {
            "codepoints": 3,
            "utf16_units": 6,
            "utf8_bytes": 12
          },
          "codepoints": "U+10330 U+10331 U+10332"
        }
      ]
    },
    "pseudo": {
      "description": "Pseudo-localized UI strings (accented, ~35% longer, bracketed)",
      "items": [
        {
          "id": "pseudo-01",
          "text": "[Şȧȧṽḗḗ]",
          "note": "Pseudo-localized: Save",
          "lang": "en-XA",
          "length": {
            "codepoints": 8,
            "utf16_units": 8,
            "utf8_bytes": 17
          },
          "codepoints": "U+005B U+015E U+0227 U+0227 U+1E7D U+1E17 U+1E17 U+005D"
        },
        {
          "id": "pseudo-02",
          "text": "[ȦȦḓḓ ŧǿǿ ƈȧȧřŧ~]",
          "note": "Pseudo-localized: Add to cart",
          "lang": "en-XA",
          "length": {
            "codepoints": 17,
            "utf16_units": 17,
            "utf8_bytes": 31
          },
          "codepoints": "U+005B U+0226 U+0226 U+1E13 U+1E13 U+0020 U+0167 U+01FF U+01FF U+0020 U+0188 U+0227 U+0227 U+0159 U+0167 U+007E U+005D"
        },
        {
          "id": "pseudo-03",
          "text": "[Şḗḗŧŧīīƞɠş~]",
          "note": "Pseudo-localized: Settings",
          "lang": "en-XA",
          "length": {
            "codepoints": 13,
            "utf16_units": 13,
            "utf8_bytes": 25
          },
          "codepoints": "U+005B U+015E U+1E17 U+1E17 U+0167 U+0167 U+012B U+012B U+019E U+0260 U+015F U+007E U+005D"
        },
        {
          "id": "pseudo-04",
          "text": "[Ḓḗḗŀḗḗŧḗḗ ȧȧƈƈǿǿŭŭƞŧ?]",
          "note": "Pseudo-localized: Delete account?",
          "lang": "en-XA",
          "length": {
            "codepoints": 23,
            "utf16_units": 23,
            "utf8_bytes": 49
          },
          "codepoints": "U+005B U+1E12 U+1E17 U+1E17 U+0140 U+1E17 U+1E17 U+0167 U+1E17 U+1E17 U+0020 U+0227 U+0227 U+0188 U+0188 U+01FF U+01FF U+016D U+016D U+019E U+0167 U+003F U+005D"
        },
        {
          "id": "pseudo-05",
          "text": "[Şīīɠƞ īīƞ ẇīīŧħ ḗḗḿȧȧīīŀ]",
          "note": "Pseudo-localized: Sign in with email",
          "lang": "en-XA",
          "length": {
            "codepoints": 26,
            "utf16_units": 26,
            "utf8_bytes": 51
          },
          "codepoints": "U+005B U+015E U+012B U+012B U+0260 U+019E U+0020 U+012B U+012B U+019E U+0020 U+1E87 U+012B U+012B U+0167 U+0127 U+0020 U+1E17 U+1E17 U+1E3F U+0227 U+0227 U+012B U+012B U+0140 U+005D"
        },
        {
          "id": "pseudo-06",
          "text": "[Ẇḗḗŀƈǿǿḿḗḗ ƀȧȧƈķ, {name}!~]",
          "note": "Pseudo-localized: Welcome back, {name}!",
          "lang": "en-XA",
          "length": {
            "codepoints": 28,
            "utf16_units": 28,
            "utf8_bytes": 49
          },
          "codepoints": "U+005B U+1E86 U+1E17 U+1E17 U+0140 U+0188 U+01FF U+01FF U+1E3F U+1E17 U+1E17 U+0020 U+0180 U+0227 U+0227 U+0188 U+0137 U+002C U+0020 U+007B U+006E U+0061 U+006D U+0065 U+007D U+0021 U+007E U+005D"
        },
        {
          "id": "pseudo-07",
          "text": "[Ẏǿǿŭŭ ħȧȧṽḗḗ %d ƞḗḗẇ ḿḗḗşşȧȧɠḗḗş]",
          "note": "Pseudo-localized: You have %d new messages",
          "lang": "en-XA",
          "length": {
            "codepoints": 34,
            "utf16_units": 34,
            "utf8_bytes": 72
          },
          "codepoints": "U+005B U+1E8E U+01FF U+01FF U+016D U+016D U+0020 U+0127 U+0227 U+0227 U+1E7D U+1E17 U+1E17 U+0020 U+0025 U+0064 U+0020 U+019E U+1E17 U+1E17 U+1E87 U+0020 U+1E3F U+1E17 U+1E17 U+015F U+015F U+0227 U+0227 U+0260 U+1E17 U+1E17 U+015F U+005D"
        },
        {
          "id": "pseudo-08",
          "text": "[{count} īīŧḗḗḿş şḗḗŀḗḗƈŧḗḗḓ]",
          "note": "Pseudo-localized: {count} items selected",
          "lang": "en-XA",
          "length": {
            "codepoints": 29,
            "utf16_units": 29,
            "utf8_bytes": 57
          },
          "codepoints": "U+005B U+007B U+0063 U+006F U+0075 U+006E U+0074 U+007D U+0020 U+012B U+012B U+0167 U+1E17 U+1E17 U+1E3F U+015F U+0020 U+015F U+1E17 U+1E17 U+0140 U+1E17 U+1E17 U+0188 U+0167 U+1E17 U+1E17 U+1E13 U+005D"
        },
        {
          "id": "pseudo-09",
          "text": "[Ƒřḗḗḗḗ şħīīƥƥīīƞɠ ǿǿƞ ǿǿřḓḗḗřş ǿǿṽḗḗř $50~]",
          "note": "Pseudo-localized: Free shipping on orders over $50",
          "lang": "en-XA",
          "length": {
            "codepoints": 44,
            "utf16_units": 44,
            "utf8_bytes": 87
          },
          "codepoints": null
        },
        {
          "id": "pseudo-10",
          "text": "[Řḗḗȧȧḓ ǿǿŭŭř <a href=\"/terms\">ŧḗḗřḿş ǿǿƒ şḗḗřṽīīƈḗḗ</a>.]",
          "note": "Pseudo-localized: Read our <a href=\"/terms\">terms of service</a>.",
          "lang": "en-XA",
          "length": {
            "codepoints": 58,
            "utf16_units": 58,
            "utf8_bytes": 99
          },
          "codepoints": null
        },
        {
          "id": "pseudo-11",
          "text": "[Ẏǿǿŭŭř ƥȧȧşşẇǿǿřḓ ḿŭŭşŧ ƀḗḗ ȧȧŧ ŀḗḗȧȧşŧ 8 ƈħȧȧřȧȧƈŧḗḗřş ŀǿǿƞɠ.~~]",
          "note": "Pseudo-localized: Your password must be at least 8 characters long.",
          "lang": "en-XA",
          "length": {
            "codepoints": 66,
            "utf16_units": 66,
            "utf8_bytes": 128
          },
          "codepoints": null
        },
        {
          "id": "pseudo-12",
          "text": "[ȦȦřḗḗ ẏǿǿŭŭ şŭŭřḗḗ ẏǿǿŭŭ ẇȧȧƞŧ ŧǿǿ ŀḗḗȧȧṽḗḗ ŧħīīş ƥȧȧɠḗḗ?]",
          "note": "Pseudo-localized: Are you sure you want to leave this page?",
          "lang": "en-XA",
          "length": {
            "codepoints": 59,
            "utf16_units": 59,
            "utf8_bytes": 121
          },
          "codepoints": null
        }
      ]
    }
  }
}