{
  "version": "v1",
  "kind": "names",
  "description": "Person names from 1 to 100+ characters: apostrophes, hyphens, particles, diacritics, scripts",
  "count": 41,
  "seed": null,
  "items": [
    {
      "id": "names-01",
      "text": "Ng",
      "note": "2-character surname",
      "lang": null,
      "length": {
        "codepoints": 2,
        "utf16_units": 2,
        "utf8_bytes": 2
      },
      "codepoints": "U+004E U+0067"
    },
    {
      "id": "names-02",
      "text": "Li Na",
      "note": "Short two-part name",
      "lang": null,
      "length": {
        "codepoints": 5,
        "utf16_units": 5,
        "utf8_bytes": 5
      },
      "codepoints": "U+004C U+0069 U+0020 U+004E U+0061"
    },
    {
      "id": "names-03",
      "text": "O",
      "note": "1-character name (exists)",
      "lang": null,
      "length": {
        "codepoints": 1,
        "utf16_units": 1,
        "utf8_bytes": 1
      },
      "codepoints": "U+004F"
    },
    {
      "id": "names-04",
      "text": "O'Connor",
      "note": "ASCII apostrophe (SQL quoting)",
      "lang": "en",
      "length": {
        "codepoints": 8,
        "utf16_units": 8,
        "utf8_bytes": 8
      },
      "codepoints": "U+004F U+0027 U+0043 U+006F U+006E U+006E U+006F U+0072"
    },
    {
      "id": "names-05",
      "text": "Mary-Jane O’Neill",
      "note": "Hyphen and typographic apostrophe (U+2019)",
      "lang": "en",
      "length": {
        "codepoints": 17,
        "utf16_units": 17,
        "utf8_bytes": 19
      },
      "codepoints": "U+004D U+0061 U+0072 U+0079 U+002D U+004A U+0061 U+006E U+0065 U+0020 U+004F U+2019 U+004E U+0065 U+0069 U+006C U+006C"
    },
    {
      "id": "names-06",
      "text": "D'Angelo-Rossi",
      "note": "Apostrophe and hyphen",
      "lang": "it",
      "length": {
        "codepoints": 14,
        "utf16_units": 14,
        "utf8_bytes": 14
      },
      "codepoints": "U+0044 U+0027 U+0041 U+006E U+0067 U+0065 U+006C U+006F U+002D U+0052 U+006F U+0073 U+0073 U+0069"
    },
    {
      "id": "names-07",
      "text": "Siobhán Ní Bhriain",
      "note": "Irish name with particle",
      "lang": "ga",
      "length": {
        "codepoints": 18,
        "utf16_units": 18,
        "utf8_bytes": 20
      },
      "codepoints": "U+0053 U+0069 U+006F U+0062 U+0068 U+00E1 U+006E U+0020 U+004E U+00ED U+0020 U+0042 U+0068 U+0072 U+0069 U+0061 U+0069 U+006E"
    },
    {
      "id": "names-08",
      "text": "José María Fernández de la Cruz y Sánchez",
      "note": "Spanish multi-part name with particles",
      "lang": "es",
      "length": {
        "codepoints": 41,
        "utf16_units": 41,
        "utf8_bytes": 45
      },
      "codepoints": null
    },
    {
      "id": "names-09",
      "text": "Zoë Ångström-Øverli",
      "note": "Diaeresis, ring and slashed O",
      "lang": null,
      "length": {
        "codepoints": 19,
        "utf16_units": 19,
        "utf8_bytes": 23
      },
      "codepoints": "U+005A U+006F U+00EB U+0020 U+00C5 U+006E U+0067 U+0073 U+0074 U+0072 U+00F6 U+006D U+002D U+00D8 U+0076 U+0065 U+0072 U+006C U+0069"
    },
    {
      "id": "names-10",
      "text": "Zoë Möller",
      "note": "Decomposed diacritics (NFD): compare with NFC before matching",
      "lang": null,
      "length": {
        "codepoints": 12,
        "utf16_units": 12,
        "utf8_bytes": 14
      },
      "codepoints": "U+005A U+006F U+0065 U+0308 U+0020 U+004D U+006F U+0308 U+006C U+006C U+0065 U+0072"
    },
    {
      "id": "names-11",
      "text": "Þórdís Guðmundsdóttir",
      "note": "Icelandic thorn and eth",
      "lang": "is",
      "length": {
        "codepoints": 21,
        "utf16_units": 21,
        "utf8_bytes": 26
      },
      "codepoints": "U+00DE U+00F3 U+0072 U+0064 U+00ED U+0073 U+0020 U+0047 U+0075 U+00F0 U+006D U+0075 U+006E U+0064 U+0073 U+0064 U+00F3 U+0074 U+0074 U+0069 U+0072"
    },
    {
      "id": "names-12",
      "text": "İlkay Işıkoğlu",
      "note": "Turkish dotted/dotless I (case-mapping trap)",
      "lang": "tr",
      "length": {
        "codepoints": 14,
        "utf16_units": 14,
        "utf8_bytes": 18
      },
      "codepoints": "U+0130 U+006C U+006B U+0061 U+0079 U+0020 U+0049 U+015F U+0131 U+006B U+006F U+011F U+006C U+0075"
    },
    {
      "id": "names-13",
      "text": "Jürgen Weiß",
      "note": "Sharp s: upper-cases to WEISS",
      "lang": "de",
      "length": {
        "codepoints": 11,
        "utf16_units": 11,
        "utf8_bytes": 13
      },
      "codepoints": "U+004A U+00FC U+0072 U+0067 U+0065 U+006E U+0020 U+0057 U+0065 U+0069 U+00DF"
    },
    {
      "id": "names-14",
      "text": "Łucja Żółkiewska-Szczęsna",
      "note": "Polish diacritics",
      "lang": "pl",
      "length": {
        "codepoints": 25,
        "utf16_units": 25,
        "utf8_bytes": 30
      },
      "codepoints": "U+0141 U+0075 U+0063 U+006A U+0061 U+0020 U+017B U+00F3 U+0142 U+006B U+0069 U+0065 U+0077 U+0073 U+006B U+0061 U+002D U+0053 U+007A U+0063 U+007A U+0119 U+0073 U+006E U+0061"
    },
    {
      "id": "names-15",
      "text": "Nguyễn Văn An",
      "note": "Vietnamese stacked diacritics",
      "lang": "vi",
      "length": {
        "codepoints": 13,
        "utf16_units": 13,
        "utf8_bytes": 16
      },
      "codepoints": "U+004E U+0067 U+0075 U+0079 U+1EC5 U+006E U+0020 U+0056 U+0103 U+006E U+0020 U+0041 U+006E"
    },
    {
      "id": "names-16",
      "text": "Trần Thị Bích Ngọc",
      "note": "Vietnamese four-part name",
      "lang": "vi",
      "length": {
        "codepoints": 18,
        "utf16_units": 18,
        "utf8_bytes": 25
      },
      "codepoints": "U+0054 U+0072 U+1EA7 U+006E U+0020 U+0054 U+0068 U+1ECB U+0020 U+0042 U+00ED U+0063 U+0068 U+0020 U+004E U+0067 U+1ECD U+0063"
    },
    {
      "id": "names-17",
      "text": "Szabó Éva",
      "note": "Hungarian family-name-first order",
      "lang": "hu",
      "length": {
        "codepoints": 9,
        "utf16_units": 9,
        "utf8_bytes": 11
      },
      "codepoints": "U+0053 U+007A U+0061 U+0062 U+00F3 U+0020 U+00C9 U+0076 U+0061"
    },
    {
      "id": "names-18",
      "text": "Hans-Jürgen van der Meer",
      "note": "Lower-case particle in surname",
      "lang": "nl",
      "length": {
        "codepoints": 24,
        "utf16_units": 24,
        "utf8_bytes": 25
      },
      "codepoints": "U+0048 U+0061 U+006E U+0073 U+002D U+004A U+00FC U+0072 U+0067 U+0065 U+006E U+0020 U+0076 U+0061 U+006E U+0020 U+0064 U+0065 U+0072 U+0020 U+004D U+0065 U+0065 U+0072"
    },
    {
      "id": "names-19",
      "text": "Fiona MacLeod-McAllister",
      "note": "Internal capitals (naive title-casing breaks them)",
      "lang": "en",
      "length": {
        "codepoints": 24,
        "utf16_units": 24,
        "utf8_bytes": 24
      },
      "codepoints": "U+0046 U+0069 U+006F U+006E U+0061 U+0020 U+004D U+0061 U+0063 U+004C U+0065 U+006F U+0064 U+002D U+004D U+0063 U+0041 U+006C U+006C U+0069 U+0073 U+0074 U+0065 U+0072"
    },
    {
      "id": "names-20",
      "text": "Kealoha Nāone-Kaʻula",
      "note": "Hawaiian ʻokina (U+02BB) and macron",
      "lang": "haw",
      "length": {
        "codepoints": 20,
        "utf16_units": 20,
        "utf8_bytes": 22
      },
      "codepoints": "U+004B U+0065 U+0061 U+006C U+006F U+0068 U+0061 U+0020 U+004E U+0101 U+006F U+006E U+0065 U+002D U+004B U+0061 U+02BB U+0075 U+006C U+0061"
    },
    {
      "id": "names-21",
      "text": "Aroha Tāwhiao",
      "note": "Māori macron",
      "lang": "mi",
      "length": {
        "codepoints": 13,
        "utf16_units": 13,
        "utf8_bytes": 14
      },
      "codepoints": "U+0041 U+0072 U+006F U+0068 U+0061 U+0020 U+0054 U+0101 U+0077 U+0068 U+0069 U+0061 U+006F"
    },
    {
      "id": "names-22",
      "text": "Wulandari",
      "note": "Mononym (single name, common in Indonesia)",
      "lang": "id",
      "length": {
        "codepoints": 9,
        "utf16_units": 9,
        "utf8_bytes": 9
      },
      "codepoints": "U+0057 U+0075 U+006C U+0061 U+006E U+0064 U+0061 U+0072 U+0069"
    },
    {
      "id": "names-23",
      "text": "Ngozi Chukwuemeka-Okafor",
      "note": "Igbo name, hyphenated",
      "lang": "ig",
      "length": {
        "codepoints": 24,
        "utf16_units": 24,
        "utf8_bytes": 24
      },
      "codepoints": "U+004E U+0067 U+006F U+007A U+0069 U+0020 U+0043 U+0068 U+0075 U+006B U+0077 U+0075 U+0065 U+006D U+0065 U+006B U+0061 U+002D U+004F U+006B U+0061 U+0066 U+006F U+0072"
    },
    {
      "id": "names-24",
      "text": "Venkataraman Subramaniam Ramachandran",
      "note": "Long South Indian name",
      "lang": "ta",
      "length": {
        "codepoints": 37,
        "utf16_units": 37,
        "utf8_bytes": 37
      },
      "codepoints": "U+0056 U+0065 U+006E U+006B U+0061 U+0074 U+0061 U+0072 U+0061 U+006D U+0061 U+006E U+0020 U+0053 U+0075 U+0062 U+0072 U+0061 U+006D U+0061 U+006E U+0069 U+0061 U+006D U+0020 U+0052 U+0061 U+006D U+0061 U+0063 U+0068 U+0061 U+006E U+0064 U+0072 U+0061 U+006E"
    },
    {
      "id": "names-25",
      "text": "K. S. Ramanathan",
      "note": "Initials with periods",
      "lang": "ta",
      "length": {
        "codepoints": 16,
        "utf16_units": 16,
        "utf8_bytes": 16
      },
      "codepoints": "U+004B U+002E U+0020 U+0053 U+002E U+0020 U+0052 U+0061 U+006D U+0061 U+006E U+0061 U+0074 U+0068 U+0061 U+006E"
    },
    {
      "id": "names-26",
      "text": "Abdullah bin Khalid al-Rashid",
      "note": "Arabic name in Latin script with particles",
      "lang": "en",
      "length": {
        "codepoints": 29,
        "utf16_units": 29,
        "utf8_bytes": 29
      },
      "codepoints": "U+0041 U+0062 U+0064 U+0075 U+006C U+006C U+0061 U+0068 U+0020 U+0062 U+0069 U+006E U+0020 U+004B U+0068 U+0061 U+006C U+0069 U+0064 U+0020 U+0061 U+006C U+002D U+0052 U+0061 U+0073 U+0068 U+0069 U+0064"
    },
    {
      "id": "names-27",
      "text": "عبد الله بن خالد الراشد",
      "note": "Arabic name (RTL)",
      "lang": "ar",
      "length": {
        "codepoints": 23,
        "utf16_units": 23,
        "utf8_bytes": 42
      },
      "codepoints": "U+0639 U+0628 U+062F U+0020 U+0627 U+0644 U+0644 U+0647 U+0020 U+0628 U+0646 U+0020 U+062E U+0627 U+0644 U+062F U+0020 U+0627 U+0644 U+0631 U+0627 U+0634 U+062F"
    },
    {
      "id": "names-28",
      "text": "אברהם בן־דוד",
      "note": "Hebrew name with maqaf (U+05BE)",
      "lang": "he",
      "length": {
        "codepoints": 12,
        "utf16_units": 12,
        "utf8_bytes": 23
      },
      "codepoints": "U+05D0 U+05D1 U+05E8 U+05D4 U+05DD U+0020 U+05D1 U+05DF U+05BE U+05D3 U+05D5 U+05D3"
    },
    {
      "id": "names-29",
      "text": "Анастасия Владимировна Кузнецова",
      "note": "Russian name with patronymic",
      "lang": "ru",
      "length": {
        "codepoints": 32,
        "utf16_units": 32,
        "utf8_bytes": 62
      },
      "codepoints": "U+0410 U+043D U+0430 U+0441 U+0442 U+0430 U+0441 U+0438 U+044F U+0020 U+0412 U+043B U+0430 U+0434 U+0438 U+043C U+0438 U+0440 U+043E U+0432 U+043D U+0430 U+0020 U+041A U+0443 U+0437 U+043D U+0435 U+0446 U+043E U+0432 U+0430"
    },
    {
      "id": "names-30",
      "text": "Αικατερίνη Παπαδοπούλου",
      "note": "Greek name",
      "lang": "el",
      "length": {
        "codepoints": 23,
        "utf16_units": 23,
        "utf8_bytes": 45
      },
      "codepoints": "U+0391 U+03B9 U+03BA U+03B1 U+03C4 U+03B5 U+03C1 U+03AF U+03BD U+03B7 U+0020 U+03A0 U+03B1 U+03C0 U+03B1 U+03B4 U+03BF U+03C0 U+03BF U+03CD U+03BB U+03BF U+03C5"
    },
    {
      "id": "names-31",
      "text": "王小明",
      "note": "Chinese placeholder name (3 characters, no space)",
      "lang": "zh-Hans",
      "length": {
        "codepoints": 3,
        "utf16_units": 3,
        "utf8_bytes": 9
      },
      "codepoints": "U+738B U+5C0F U+660E"
    },
    {
      "id": "names-32",
      "text": "山田太郎",
      "note": "Japanese placeholder name (no space)",
      "lang": "ja",
      "length": {
        "codepoints": 4,
        "utf16_units": 4,
        "utf8_bytes": 12
      },
      "codepoints": "U+5C71 U+7530 U+592A U+90CE"
    },
    {
      "id": "names-33",
      "text": "山田　花子",
      "note": "Japanese name with ideographic space (U+3000)",
      "lang": "ja",
      "length": {
        "codepoints": 5,
        "utf16_units": 5,
        "utf8_bytes": 15
      },
      "codepoints": "U+5C71 U+7530 U+3000 U+82B1 U+5B50"
    },
    {
      "id": "names-34",
      "text": "김민준",
      "note": "Korean name",
      "lang": "ko",
      "length": {
        "codepoints": 3,
        "utf16_units": 3,
        "utf8_bytes": 9
      },
      "codepoints": "U+AE40 U+BBFC U+C900"
    },
    {
      "id": "names-35",
      "text": "สมชาย ใจดี",
      "note": "Thai placeholder name",
      "lang": "th",
      "length": {
        "codepoints": 10,
        "utf16_units": 10,
        "utf8_bytes": 28
      },
      "codepoints": "U+0E2A U+0E21 U+0E0A U+0E32 U+0E22 U+0020 U+0E43 U+0E08 U+0E14 U+0E35"
    },
    {
      "id": "names-36",
      "text": "Alex Null",
      "note": "Surname equal to a reserved word",
      "lang": "en",
      "length": {
        "codepoints": 9,
        "utf16_units": 9,
        "utf8_bytes": 9
      },
      "codepoints": "U+0041 U+006C U+0065 U+0078 U+0020 U+004E U+0075 U+006C U+006C"
    },
    {
      "id": "names-37",
      "text": "Anna 🌸",
      "note": "Name containing an emoji",
      "lang": null,
      "length": {
        "codepoints": 6,
        "utf16_units": 7,
        "utf8_bytes": 9
      },
      "codepoints": "U+0041 U+006E U+006E U+0061 U+0020 U+1F338"
    },
    {
      "id": "names-38",
      "text": "Robert J. Smith-Jones III",
      "note": "Middle initial and generational suffix",
      "lang": "en",
      "length": {
        "codepoints": 25,
        "utf16_units": 25,
        "utf8_bytes": 25
      },
      "codepoints": "U+0052 U+006F U+0062 U+0065 U+0072 U+0074 U+0020 U+004A U+002E U+0020 U+0053 U+006D U+0069 U+0074 U+0068 U+002D U+004A U+006F U+006E U+0065 U+0073 U+0020 U+0049 U+0049 U+0049"
    },
    {
      "id": "names-39",
      "text": "Wolfeschlegelsteinhausenbergerdorff",
      "note": "Very long single surname",
      "lang": "de",
      "length": {
        "codepoints": 35,
        "utf16_units": 35,
        "utf8_bytes": 35
      },
      "codepoints": "U+0057 U+006F U+006C U+0066 U+0065 U+0073 U+0063 U+0068 U+006C U+0065 U+0067 U+0065 U+006C U+0073 U+0074 U+0065 U+0069 U+006E U+0068 U+0061 U+0075 U+0073 U+0065 U+006E U+0062 U+0065 U+0072 U+0067 U+0065 U+0072 U+0064 U+006F U+0072 U+0066 U+0066"
    },
    {
      "id": "names-40",
      "text": "Maximiliana Josefina Alexandra Charlotte Viktoria von und zu Hohenberg-Lichtenfels-Sankt Gallen",
      "note": "Name over 80 characters",
      "lang": "de",
      "length": {
        "codepoints": 95,
        "utf16_units": 95,
        "utf8_bytes": 95
      },
      "codepoints": null
    },
    {
      "id": "names-41",
      "text": "Johanna Marie Wolfeschlegelsteinhausenbergerdorff Senior-Ramírez de la Fuente y Montenegro-Castellanos",
      "note": "Name over 100 characters",
      "lang": null,
      "length": {
        "codepoints": 102,
        "utf16_units": 102,
        "utf8_bytes": 103
      },
      "codepoints": null
    }
  ]
}