# How to update the Unicode files # # 1. Update CompositionExclusions.txt by copying original as is # 2. Update GraphemeBreakProperty.txt by copying original as is # 3. Update SpecialCasing.txt by removing comments and conditional mappings from original # 4. Update WhiteSpace.txt by copying the proper excerpt from PropList.txt # 5. Update GraphemeBreakTest.txt and run graphemes_test.exs # 6. Update String.Unicode.version/0 and on String module docs # defmodule String.Unicode do @moduledoc false def version, do: {9, 0, 0} cluster_path = Path.join(__DIR__, "GraphemeBreakProperty.txt") regex = ~r/(?:^([0-9A-F]+)(?:\.\.([0-9A-F]+))?)\s+;\s(\w+)/m cluster = Enum.reduce File.stream!(cluster_path), %{}, fn line, acc -> case Regex.run(regex, line, capture: :all_but_first) do ["D800", "DFFF", _class] -> acc [first, "", class] -> codepoint = <> Map.update(acc, class, [codepoint], &[<> | &1]) [first, last, class] -> range = String.to_integer(first, 16)..String.to_integer(last, 16) codepoints = Enum.map(range, fn int -> <> end) Map.update(acc, class, codepoints, &(codepoints ++ &1)) nil -> acc end end # Don't break CRLF def next_grapheme_size(<>) do {2, rest} end # Break on control for codepoint <- cluster["CR"] ++ cluster["LF"] ++ cluster["Control"] do def next_grapheme_size(<>) do {unquote(byte_size(codepoint)), rest} end end # Break on Prepend* for codepoint <- cluster["Prepend"] do def next_grapheme_size(<>) do next_prepend_size(rest, unquote(byte_size(codepoint))) end end # Handle Regional for codepoint <- cluster["Regional_Indicator"] do def next_grapheme_size(<>) do next_regional_size(rest, unquote(byte_size(codepoint))) end end # Handle Hangul L for codepoint <- cluster["L"] do def next_grapheme_size(<>) do next_hangul_l_size(rest, unquote(byte_size(codepoint))) end end # Handle Hangul V for codepoint <- cluster["LV"] ++ cluster["V"] do def next_grapheme_size(<>) do next_hangul_v_size(rest, unquote(byte_size(codepoint))) end end # Handle Hangul T for codepoint <- cluster["LVT"] ++ cluster["T"] do def next_grapheme_size(<>) do next_hangul_t_size(rest, unquote(byte_size(codepoint))) end end # Handle E_Base for codepoint <- cluster["E_Base"] ++ cluster["E_Base_GAZ"] do def next_grapheme_size(<>) do next_extend_size(rest, unquote(byte_size(codepoint)), :e_base) end end # Handle ZWJ for codepoint <- cluster["ZWJ"] do def next_grapheme_size(<>) do next_extend_size(rest, unquote(byte_size(codepoint)), :zwj) end end # Handle extended entries def next_grapheme_size(<>) do case cp do x when x <= 0x007F -> next_extend_size(rest, 1, :other) x when x <= 0x07FF -> next_extend_size(rest, 2, :other) x when x <= 0xFFFF -> next_extend_size(rest, 3, :other) _ -> next_extend_size(rest, 4, :other) end end def next_grapheme_size(<<_, rest::binary>>) do {1, rest} end def next_grapheme_size(<<>>) do nil end # Handle hanguls defp next_hangul_l_size(rest, size) do case next_hangul(rest, size) do {:l, rest, size} -> next_hangul_l_size(rest, size) {:v, rest, size} -> next_hangul_v_size(rest, size) {:lv, rest, size} -> next_hangul_v_size(rest, size) {:lvt, rest, size} -> next_hangul_t_size(rest, size) _ -> next_extend_size(rest, size, :other) end end defp next_hangul_v_size(rest, size) do case next_hangul(rest, size) do {:v, rest, size} -> next_hangul_v_size(rest, size) {:t, rest, size} -> next_hangul_t_size(rest, size) _ -> next_extend_size(rest, size, :other) end end defp next_hangul_t_size(rest, size) do case next_hangul(rest, size) do {:t, rest, size} -> next_hangul_t_size(rest, size) _ -> next_extend_size(rest, size, :other) end end for codepoint <- cluster["L"] do defp next_hangul(<>, size) do {:l, rest, size + unquote(byte_size(codepoint))} end end for codepoint <- cluster["V"] do defp next_hangul(<>, size) do {:v, rest, size + unquote(byte_size(codepoint))} end end for codepoint <- cluster["T"] do defp next_hangul(<>, size) do {:t, rest, size + unquote(byte_size(codepoint))} end end for codepoint <- cluster["LV"] do defp next_hangul(<>, size) do {:lv, rest, size + unquote(byte_size(codepoint))} end end for codepoint <- cluster["LVT"] do defp next_hangul(<>, size) do {:lvt, rest, size + unquote(byte_size(codepoint))} end end defp next_hangul(_, _) do false end # Handle regional for codepoint <- cluster["Regional_Indicator"] do defp next_regional_size(<>, size) do next_extend_size(rest, size + unquote(byte_size(codepoint)), :other) end end defp next_regional_size(rest, size) do next_extend_size(rest, size, :other) end # Handle Extend+SpacingMark+ZWJ for codepoint <- cluster["Extend"] do defp next_extend_size(<>, size, marker) do next_extend_size(rest, size + unquote(byte_size(codepoint)), keep_ebase(marker)) end end for codepoint <- cluster["SpacingMark"] do defp next_extend_size(<>, size, _marker) do next_extend_size(rest, size + unquote(byte_size(codepoint)), :other) end end for codepoint <- cluster["ZWJ"] do defp next_extend_size(<>, size, _marker) do next_extend_size(rest, size + unquote(byte_size(codepoint)), :zwj) end end for codepoint <- cluster["E_Modifier"] do defp next_extend_size(<>, size, :e_base) do next_extend_size(rest, size + unquote(byte_size(codepoint)), :other) end end for codepoint <- cluster["Glue_After_Zwj"] do defp next_extend_size(<>, size, :zwj) do next_extend_size(rest, size + unquote(byte_size(codepoint)), :other) end end for codepoint <- cluster["E_Base_GAZ"] do defp next_extend_size(<>, size, :zwj) do next_extend_size(rest, size + unquote(byte_size(codepoint)), :e_base) end end defp next_extend_size(rest, size, _) do {size, rest} end defp keep_ebase(:e_base), do: :e_base defp keep_ebase(_), do: :other # Handle Prepend for codepoint <- cluster["Prepend"] do defp next_prepend_size(<>, size) do next_prepend_size(rest, size + unquote(byte_size(codepoint))) end end # However, if we see a control character, we have to break it for codepoint <- cluster["CR"] ++ cluster["LF"] ++ cluster["Control"] do defp next_prepend_size(<> = rest, size) do {size, rest} end end defp next_prepend_size(rest, size) do case next_grapheme_size(rest) do {more, rest} -> {more + size, rest} nil -> {size, rest} end end # Graphemes def graphemes(binary) when is_binary(binary) do do_graphemes(next_grapheme_size(binary), binary) end defp do_graphemes({size, rest}, binary) do [:binary.part(binary, 0, size) | do_graphemes(next_grapheme_size(rest), rest)] end defp do_graphemes(nil, _) do [] end # Length def length(string) do do_length(next_grapheme_size(string), 0) end defp do_length({_, rest}, acc) do do_length(next_grapheme_size(rest), acc + 1) end defp do_length(nil, acc), do: acc # Split at def split_at(string, pos) do do_split_at(string, 0, pos, 0) end defp do_split_at(string, acc, desired_pos, current_pos) when desired_pos > current_pos do case next_grapheme_size(string) do {count, rest} -> do_split_at(rest, acc + count, desired_pos, current_pos + 1) nil -> {acc, nil} end end defp do_split_at(string, acc, desired_pos, desired_pos) do {acc, string} end # Codepoints def next_codepoint(<>) do {<>, rest} end def next_codepoint(<>) do {<>, rest} end def next_codepoint(<<>>) do nil end def codepoints(binary) when is_binary(binary) do do_codepoints(next_codepoint(binary)) end defp do_codepoints({c, rest}) do [c | do_codepoints(next_codepoint(rest))] end defp do_codepoints(nil) do [] end end to_binary = fn "" -> nil codepoints -> codepoints |> :binary.split(" ", [:global]) |> Enum.map(&<>) |> IO.iodata_to_binary end data_path = Path.join(__DIR__, "UnicodeData.txt") {codes, non_breakable, decompositions, combining_classes} = Enum.reduce File.stream!(data_path), {[], [], %{}, %{}}, fn line, {cacc, wacc, dacc, kacc} -> [codepoint, _name, _category, class, _bidi, decomposition, _numeric_1, _numeric_2, _numeric_3, _bidi_mirror, _unicode_1, _iso, upper, lower, title] = :binary.split(line, ";", [:global]) title = :binary.part(title, 0, byte_size(title) - 1) cacc = if upper != "" or lower != "" or title != "" do [{to_binary.(codepoint), to_binary.(upper), to_binary.(lower), to_binary.(title)} | cacc] else cacc end wacc = case decomposition do "" <> _ -> [to_binary.(codepoint) | wacc] _ -> wacc end dacc = case decomposition do <> when h != ?< -> # Decomposition decomposition = decomposition |> :binary.split(" ", [:global]) |> Enum.map(&String.to_integer(&1, 16)) Map.put(dacc, String.to_integer(codepoint, 16), decomposition) _ -> dacc end kacc = case Integer.parse(class) do {0, ""} -> kacc {n, ""} -> Map.put(kacc, String.to_integer(codepoint, 16), n) end {cacc, wacc, dacc, kacc} end defmodule String.Casing do @moduledoc false special_path = Path.join(__DIR__, "SpecialCasing.txt") codes = Enum.reduce File.stream!(special_path), codes, fn line, acc -> [codepoint, lower, title, upper, _] = :binary.split(line, "; ", [:global]) key = to_binary.(codepoint) :lists.keystore(key, 1, acc, {key, to_binary.(upper), to_binary.(lower), to_binary.(title)}) end # Downcase def downcase(string), do: downcase(string, "") for {codepoint, _upper, lower, _title} <- codes, lower && lower != codepoint do defp downcase(unquote(codepoint) <> rest, acc) do downcase(rest, acc <> unquote(lower)) end end defp downcase(<>, acc) do downcase(rest, <>) end defp downcase("", acc), do: acc # Upcase def upcase(string), do: upcase(string, "") for {codepoint, upper, _lower, _title} <- codes, upper && upper != codepoint do defp upcase(unquote(codepoint) <> rest, acc) do upcase(rest, acc <> unquote(upper)) end end defp upcase(<>, acc) do upcase(rest, <>) end defp upcase("", acc), do: acc # Titlecase once def titlecase_once(""), do: {"", ""} for {codepoint, _upper, _lower, title} <- codes, title && title != codepoint do def titlecase_once(unquote(codepoint) <> rest) do {unquote(title), rest} end end def titlecase_once(<>) do {<>, rest} end end defmodule String.Break do @moduledoc false @whitespace_max_size 3 prop_path = Path.join(__DIR__, "WhiteSpace.txt") whitespace = Enum.reduce File.stream!(prop_path), [], fn line, acc -> case line |> :binary.split(";") |> hd do <> -> first = String.to_integer(first, 16) last = String.to_integer(last, 16) Enum.map(first..last, fn int -> <> end) ++ acc <> -> [<> | acc] end end # trim_leading for codepoint <- whitespace do def trim_leading(unquote(codepoint) <> rest), do: trim_leading(rest) end def trim_leading(""), do: "" def trim_leading(string) when is_binary(string), do: string # trim_trailing for codepoint <- whitespace do # We need to increment @whitespace_max_size as well # as the small table (_s) if we add a new entry here. case byte_size(codepoint) do 3 -> defp do_trim_trailing_l(unquote(codepoint)), do: -3 2 -> defp do_trim_trailing_l(<<_, unquote(codepoint)>>), do: -2 defp do_trim_trailing_s(unquote(codepoint)), do: <<>> 1 -> defp do_trim_trailing_l(<>), do: -3 defp do_trim_trailing_l(<<_, unquote(codepoint), unquote(codepoint)>>), do: -2 defp do_trim_trailing_l(<<_, _, unquote(codepoint)>>), do: -1 defp do_trim_trailing_s(<>), do: do_trim_trailing_s(<>) defp do_trim_trailing_s(unquote(codepoint)), do: <<>> end end defp do_trim_trailing_l(_), do: 0 defp do_trim_trailing_s(o), do: o def trim_trailing(string) when is_binary(string) do trim_trailing(string, byte_size(string)) end defp trim_trailing(string, size) when size < @whitespace_max_size do do_trim_trailing_s(string) end defp trim_trailing(string, size) do trail = binary_part(string, size, -@whitespace_max_size) case do_trim_trailing_l(trail) do 0 -> string x -> trim_trailing(binary_part(string, 0, size + x), size + x) end end # Split def split(string) do for piece <- :binary.split(string, unquote(whitespace -- non_breakable), [:global]), piece != "", do: piece end # Decompose def decompose(entries, map) do for entry <- entries do case map do %{^entry => match} -> decompose(match, map) %{} -> <> end end end end defmodule String.Normalizer do @moduledoc false exclusions_path = Path.join(__DIR__, "CompositionExclusions.txt") compositions = Enum.reduce File.stream!(exclusions_path), decompositions, fn <> = line, acc when h in ?0..?9 or h in ?A..?F -> [codepoint, _] = :binary.split(line, " ") Map.delete(acc, String.to_integer(codepoint, 16)) _, acc -> acc end # Normalize def normalize(string, :nfd) when is_binary(string) do normalize_nfd(string, "") end def normalize(string, :nfc) when is_binary(string) do normalize_nfc(string, "") end defp normalize_nfd("", acc), do: acc defp normalize_nfd(<>, acc) when cp in 0xAC00..0xD7A3 do {syllable_index, t_count, n_count} = {cp - 0xAC00, 28, 588} lead = 0x1100 + div(syllable_index, n_count) vowel = 0x1161 + div(rem(syllable_index, n_count), t_count) trail = 0x11A7 + rem(syllable_index, t_count) binary = if trail == 0x11A7 do <> else <> end normalize_nfd(rest, acc <> binary) end defp normalize_nfd(binary, acc) do {n, rest} = String.Unicode.next_grapheme_size(binary) part = :binary.part(binary, 0, n) case n do 1 -> normalize_nfd(rest, acc <> part) _ -> normalize_nfd(rest, acc <> canonical_order(part, [])) end end defp normalize_nfc("", acc), do: acc defp normalize_nfc(<>, acc) when cp in 0xAC00..0xD7A3 do normalize_nfc(rest, acc <> <>) end defp normalize_nfc(binary, acc) do {n, rest} = String.Unicode.next_grapheme_size(binary) part = :binary.part(binary, 0, n) case n do 1 -> normalize_nfc(rest, acc <> part) _ -> normalize_nfc(rest, acc <> compose(normalize_nfd(part, ""))) end end for {cp, decomposition} <- decompositions do decomposition = decomposition |> String.Break.decompose(decompositions) |> IO.iodata_to_binary() defp canonical_order(unquote(<>) <> rest, acc) do canonical_order(unquote(decomposition) <> rest, acc) end end defp canonical_order(<>, acc) do case combining_class(h) do 0 -> canonical_order(acc) <> canonical_order(t, [{h, 0}]) n -> canonical_order(t, [{h, n} | acc]) end end defp canonical_order(<<>>, acc) do canonical_order(acc) end defp canonical_order([{x, _}]) do <> end defp canonical_order(acc) do :lists.keysort(2, Enum.reverse(acc)) |> Enum.map(&<>) |> IO.iodata_to_binary end for {codepoint, class} <- combining_classes do defp combining_class(unquote(codepoint)), do: unquote(class) end defp combining_class(_), do: 0 defp compose(<>) when lead in 0x1100..0x1112 and vowel in 0x1161..0x1175 do codepoint = 0xAC00 + ((lead - 0x1100) * 588) + ((vowel - 0x1161) * 28) case rest do <> when trail in 0x11A7..0x11C2 -> <> _ -> <> end end defp compose(binary) do compose_one(binary) || ( <> = binary compose_many(rest, <>, "", combining_class(cp) - 1) ) end defp compose_many("", base, accents, _), do: base <> accents defp compose_many(<>, base, accents, last_class) do part_class = combining_class(cp) combined = <> if composed = (last_class < part_class && compose_one(combined)) do compose_many(rest, composed, accents, last_class) else compose_many(rest, base, <>, part_class) end end # Compositions: # 1. We must exclude compositions with a single codepoint # 2. We must exclude compositions that do not start with 0 combining class for {cp, [fst, snd]} <- compositions, Map.get(combining_classes, fst, 0) == 0 do defp compose_one(unquote(<>)), do: unquote(<>) end defp compose_one(_), do: nil end