Merge pull request #49 from JuliaLang/cjh/mw

Try again to update Unicode 8 data
2015-06-26 14:28:30 -04:00
parent 79232c46ea 1cc58b2bc9
commit 7d52470346
4 changed files with 3070 additions and 3125 deletions
--- a/data/charwidths.jl
+++ b/data/charwidths.jl
@@ -14,6 +14,40 @@ end
 CharWidths = Dict{Int,Int}()
 #############################################################################
 # Use ../libutf8proc for category codes, rather than the one in Julia,
 # to minimize bootstrapping complexity when a new version of Unicode comes out.
 catcode(c) = ccall((:utf8proc_category,"../libutf8proc"), Cint, (Int32,), c)
 # use Base.UTF8proc module to get category codes constants, since
 # we won't change these in utf8proc.
 import Base.UTF8proc
 #############################################################################
 # Use a default width of 1 for all character categories that are
 # letter/symbol/number-like.  This can be overriden by Unifont or UAX 11
 # below, but provides a useful nonzero fallback for new codepoints when
 # a new Unicode version has been released but Unifont hasn't been updated yet.
 zerowidth = Set{Int}() # categories that may contain zero-width chars
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_CN)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_MN)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_MC)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_ME)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_SK)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_ZS)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_ZL)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_ZP)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_CC)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_CF)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_CS)
 push!(zerowidth, UTF8proc.UTF8PROC_CATEGORY_CO)
 for c in 0x0000:0x110000
    if catcode(c) ∉ zerowidth
        CharWidths[c] = 1
    end
 end
 #############################################################################
 # Widths from GNU Unifont
@@ -40,7 +74,13 @@ function parsesfd(filename::String, CharWidths::Dict{Int,Int}=Dict{Int,Int}())
            contains(line, "Encoding:") && (codepoint = int(split(line)[3]))
            contains(line, "Width:") && (width = int(split(line)[2]))
            if codepoint!=nothing && width!=nothing && codepoint >= 0
-                CharWidths[codepoint]=div(width, 512) # 512 units to the en
+                w=div(width, 512) # 512 units to the en
                if w > 0
                    # only add nonzero widths, since (1) the default is zero
                    # and (2) this circumvents some apparent bugs in Unifont
                    # (https://savannah.gnu.org/bugs/index.php?45395)
                    CharWidths[codepoint] = w
                end
                state = :seekchar
            end
        end
@@ -84,17 +124,6 @@ end
 # A few exceptions to the above cases, found by manual comparison
 # to other wcwidth functions and similar checks.
 # Use ../libutf8proc for category codes, rather than the one in Julia,
 # to minimize bootstrapping complexity when a new version of Unicode comes out.
 function catcode(c)
    uint(c) > 0x10FFFF && return 0x0000 # see utf8proc_get_property docs
    return unsafe_load(ccall((:utf8proc_get_property,"../libutf8proc"), Ptr{UInt16}, (Int32,), c))
 end
 # use Base.UTF8proc module to get category codes constants, since
 # we won't change these in utf8proc.
 import Base.UTF8proc
 for c in keys(CharWidths)
    cat = catcode(c)
--- a/data/data_generator.rb
+++ b/data/data_generator.rb
@@ -313,8 +313,8 @@ $stdout << "};\n\n"
 $stdout << "const utf8proc_int32_t utf8proc_combinations[] = {\n  "
 i = 0
-comb1st_indicies.keys.each_index do |a|
+comb1st_indicies.keys.sort.each_index do |a|
-  comb2nd_indicies.keys.each_index do |b|
+  comb2nd_indicies.keys.sort.each_index do |b|
    i += 1
    if i == 8
      i = 0
--- a/test/charwidth.c
+++ b/test/charwidth.c
@@ -10,7 +10,7 @@ int my_isprint(int c) {
 int main(int argc, char **argv)
 {
-     int c, error = 0;
+     int c, error = 0, updates = 0;
     (void) argc; /* unused */
     (void) argv; /* unused */
@@ -24,6 +24,13 @@ int main(int argc, char **argv)
               fprintf(stderr, "nonzero width %d for combining char %x\n", w, c);
               error = 1;
          }
          if (w == 0 &&
 			  ((cat >= UTF8PROC_CATEGORY_LU && cat <= UTF8PROC_CATEGORY_LO) ||
 			   (cat >= UTF8PROC_CATEGORY_ND && cat <= UTF8PROC_CATEGORY_SC) ||
 			   (cat >= UTF8PROC_CATEGORY_SO && cat <= UTF8PROC_CATEGORY_ZS))) {
               fprintf(stderr, "zero width for symbol-like char %x\n", c);
               error = 1;
          }
          if (c <= 127 && ((!isprint(c) && w > 0) ||
                           (isprint(c) && wcwidth(c) != w))) {
               fprintf(stderr, "wcwidth %d mismatch %d for %s ASCII %x\n",
@@ -44,17 +51,20 @@ int main(int argc, char **argv)
          int w = utf8proc_charwidth(c);
          int wc = wcwidth(c);
          if (sizeof(wchar_t) == 2 && c >= (1<<16)) continue;
 #if 0
          /* lots of these errors for out-of-date system unicode tables */
-          if (wc == -1 && my_isprint(c) && w > 0)
+          if (wc == -1 && my_isprint(c) && w > 0) {
 			   updates += 1;
 #if 0
               printf("  wcwidth(%x) = -1 for printable char\n", c);
 #endif
 		  }
          if (wc == -1 && !my_isprint(c) && w > 0)
               printf("  wcwidth(%x) = -1 for non-printable width-%d char\n", c, w);
          if (wc >= 0 && wc != w)
               printf("  wcwidth(%x) = %d != charwidth %d\n", c, wc, w);
     }
-
+	 printf("   ... (positive widths for %d chars unknown to wcwidth) ...\n",
 			updates);
     printf("Character-width tests SUCCEEDED.\n");
     return 0;
--- a/utf8proc_data.c
+++ b/utf8proc_data.c