This is the mail archive of the java-patches@gcc.gnu.org mailing list for the Java project.


Index Nav: [Date Index] [Subject Index] [Author Index] [Thread Index]
Message Nav: [Date Prev] [Date Next] [Thread Prev] [Thread Next]

Patch: character set names -vs- case


David Brownell points out that IANA character set names are
case-insensitive by definition.  This patch implements this.
I'm checking this in.

2001-06-25  Tom Tromey  <tromey@redhat.com>

	* scripts/encodings.pl: Generate lower-case names.  Updated URL
	for `character-sets' file.
	* gnu/gcj/convert/IOConverter.java (canonicalize): Convert name to
	lower case.
	Rebuilt list of aliases.

Tom

Index: scripts/encodings.pl
===================================================================
RCS file: /cvs/gcc/gcc/libjava/scripts/encodings.pl,v
retrieving revision 1.2
diff -u -r1.2 encodings.pl
--- encodings.pl	2000/11/01 17:00:02	1.2
+++ encodings.pl	2001/06/26 04:34:30
@@ -17,7 +17,7 @@
     if (! -f $file)
     {
 	# Too painful to figure out how to get Perl to do it.
-	system 'wget -o .wget-log http://www.isi.edu/in-notes/iana/assignments/character-sets';
+	system 'wget -o .wget-log http://www.iana.org/assignments/character-sets';
     }
 }
 else
@@ -42,12 +42,15 @@
     }
 
     ($type, $name) = split (/\s+/);
+    # Encoding names are case-insensitive.  We do all processing on
+    # the lower-case form.
+    my $lower = lc ($name);
     if ($type eq 'Name:')
     {
 	$current = $map{$name};
 	if ($current)
 	{
-	    print "    hash.put (\"$name\", \"$current\");\n";
+	    print "    hash.put (\"$lower\", \"$current\");\n";
 	}
     }
     elsif ($type eq 'Alias:')
@@ -55,7 +58,7 @@
 	# The IANA list has some ugliness.
 	if ($name ne '' && $name ne 'NONE' && $current)
 	{
-	    print "    hash.put (\"$name\", \"$current\");\n";
+	    print "    hash.put (\"$lower\", \"$current\");\n";
 	}
     }
 }
Index: gnu/gcj/convert/IOConverter.java
===================================================================
RCS file: /cvs/gcc/gcc/libjava/gnu/gcj/convert/IOConverter.java,v
retrieving revision 1.2
diff -u -r1.2 IOConverter.java
--- IOConverter.java	2000/11/01 17:00:01	1.2
+++ IOConverter.java	2001/06/26 04:34:30
@@ -1,4 +1,4 @@
-/* Copyright (C) 2000  Free Software Foundation
+/* Copyright (C) 2000, 2001  Free Software Foundation
 
    This file is part of libgcj.
 
@@ -29,33 +29,34 @@
     hash.put ("ISO-Latin-1", "8859_1");
     // All aliases after this point are automatically generated by the
     // `encodings.pl' script.  Run it to make any corrections.
-    hash.put ("ANSI_X3.4-1968", "ASCII");
+    hash.put ("ansi_x3.4-1968", "ASCII");
     hash.put ("iso-ir-6", "ASCII");
-    hash.put ("ANSI_X3.4-1986", "ASCII");
-    hash.put ("ISO_646.irv:1991", "ASCII");
-    hash.put ("ASCII", "ASCII");
-    hash.put ("ISO646-US", "ASCII");
-    hash.put ("US-ASCII", "ASCII");
+    hash.put ("ansi_x3.4-1986", "ASCII");
+    hash.put ("iso_646.irv:1991", "ASCII");
+    hash.put ("ascii", "ASCII");
+    hash.put ("iso646-us", "ASCII");
+    hash.put ("us-ascii", "ASCII");
     hash.put ("us", "ASCII");
-    hash.put ("IBM367", "ASCII");
+    hash.put ("ibm367", "ASCII");
     hash.put ("cp367", "ASCII");
-    hash.put ("csASCII", "ASCII");
-    hash.put ("ISO_8859-1:1987", "8859_1");
+    hash.put ("csascii", "ASCII");
+    hash.put ("iso_8859-1:1987", "8859_1");
     hash.put ("iso-ir-100", "8859_1");
-    hash.put ("ISO_8859-1", "8859_1");
-    hash.put ("ISO-8859-1", "8859_1");
+    hash.put ("iso_8859-1", "8859_1");
+    hash.put ("iso-8859-1", "8859_1");
     hash.put ("latin1", "8859_1");
     hash.put ("l1", "8859_1");
-    hash.put ("IBM819", "8859_1");
-    hash.put ("CP819", "8859_1");
-    hash.put ("csISOLatin1", "8859_1");
-    hash.put ("UTF-8", "UTF8");
-    hash.put ("Shift_JIS", "SJIS");
-    hash.put ("MS_Kanji", "SJIS");
-    hash.put ("csShiftJIS", "SJIS");
-    hash.put ("Extended_UNIX_Code_Packed_Format_for_Japanese", "EUCJIS");
-    hash.put ("csEUCPkdFmtJapanese", "EUCJIS");
-    hash.put ("EUC-JP", "EUCJIS");
+    hash.put ("ibm819", "8859_1");
+    hash.put ("cp819", "8859_1");
+    hash.put ("csisolatin1", "8859_1");
+    hash.put ("utf-8", "UTF8");
+    hash.put ("none", "UTF8");
+    hash.put ("shift_jis", "SJIS");
+    hash.put ("ms_kanji", "SJIS");
+    hash.put ("csshiftjis", "SJIS");
+    hash.put ("extended_unix_code_packed_format_for_japanese", "EUCJIS");
+    hash.put ("cseucpkdfmtjapanese", "EUCJIS");
+    hash.put ("euc-jp", "EUCJIS");
 
     iconv_byte_swap = iconv_init ();
   }
@@ -65,7 +66,7 @@
   // Turn an alias into the canonical form.
   protected static final String canonicalize (String name)
   {
-    String c = (String) hash.get (name);
+    String c = (String) hash.get (name.toLowerCase ());
     return c == null ? name : c;
   }
 }


Index Nav: [Date Index] [Subject Index] [Author Index] [Thread Index]
Message Nav: [Date Prev] [Date Next] [Thread Prev] [Thread Next]