Tidy up this developer's test program and add more descriptive comments.

git-svn-id: svn://vcs.exim.org/pcre/code/trunk@946 2f5784b3-3f2a-0410-8824-cb99058d5e15
author: ph10 <ph10@2f5784b3-3f2a-0410-8824-cb99058d5e15> 2012-02-29 18:00:55 +0000
committer: ph10 <ph10@2f5784b3-3f2a-0410-8824-cb99058d5e15> 2012-02-29 18:00:55 +0000
commit: 1a75f94473f2731c1f85923b881b97ceaa9f0d73 (patch)
tree: d592cef4e133bd8b43630c5c3c4b80354010e6df /maint
parent: 3b7e296694f9fbc301b2b3bab17a28b416e8f1f6 (diff)
download: pcre-1a75f94473f2731c1f85923b881b97ceaa9f0d73.tar.gz
1 files changed, 58 insertions, 9 deletions
diff --git a/maint/utf8.c b/maint/utf8.c
index f0897f5..dda84e6 100644
--- a/maint/utf8.c
+++ b/maint/utf8.c
@@ -1,4 +1,29 @@
-/* A program for converting characters to UTF-8 and vice versa */
+/* A test program for converting characters to UTF-8 and vice versa. Note that
+this program conforms to the original definition of UTF-8, which allows
+codepoints up to 7fffffff. The more recent definition limits the validity of
+UTF-8 codepoints to a maximum of 10ffffff.
+
+The arguments are either single codepoint values, written as 0xhhhh, for 
+conversion to UTF-8, or sequences of hex values, written without 0x and 
+optionally including spaces (but such arguments must be quoted), for conversion 
+from UTF-8 to codepoints. For example:
+
+./utf8 0x1234
+0x00001234 => e1 88 b4
+
+./utf8 "e1 88 b4"
+0x00001234 <= e1 88 b4
+
+In the second case, a number of characters can be present in one argument:
+
+./utf8 "65 e188b4 77"
+0x00000065 <= 65 
+0x00001234 <= e1 88 b4 
+0x00000077 <= 77 
+
+If the option -s is given, the sequence of UTF-bytes is written out between 
+angle brackets at the end of the line. On a UTF-8 terminal, this will show the
+appropriate graphic for the codepoint. */
 
 #include <stdio.h>
 #include <stdlib.h>
@@ -137,9 +162,9 @@ main(int argc, char **argv)
 {
 int i = 1;
 int show = 0;
-unsigned char buffer[8];
+unsigned char buffer[64];
 
-if (strcmp(argv[1], "-s") == 0)
+if (argc > 1 && strcmp(argv[1], "-s") == 0)
   {
   show = 1;
   i = 2;
@@ -171,7 +196,9 @@ for (; i < argc; i++)
     int d, rc; 
     int j = 0;
     int y = 0; 
-    int z = 0; 
+    int z = 0;
+    unsigned char *bptr;
+       
     for (;;) 
       { 
       while (*x == ' ') x++; 
@@ -191,11 +218,33 @@ for (; i < argc; i++)
         }
       z ^= 1;     
       } 
-    if (j < 0) continue;     
-    buffer[j] = 0;   
-    rc = utf82ord(buffer, &d);
-    if (rc > 0) printf("0x%08x <= %s\n", d, argv[i]); 
-      else printf("Error %d <= %s\n", rc, argv[i]); 
+    buffer[j] = 0;
+    bptr = buffer;
+
+    while (*bptr != 0)
+      { 
+      rc = utf82ord(bptr, &d);
+      if (rc > 0) 
+        {
+        printf("0x%08x <= ", d);
+        for (j = 0; j < rc; j++) printf("%02x ", bptr[j]);
+        if (show)
+          {
+          printf(">");
+          for (j = 0; j < rc; j++) printf("%c", bptr[j]);
+          printf("<"); 
+          }  
+        printf("\n");
+        bptr += rc; 
+        } 
+      else 
+        {
+        printf("Malformed UTF-8 at offset %d <= ", -rc);
+        while (*bptr != 0) printf("%02x ", *bptr++);
+        printf("\n"); 
+        break;  
+        } 
+      }   
     }       
   } 
 return 0;
author	ph10 <ph10@2f5784b3-3f2a-0410-8824-cb99058d5e15>	2012-02-29 18:00:55 +0000
committer	ph10 <ph10@2f5784b3-3f2a-0410-8824-cb99058d5e15>	2012-02-29 18:00:55 +0000
commit	1a75f94473f2731c1f85923b881b97ceaa9f0d73 (patch)
tree	d592cef4e133bd8b43630c5c3c4b80354010e6df /maint
parent	3b7e296694f9fbc301b2b3bab17a28b416e8f1f6 (diff)
download	pcre-1a75f94473f2731c1f85923b881b97ceaa9f0d73.tar.gz