changed names of resource-1.3; added a note on homepage on release

author: aarne <aarne@cs.chalmers.se> 2008-06-25 16:54:35 +0000
committer: aarne <aarne@cs.chalmers.se> 2008-06-25 16:54:35 +0000
commit: e9e80fc389365e24d4300d7d5390c7d833a96c50 (patch)
tree: f0b58473adaa670bd8fc52ada419d8cad470ee03 /src/GF/Text/UTF8.hs
parent: b96b36f43de3e2f8b58d5f539daa6f6d47f25870 (diff)
1 files changed, 48 insertions, 0 deletions
diff --git a/src/GF/Text/UTF8.hs b/src/GF/Text/UTF8.hs
new file mode 100644
index 000000000..5e9687684
--- /dev/null
+++ b/src/GF/Text/UTF8.hs
@@ -0,0 +1,48 @@
+----------------------------------------------------------------------
+-- |
+-- Module      : UTF8
+-- Maintainer  : AR
+-- Stability   : (stable)
+-- Portability : (portable)
+--
+-- > CVS $Date: 2005/04/21 16:23:42 $ 
+-- > CVS $Author: bringert $
+-- > CVS $Revision: 1.5 $
+--
+-- From the Char module supplied with HBC.
+-- code by Thomas Hallgren (Jul 10 1999)
+-----------------------------------------------------------------------------
+
+module GF.Text.UTF8 (decodeUTF8, encodeUTF8) where
+
+-- | Take a Unicode string and encode it as a string
+-- with the UTF8 method.
+decodeUTF8 :: String -> String
+decodeUTF8 "" = ""
+decodeUTF8 (c:cs) | c < '\x80' = c : decodeUTF8 cs
+decodeUTF8 (c:c':cs) | '\xc0' <= c  && c  <= '\xdf' && 
+		      '\x80' <= c' && c' <= '\xbf' =
+	toEnum ((fromEnum c `mod` 0x20) * 0x40 + fromEnum c' `mod` 0x40) : decodeUTF8 cs
+decodeUTF8 (c:c':c'':cs) | '\xe0' <= c   && c   <= '\xef' && 
+		          '\x80' <= c'  && c'  <= '\xbf' &&
+		          '\x80' <= c'' && c'' <= '\xbf' =
+	toEnum ((fromEnum c `mod` 0x10 * 0x1000) + (fromEnum c' `mod` 0x40) * 0x40 + fromEnum c'' `mod` 0x40) : decodeUTF8 cs
+decodeUTF8 s = s ---- AR workaround 22/6/2006
+----decodeUTF8 _ = error "UniChar.decodeUTF8: bad data"
+
+encodeUTF8 :: String -> String
+encodeUTF8 "" = ""
+encodeUTF8 (c:cs) =
+	if c > '\x0000' && c < '\x0080' then
+	    c : encodeUTF8 cs
+	else if c < toEnum 0x0800 then
+	    let i = fromEnum c
+	    in  toEnum (0xc0 + i `div` 0x40) : 
+	        toEnum (0x80 + i `mod` 0x40) : 
+		encodeUTF8 cs
+	else
+	    let i = fromEnum c
+	    in  toEnum (0xe0 + i `div` 0x1000) : 
+	        toEnum (0x80 + (i `mod` 0x1000) `div` 0x40) : 
+		toEnum (0x80 + i `mod` 0x40) : 
+		encodeUTF8 cs
author	aarne <aarne@cs.chalmers.se>	2008-06-25 16:54:35 +0000
committer	aarne <aarne@cs.chalmers.se>	2008-06-25 16:54:35 +0000
commit	e9e80fc389365e24d4300d7d5390c7d833a96c50 (patch)
tree	f0b58473adaa670bd8fc52ada419d8cad470ee03 /src/GF/Text/UTF8.hs
parent	b96b36f43de3e2f8b58d5f539daa6f6d47f25870 (diff)