Import parser construction utility library

svn path=/trunk/libparserutils/; revision=4111
author: John Mark Bell <jmb@netsurf-browser.org> 2008-05-01 16:34:46 +0000
committer: John Mark Bell <jmb@netsurf-browser.org> 2008-05-01 16:34:46 +0000
commit: 2777a04ed2ba4fd36138b991d66a32a283361f7e (patch)
tree: b0c3730533c36ca41402b6d0c5b98413f0a57bee /src/charset/codecs
download: libparserutils-2777a04ed2ba4fd36138b991d66a32a283361f7e.tar.gz
libparserutils-2777a04ed2ba4fd36138b991d66a32a283361f7e.tar.bz2
5 files changed, 1867 insertions, 0 deletions
diff --git a/src/charset/codecs/Makefile b/src/charset/codecs/Makefile
new file mode 100644
index 0000000..6d3b78e
--- /dev/null
+++ b/src/charset/codecs/Makefile
@@ -0,0 +1,46 @@
+# Child makefile fragment
+#
+# Toolchain is provided by top-level makefile
+#
+# Variables provided by top-level makefile
+#
+# COMPONENT		The name of the component
+# EXPORT		The location of the export directory
+# TOP			The location of the source tree root
+# RELEASEDIR		The place to put release objects
+# DEBUGDIR		The place to put debug objects
+#
+# do_include		Canned command sequence to include a child makefile
+#
+# Variables provided by parent makefile:
+#
+# DIR			The name of the directory we're in, relative to $(TOP)
+#
+# Variables we can manipulate:
+#
+# ITEMS_CLEAN		The list of items to remove for "make clean"
+# ITEMS_DISTCLEAN	The list of items to remove for "make distclean"
+# TARGET_TESTS		The list of target names to run for "make test"
+#
+# SOURCES		The list of sources to build for $(COMPONENT)
+#
+# Plus anything from the toolchain
+
+# Push parent directory onto the directory stack
+sp             := $(sp).x
+dirstack_$(sp) := $(d)
+d              := $(DIR)
+
+# Sources
+SRCS_$(d) := codec_iconv.c codec_utf8.c codec_utf16.c
+
+# Append to sources for component
+SOURCES += $(addprefix $(d), $(SRCS_$(d)))
+
+# Now include any children we may have
+MAKE_INCLUDES := $(wildcard $(d)*/Makefile)
+$(eval $(foreach INC, $(MAKE_INCLUDES), $(call do_include,$(INC))))
+
+# Finally, pop off the directory stack
+d  := $(dirstack_$(sp))
+sp := $(basename $(sp))
diff --git a/src/charset/codecs/codec_iconv.c b/src/charset/codecs/codec_iconv.c
new file mode 100644
index 0000000..bbe8bc4
--- /dev/null
+++ b/src/charset/codecs/codec_iconv.c
@@ -0,0 +1,683 @@
+/*
+ * This file is part of LibParserUtils.
+ * Licensed under the MIT License,
+ *                http://www.opensource.org/licenses/mit-license.php
+ * Copyright 2007 John-Mark Bell <jmb@netsurf-browser.org>
+ */
+
+/* This codec is hideously slow. Only use it as a last resort */
+
+#include <errno.h>
+#include <stdlib.h>
+#include <string.h>
+
+/* We put this here rather than at the top as GCC complains 
+ * about the source file being empty otherwise. */
+#ifdef WITH_ICONV_CODEC
+
+#include <iconv.h>
+
+/* These two are for htonl / ntohl */
+#include <arpa/inet.h>
+#include <netinet/in.h>
+
+#include <parserutils/charset/mibenum.h>
+
+#include "charset/codecs/codec_impl.h"
+#include "utils/utils.h"
+
+/**
+ * Iconv-based charset codec
+ */
+typedef struct iconv_codec {
+	parserutils_charset_codec base;	/**< Base class */
+
+	iconv_t read_cd;		/**< Iconv handle for reading */
+#define INVAL_BUFSIZE (32)
+	uint8_t inval_buf[INVAL_BUFSIZE];	/**< Buffer for fixing up
+						 * incomplete input
+						 * sequences */
+	size_t inval_len;		/**< Number of bytes in inval_buf */
+
+#define READ_BUFSIZE (8)
+	uint32_t read_buf[READ_BUFSIZE];	/**< Buffer for partial
+						 * output sequences (decode)
+						 */
+	size_t read_len;		/**< Number of characters in
+					 * read_buf */
+
+	iconv_t write_cd;		/**< Iconv handle for writing */
+#define WRITE_BUFSIZE (8)
+	uint32_t write_buf[WRITE_BUFSIZE];	/**< Buffer for partial
+						 * output sequences (encode)
+						 */
+	size_t write_len;		/**< Number of characters in
+					 * write_buf */
+} iconv_codec;
+
+
+static bool iconv_codec_handles_charset(const char *charset);
+static parserutils_charset_codec *iconv_codec_create(const char *charset,
+		parserutils_alloc alloc, void *pw);
+static void iconv_codec_destroy (parserutils_charset_codec *codec);
+static parserutils_error iconv_codec_encode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error iconv_codec_decode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error iconv_codec_reset(parserutils_charset_codec *codec);
+static parserutils_error iconv_codec_output_decoded_char(
+		iconv_codec *c, uint32_t ucs4, uint8_t **dest,
+		size_t *destlen);
+static parserutils_error iconv_codec_read_char(iconv_codec *c,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error iconv_codec_write_char(iconv_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen);
+
+/**
+ * Determine whether this codec handles a specific charset
+ *
+ * \param charset  Charset to test
+ * \return true if handleable, false otherwise
+ */
+bool iconv_codec_handles_charset(const char *charset)
+{
+	iconv_t cd;
+	bool ret;
+
+	cd = iconv_open("UCS-4", charset);
+
+	ret = (cd != (iconv_t) -1);
+
+	if (ret)
+		iconv_close(cd);
+
+	return ret;
+}
+
+/**
+ * Create an iconv-based codec
+ *
+ * \param charset  The charset to read from / write to
+ * \param alloc    Memory (de)allocation function
+ * \param pw       Pointer to client-specific private data (may be NULL)
+ * \return Pointer to codec, or NULL on failure
+ */
+parserutils_charset_codec *iconv_codec_create(const char *charset,
+		parserutils_alloc alloc, void *pw)
+{
+	iconv_codec *codec;
+
+	codec = alloc(NULL, sizeof(iconv_codec), pw);
+	if (codec == NULL)
+		return NULL;
+
+	codec->read_cd = iconv_open("UCS-4", charset);
+	if (codec->read_cd == (iconv_t) -1) {
+		alloc(codec, 0, pw);
+		return NULL;
+	}
+
+	codec->write_cd = iconv_open(charset, "UCS-4");
+	if (codec->write_cd == (iconv_t) -1) {
+		iconv_close(codec->read_cd);
+		alloc(codec, 0, pw);
+		return NULL;
+	}
+
+	codec->inval_buf[0] = '\0';
+	codec->inval_len = 0;
+
+	codec->read_buf[0] = 0;
+	codec->read_len = 0;
+
+	codec->write_buf[0] = 0;
+	codec->write_len = 0;
+
+	/* Finally, populate vtable */
+	codec->base.handler.destroy = iconv_codec_destroy;
+	codec->base.handler.encode = iconv_codec_encode;
+	codec->base.handler.decode = iconv_codec_decode;
+	codec->base.handler.reset = iconv_codec_reset;
+
+	return (parserutils_charset_codec *) codec;
+}
+
+/**
+ * Destroy an iconv-based codec
+ *
+ * \param codec  The codec to destroy
+ */
+void iconv_codec_destroy (parserutils_charset_codec *codec)
+{
+	iconv_codec *c = (iconv_codec *) codec;
+
+	iconv_close(c->read_cd);
+	iconv_close(c->write_cd);
+
+	return;
+}
+
+/**
+ * Encode a chunk of UCS4 data into an iconv-based codec's charset
+ *
+ * \param codec      The codec to use
+ * \param source     Pointer to pointer to source data
+ * \param sourcelen  Pointer to length (in bytes) of source data
+ * \param dest       Pointer to pointer to output buffer
+ * \param destlen    Pointer to length (in bytes) of output buffer
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                             codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read. Any remaining output for the character will be buffered by the
+ * codec for writing on the next call.
+ *
+ * Note that, if failure occurs whilst attempting to write any output
+ * buffered by the last call, then ::source and ::sourcelen will remain
+ * unchanged (as nothing more has been read).
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ */
+parserutils_error iconv_codec_encode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	iconv_codec *c = (iconv_codec *) codec;
+	uint32_t ucs4;
+	const uint32_t *towrite;
+	size_t towritelen;
+	parserutils_error error;
+
+	/* Process any outstanding characters from the previous call */
+	if (c->write_len > 0) {
+		uint32_t *pwrite = c->write_buf;
+
+		while (c->write_len > 0) {
+			error = iconv_codec_write_char(c, pwrite[0],
+					dest, destlen);
+			if (error != PARSERUTILS_OK) {
+				/* Copy outstanding chars down, skipping
+				 * invalid one, if present, so as to avoid
+				 * reprocessing the invalid character */
+				if (error == PARSERUTILS_INVALID) {
+					for (ucs4 = 1; ucs4 < c->write_len;
+							ucs4++) {
+						c->write_buf[ucs4] =
+								pwrite[ucs4];
+					}
+				}
+
+				return error;
+			}
+
+			pwrite++;
+			c->write_len--;
+		}
+	}
+
+	/* Now process the characters for this call */
+	while (*sourcelen > 0) {
+		towrite = (const uint32_t *) (const void *) *source;
+		towritelen = 1;
+		ucs4 = *towrite;
+
+		/* Output current character(s) */
+		while (towritelen > 0) {
+			error = iconv_codec_write_char(c, towrite[0],
+					dest, destlen);
+
+			if (error != PARSERUTILS_OK) {
+				ucs4 = (error == PARSERUTILS_INVALID) ? 1 : 0;
+
+				if (towritelen - ucs4 >= WRITE_BUFSIZE)
+					abort();
+
+				c->write_len = towritelen - ucs4;
+
+				/* Copy pending chars to save area, for
+				 * processing next call; skipping invalid
+				 * character, if present, so it's not
+				 * reprocessed. */
+				for (; ucs4 < towritelen; ucs4++) {
+					c->write_buf[ucs4] = towrite[ucs4];
+				}
+
+				/* Claim character we've just buffered,
+				 * so it's not repreocessed */
+				*source += 4;
+				*sourcelen -= 4;
+
+				return error;
+			}
+
+			towrite++;
+			towritelen--;
+		}
+
+		*source += 4;
+		*sourcelen -= 4;
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Decode a chunk of data in an iconv-based codec's charset into UCS4
+ *
+ * \param codec      The codec to use
+ * \param source     Pointer to pointer to source data
+ * \param sourcelen  Pointer to length (in bytes) of source data
+ * \param dest       Pointer to pointer to output buffer
+ * \param destlen    Pointer to length (in bytes) of output buffer
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read, if the result is _OK or _NOMEM. Any remaining output for the
+ * character will be buffered by the codec for writing on the next call.
+ *
+ * In the case of the result being _INVALID, ::source will point _at_ the 
+ * last input character read; nothing will be written or buffered for the 
+ * failed character. It is up to the client to fix the cause of the failure 
+ * and retry the decoding process.
+ *
+ * Note that, if failure occurs whilst attempting to write any output
+ * buffered by the last call, then ::source and ::sourcelen will remain
+ * unchanged (as nothing more has been read).
+ *
+ * If STRICT error handling is configured and an illegal sequence is split
+ * over two calls, then _INVALID will be returned from the second call,
+ * but ::source will point mid-way through the invalid sequence (i.e. it
+ * will be unmodified over the second call). In addition, the internal
+ * incomplete-sequence buffer will be emptied, such that subsequent calls
+ * will progress, rather than re-evaluating the same invalid sequence.
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ *
+ * Call this with a source length of 0 to flush the output buffer.
+ */
+parserutils_error iconv_codec_decode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	iconv_codec *c = (iconv_codec *) codec;
+	parserutils_error error;
+
+	if (c->read_len > 0) {
+		/* Output left over from last decode
+		 * Attempt to finish this here */
+		uint32_t *pread = c->read_buf;
+
+		while (c->read_len > 0 && *destlen >= c->read_len * 4) {
+			*((uint32_t *) (void *) *dest) = pread[0];
+
+			*dest += 4;
+			*destlen -= 4;
+
+			pread++;
+			c->read_len--;
+		}
+
+		if (*destlen < c->read_len * 4) {
+			/* Run out of output buffer */
+			size_t i;
+
+			/* Shuffle remaining output down */
+			for (i = 0; i < c->read_len; i++) {
+				c->read_buf[i] = pread[i];
+			}
+
+			return PARSERUTILS_NOMEM;
+		}
+	}
+
+	if (c->inval_len > 0) {
+		/* The last decode ended in an incomplete sequence.
+		 * Fill up inval_buf with data from the start of the
+		 * new chunk and process it. */
+		uint8_t *in = c->inval_buf;
+		size_t ol = c->inval_len;
+		size_t l = min(INVAL_BUFSIZE - ol - 1, *sourcelen);
+		size_t orig_l = l;
+
+		memcpy(c->inval_buf + ol, *source, l);
+
+		l += c->inval_len;
+
+		error = iconv_codec_read_char(c,
+				(const uint8_t **) &in, &l, dest, destlen);
+		if (error != PARSERUTILS_OK && error != PARSERUTILS_NOMEM) {
+			return error;
+		}
+
+
+		/* And now, fix everything up so the normal processing
+		 * does the right thing. */
+		*source += max((signed) (orig_l - l), 0);
+		*sourcelen -= max((signed) (orig_l - l), 0);
+
+		/* Failed to resolve an incomplete character and
+		 * ran out of buffer space. No recovery strategy
+		 * possible, so explode everywhere. */
+		if ((orig_l + ol) - l == 0)
+			abort();
+
+		/* Handle memry exhaustion case from above */
+		if (error != PARSERUTILS_OK)
+			return error;
+	}
+
+	while (*sourcelen > 0) {
+		error = iconv_codec_read_char(c,
+				source, sourcelen, dest, destlen);
+		if (error != PARSERUTILS_OK) {
+			return error;
+		}
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Clear an iconv-based codec's encoding state
+ *
+ * \param codec  The codec to reset
+ * \return PARSERUTILS_OK on success, appropriate error otherwise
+ */
+parserutils_error iconv_codec_reset(parserutils_charset_codec *codec)
+{
+	iconv_codec *c = (iconv_codec *) codec;
+
+	iconv(c->read_cd, NULL, NULL, NULL, NULL);
+	iconv(c->write_cd, NULL, NULL, NULL, NULL);
+
+	c->inval_buf[0] = '\0';
+	c->inval_len = 0;
+
+	c->read_buf[0] = 0;
+	c->read_len = 0;
+
+	c->write_buf[0] = 0;
+	c->write_len = 0;
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Output a UCS4 character
+ *
+ * \param c        Codec to use
+ * \param ucs4     UCS4 character (big endian)
+ * \param dest     Pointer to pointer to output buffer
+ * \param destlen  Pointer to output buffer length
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ */
+parserutils_error iconv_codec_output_decoded_char(iconv_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen)
+{
+	if (*destlen < 4) {
+		/* Run out of output buffer */
+
+		c->read_len = 1;
+		c->read_buf[0] = ucs4;
+
+		return PARSERUTILS_NOMEM;
+	}
+
+	*((uint32_t *) (void *) *dest) = ucs4;
+	*dest += 4;
+	*destlen -= 4;
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Read a character from the codec's native charset to UCS4 (big endian)
+ *
+ * \param c          The codec
+ * \param source     Pointer to pointer to source buffer (updated on exit)
+ * \param sourcelen  Pointer to length of source buffer (updated on exit)
+ * \param dest       Pointer to pointer to output buffer (updated on exit)
+ * \param destlen    Pointer to length of output buffer (updated on exit)
+ * \return PARSERUTILS_OK on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read, if the result is _OK or _NOMEM. Any remaining output for the
+ * character will be buffered by the codec for writing on the next call.
+ *
+ * In the case of the result being _INVALID, ::source will point _at_ the 
+ * last input character read; nothing will be written or buffered for the 
+ * failed character. It is up to the client to fix the cause of the failure 
+ * and retry the decoding process.
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ */
+parserutils_error iconv_codec_read_char(iconv_codec *c,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	size_t iconv_ret;
+	const uint8_t *origsrc = *source;
+	size_t origsrclen = *sourcelen;
+	uint32_t ucs4;
+	uint8_t *pucs4 = (uint8_t *) &ucs4;
+	size_t sucs4 = 4;
+	parserutils_error error;
+
+	/* Use iconv to convert a single character
+	 * Side effect: Updates *source to point at next input
+	 * character and *sourcelen to reflect reduced input length
+	 */
+	iconv_ret = iconv(c->read_cd, (char **) source, sourcelen,
+			(char **) (void *) &pucs4, &sucs4);
+
+	if (iconv_ret != (size_t) -1 ||
+			(*source != origsrc && sucs4 == 0)) {
+		/* Read a character */
+		error = iconv_codec_output_decoded_char(c, ucs4, dest, destlen);
+		if (error != PARSERUTILS_OK && error != PARSERUTILS_NOMEM) {
+			/* output failed; restore source pointers */
+			*source = origsrc;
+			*sourcelen = origsrclen;
+		}
+
+		/* Clear inval buffer */
+		c->inval_buf[0] = '\0';
+		c->inval_len = 0;
+
+		return error;
+	} else if (errno == E2BIG) {
+		/* Should never happen */
+		abort();
+	} else if (errno == EINVAL) {
+		/* Incomplete input sequence */
+		if (*sourcelen > INVAL_BUFSIZE)
+			abort();
+
+		memmove(c->inval_buf, (const char *) *source, *sourcelen);
+		c->inval_buf[*sourcelen] = '\0';
+		c->inval_len = *sourcelen;
+
+		*source += *sourcelen;
+		*sourcelen = 0;
+
+		return PARSERUTILS_OK;
+	} else if (errno == EILSEQ) {
+		/* Illegal input sequence */
+		bool found = false;
+		const uint8_t *oldsrc;
+		size_t oldsrclen;
+
+		/* Clear inval buffer */
+		c->inval_buf[0] = '\0';
+		c->inval_len = 0;
+
+		/* Strict errormode; simply flag invalid character */
+		if (c->base.errormode == 
+				PARSERUTILS_CHARSET_CODEC_ERROR_STRICT) {
+			/* restore source pointers */
+			*source = origsrc;
+			*sourcelen = origsrclen;
+
+			return PARSERUTILS_INVALID;
+		}
+
+		/* Ok, this becomes problematic. The iconv API here
+		* is particularly unhelpful; *source will point at
+		* the _start_ of the illegal sequence. This means
+		* that we must find the end of the sequence */
+
+		/* Search for the start of the next valid input
+		 * sequence (or the end of the input stream) */
+		while (*sourcelen > 1) {
+			pucs4 = (uint8_t *) &ucs4;
+			sucs4 = 4;
+
+			(*source)++;
+			(*sourcelen)--;
+
+			oldsrc = *source;
+			oldsrclen = *sourcelen;
+
+			iconv_ret = iconv(c->read_cd,
+					(char **) source, sourcelen,
+					(char **) (void *) &pucs4, &sucs4);
+			if (iconv_ret != (size_t) -1 || errno != EILSEQ) {
+				found = true;
+				break;
+			}
+		}
+
+		if (found) {
+			/* Found start of next valid sequence */
+			*source = oldsrc;
+			*sourcelen = oldsrclen;
+		} else {
+			/* Not found - skip last byte in buffer */
+			(*source)++;
+			(*sourcelen)--;
+
+			if (*sourcelen != 0)
+				abort();
+		}
+
+		/* output U+FFFD and continue processing. */
+		error = iconv_codec_output_decoded_char(c,
+				htonl(0xFFFD), dest, destlen);
+		if (error != PARSERUTILS_OK && error != PARSERUTILS_NOMEM) {
+			/* output failed; restore source pointers */
+			*source = origsrc;
+			*sourcelen = origsrclen;
+		}
+
+		return error;
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Write a UCS4 character in a codec's native charset
+ *
+ * \param c        The codec
+ * \param ucs4     The UCS4 character to write (big endian)
+ * \param dest     Pointer to pointer to output buffer (updated on exit)
+ * \param destlen  Pointer to length of output buffer (updated on exit)
+ * \return PARSERUTILS_OK       on success,
+ *         PARSERUTILS_NOMEM    if output buffer is too small,
+ *         PARSERUTILS_INVALID  if character cannot be represented and the
+ *                         codec's error handling mode is set to STRICT.
+ */
+parserutils_error iconv_codec_write_char(iconv_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen)
+{
+	size_t iconv_ret;
+	uint8_t *pucs4 = (uint8_t *) &ucs4;
+	size_t sucs4 = 4;
+	uint8_t *origdest = *dest;
+
+	iconv_ret = iconv(c->write_cd, (char **) (void *) &pucs4,
+			&sucs4, (char **) dest, destlen);
+
+	if (iconv_ret == (size_t) -1 && errno == E2BIG) {
+		/* Output buffer is too small */
+		return PARSERUTILS_NOMEM;
+	} else if (iconv_ret == (size_t) -1 && errno == EILSEQ) {
+		/* Illegal multibyte sequence */
+		/* This should never happen */
+		abort();
+	} else if (iconv_ret == (size_t) -1 && errno == EINVAL) {
+		/* Incomplete input character */
+		/* This should never happen */
+		abort();
+	} else if (*dest == origdest) {
+		/* Nothing was output */
+		switch (c->base.errormode) {
+		case PARSERUTILS_CHARSET_CODEC_ERROR_STRICT:
+			return PARSERUTILS_INVALID;
+
+		case PARSERUTILS_CHARSET_CODEC_ERROR_TRANSLIT:
+			/** \todo transliteration */
+		case PARSERUTILS_CHARSET_CODEC_ERROR_LOOSE:
+		{
+			pucs4 = (uint8_t *) &ucs4;
+			sucs4 = 4;
+
+			ucs4 = parserutils_charset_mibenum_is_unicode(
+					c->base.mibenum)
+					? htonl(0xFFFD) : htonl(0x3F);
+
+			iconv_ret = iconv(c->write_cd,
+					(char **) (void *) &pucs4, &sucs4,
+					(char **) dest, destlen);
+
+			if (iconv_ret == (size_t) -1 && errno == E2BIG) {
+				return PARSERUTILS_NOMEM;
+			} else if (iconv_ret == (size_t) -1 &&
+					errno == EILSEQ) {
+				/* Illegal multibyte sequence */
+				/* This should never happen */
+				abort();
+			} else if (iconv_ret == (size_t) -1 &&
+					errno == EINVAL) {
+				/* Incomplete input character */
+				/* This should never happen */
+				abort();
+			}
+		}
+			break;
+		}
+	}
+
+	return PARSERUTILS_OK;
+}
+
+const parserutils_charset_handler iconv_codec_handler = {
+	iconv_codec_handles_charset,
+	iconv_codec_create
+};
+
+#endif
diff --git a/src/charset/codecs/codec_impl.h b/src/charset/codecs/codec_impl.h
new file mode 100644
index 0000000..9183594
--- /dev/null
+++ b/src/charset/codecs/codec_impl.h
@@ -0,0 +1,48 @@
+/*
+ * This file is part of LibParserUtils.
+ * Licensed under the MIT License,
+ *                http://www.opensource.org/licenses/mit-license.php
+ * Copyright 2007 John-Mark Bell <jmb@netsurf-browser.org>
+ */
+
+#ifndef parserutils_charset_codecs_codecimpl_h_
+#define parserutils_charset_codecs_codecimpl_h_
+
+#include <stdbool.h>
+#include <inttypes.h>
+
+#include <parserutils/charset/codec.h>
+
+/**
+ * Core charset codec definition; implementations extend this
+ */
+struct parserutils_charset_codec {
+	uint16_t mibenum;			/**< MIB enum for charset */
+
+	parserutils_charset_codec_errormode errormode;	/**< error mode */
+
+	parserutils_alloc alloc;		/**< allocation function */
+	void *alloc_pw;				/**< private word */
+
+	struct {
+		void (*destroy)(parserutils_charset_codec *codec);
+		parserutils_error (*encode)(parserutils_charset_codec *codec,
+				const uint8_t **source, size_t *sourcelen,
+				uint8_t **dest, size_t *destlen);
+		parserutils_error (*decode)(parserutils_charset_codec *codec,
+				const uint8_t **source, size_t *sourcelen,
+				uint8_t **dest, size_t *destlen);
+		parserutils_error (*reset)(parserutils_charset_codec *codec);
+	} handler; /**< Vtable for handler code */
+};
+
+/**
+ * Codec factory component definition
+ */
+typedef struct parserutils_charset_handler {
+	bool (*handles_charset)(const char *charset);
+	parserutils_charset_codec *(*create)(const char *charset,
+			parserutils_alloc alloc, void *pw);
+} parserutils_charset_handler;
+
+#endif
diff --git a/src/charset/codecs/codec_utf16.c b/src/charset/codecs/codec_utf16.c
new file mode 100644
index 0000000..0dd7a07
--- /dev/null
+++ b/src/charset/codecs/codec_utf16.c
@@ -0,0 +1,544 @@
+/*
+ * This file is part of LibParserUtils.
+ * Licensed under the MIT License,
+ *                http://www.opensource.org/licenses/mit-license.php
+ * Copyright 2007 John-Mark Bell <jmb@netsurf-browser.org>
+ */
+
+#include <stdlib.h>
+#include <string.h>
+
+/* These two are for htonl / ntohl */
+#include <arpa/inet.h>
+#include <netinet/in.h>
+
+#include <parserutils/charset/mibenum.h>
+#include <parserutils/charset/utf16.h>
+
+#include "charset/codecs/codec_impl.h"
+#include "utils/utils.h"
+
+/**
+ * UTF-16 charset codec
+ */
+typedef struct charset_utf16_codec {
+	parserutils_charset_codec base;	/**< Base class */
+
+#define INVAL_BUFSIZE (32)
+	uint8_t inval_buf[INVAL_BUFSIZE];	/**< Buffer for fixing up
+						 * incomplete input
+						 * sequences */
+	size_t inval_len;		/*< Byte length of inval_buf **/
+
+#define READ_BUFSIZE (8)
+	uint32_t read_buf[READ_BUFSIZE];	/**< Buffer for partial
+						 * output sequences (decode)
+						 * (host-endian) */
+	size_t read_len;		/**< Character length of read_buf */
+
+#define WRITE_BUFSIZE (8)
+	uint32_t write_buf[WRITE_BUFSIZE];	/**< Buffer for partial
+						 * output sequences (encode)
+						 * (host-endian) */
+	size_t write_len;		/**< Character length of write_buf */
+
+} charset_utf16_codec;
+
+static bool charset_utf16_codec_handles_charset(const char *charset);
+static parserutils_charset_codec *charset_utf16_codec_create(
+		const char *charset, parserutils_alloc alloc, void *pw);
+static void charset_utf16_codec_destroy (parserutils_charset_codec *codec);
+static parserutils_error charset_utf16_codec_encode(
+		parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error charset_utf16_codec_decode(
+		parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error charset_utf16_codec_reset(
+		parserutils_charset_codec *codec);
+static inline parserutils_error charset_utf16_codec_read_char(
+		charset_utf16_codec *c,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static inline parserutils_error charset_utf16_codec_output_decoded_char(
+		charset_utf16_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen);
+
+/**
+ * Determine whether this codec handles a specific charset
+ *
+ * \param charset  Charset to test
+ * \return true if handleable, false otherwise
+ */
+bool charset_utf16_codec_handles_charset(const char *charset)
+{
+	return parserutils_charset_mibenum_from_name(charset, strlen(charset)) 
+		==
+		parserutils_charset_mibenum_from_name("UTF-16", SLEN("UTF-16"));
+}
+
+/**
+ * Create a utf16 codec
+ *
+ * \param charset  The charset to read from / write to
+ * \param alloc    Memory (de)allocation function
+ * \param pw       Pointer to client-specific private data (may be NULL)
+ * \return Pointer to codec, or NULL on failure
+ */
+parserutils_charset_codec *charset_utf16_codec_create(const char *charset,
+		parserutils_alloc alloc, void *pw)
+{
+	charset_utf16_codec *codec;
+
+	UNUSED(charset);
+
+	codec = alloc(NULL, sizeof(charset_utf16_codec), pw);
+	if (codec == NULL)
+		return NULL;
+
+	codec->inval_buf[0] = '\0';
+	codec->inval_len = 0;
+
+	codec->read_buf[0] = 0;
+	codec->read_len = 0;
+
+	codec->write_buf[0] = 0;
+	codec->write_len = 0;
+
+	/* Finally, populate vtable */
+	codec->base.handler.destroy = charset_utf16_codec_destroy;
+	codec->base.handler.encode = charset_utf16_codec_encode;
+	codec->base.handler.decode = charset_utf16_codec_decode;
+	codec->base.handler.reset = charset_utf16_codec_reset;
+
+	return (parserutils_charset_codec *) codec;
+}
+
+/**
+ * Destroy a utf16 codec
+ *
+ * \param codec  The codec to destroy
+ */
+void charset_utf16_codec_destroy (parserutils_charset_codec *codec)
+{
+	UNUSED(codec);
+}
+
+/**
+ * Encode a chunk of UCS4 data into utf16
+ *
+ * \param codec      The codec to use
+ * \param source     Pointer to pointer to source data
+ * \param sourcelen  Pointer to length (in bytes) of source data
+ * \param dest       Pointer to pointer to output buffer
+ * \param destlen    Pointer to length (in bytes) of output buffer
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read. Any remaining output for the character will be buffered by the
+ * codec for writing on the next call. 
+ *
+ * Note that, if failure occurs whilst attempting to write any output
+ * buffered by the last call, then ::source and ::sourcelen will remain
+ * unchanged (as nothing more has been read).
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ */
+parserutils_error charset_utf16_codec_encode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	charset_utf16_codec *c = (charset_utf16_codec *) codec;
+	uint32_t ucs4;
+	uint32_t *towrite;
+	size_t towritelen;
+	parserutils_error error;
+
+	/* Process any outstanding characters from the previous call */
+	if (c->write_len > 0) {
+		uint32_t *pwrite = c->write_buf;
+		uint8_t buf[4];
+		size_t len;
+
+		while (c->write_len > 0) {
+			error = parserutils_charset_utf16_from_ucs4(
+					pwrite[0], buf, &len);
+			if (error != PARSERUTILS_OK)
+				abort();
+
+			if (*destlen < len) {
+				/* Insufficient output buffer space */
+				for (len = 0; len < c->write_len; len++)
+					c->write_buf[len] = pwrite[len];
+
+				return PARSERUTILS_NOMEM;
+			}
+
+			memcpy(*dest, buf, len);
+
+			*dest += len;
+			*destlen -= len;
+
+			pwrite++;
+			c->write_len--;
+		}
+	}
+
+	/* Now process the characters for this call */
+	while (*sourcelen > 0) {
+		ucs4 = ntohl(*((uint32_t *) (void *) *source));
+		towrite = &ucs4;
+		towritelen = 1;
+
+		/* Output current characters */
+		while (towritelen > 0) {
+			uint8_t buf[4];
+			size_t len;
+
+			error = parserutils_charset_utf16_from_ucs4(
+					towrite[0], buf, &len);
+			if (error != PARSERUTILS_OK)
+				abort();
+
+			if (*destlen < len) {
+				/* Insufficient output space */
+				if (towritelen >= WRITE_BUFSIZE)
+					abort();
+
+				c->write_len = towritelen;
+
+				/* Copy pending chars to save area, for
+				 * processing next call. */
+				for (len = 0; len < towritelen; len++)
+					c->write_buf[len] = towrite[len];
+
+				/* Claim character we've just buffered,
+				 * so it's not reprocessed */
+				*source += 4;
+				*sourcelen -= 4;
+
+				return PARSERUTILS_NOMEM;
+			}
+
+			memcpy(*dest, buf, len);
+
+			*dest += len;
+			*destlen -= len;
+
+			towrite++;
+			towritelen--;
+		}
+
+		*source += 4;
+		*sourcelen -= 4;
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Decode a chunk of utf16 data into UCS4
+ *
+ * \param codec      The codec to use
+ * \param source     Pointer to pointer to source data
+ * \param sourcelen  Pointer to length (in bytes) of source data
+ * \param dest       Pointer to pointer to output buffer
+ * \param destlen    Pointer to length (in bytes) of output buffer
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read, if the result is _OK or _NOMEM. Any remaining output for the
+ * character will be buffered by the codec for writing on the next call.
+ *
+ * In the case of the result being _INVALID, ::source will point _at_ the 
+ * last input character read; nothing will be written or buffered for the 
+ * failed character. It is up to the client to fix the cause of the failure 
+ * and retry the decoding process.
+ *
+ * Note that, if failure occurs whilst attempting to write any output
+ * buffered by the last call, then ::source and ::sourcelen will remain
+ * unchanged (as nothing more has been read).
+ *
+ * If STRICT error handling is configured and an illegal sequence is split
+ * over two calls, then _INVALID will be returned from the second call,
+ * but ::source will point mid-way through the invalid sequence (i.e. it
+ * will be unmodified over the second call). In addition, the internal
+ * incomplete-sequence buffer will be emptied, such that subsequent calls
+ * will progress, rather than re-evaluating the same invalid sequence.
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ *
+ * Call this with a source length of 0 to flush the output buffer.
+ */
+parserutils_error charset_utf16_codec_decode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	charset_utf16_codec *c = (charset_utf16_codec *) codec;
+	parserutils_error error;
+
+	if (c->read_len > 0) {
+		/* Output left over from last decode */
+		uint32_t *pread = c->read_buf;
+
+		while (c->read_len > 0 && *destlen >= c->read_len * 4) {
+			*((uint32_t *) (void *) *dest) = htonl(pread[0]);
+
+			*dest += 4;
+			*destlen -= 4;
+
+			pread++;
+			c->read_len--;
+		}
+
+		if (*destlen < c->read_len * 4) {
+			/* Ran out of output buffer */
+			size_t i;
+
+			/* Shuffle remaining output down */
+			for (i = 0; i < c->read_len; i++)
+				c->read_buf[i] = pread[i];
+
+			return PARSERUTILS_NOMEM;
+		}
+	}
+
+	if (c->inval_len > 0) {
+		/* The last decode ended in an incomplete sequence.
+		 * Fill up inval_buf with data from the start of the
+		 * new chunk and process it. */
+		uint8_t *in = c->inval_buf;
+		size_t ol = c->inval_len;
+		size_t l = min(INVAL_BUFSIZE - ol - 1, *sourcelen);
+		size_t orig_l = l;
+
+		memcpy(c->inval_buf + ol, *source, l);
+
+		l += c->inval_len;
+
+		error = charset_utf16_codec_read_char(c,
+				(const uint8_t **) &in, &l, dest, destlen);
+		if (error != PARSERUTILS_OK && error != PARSERUTILS_NOMEM) {
+			return error;
+		}
+
+		/* And now, fix up source pointers */
+		*source += max((signed) (orig_l - l), 0);
+		*sourcelen -= max((signed) (orig_l - l), 0);
+
+		/* Failed to resolve an incomplete character and
+		 * ran out of buffer space. No recovery strategy
+		 * possible, so explode everywhere. */
+		if ((orig_l + ol) - l == 0)
+			abort();
+
+		/* Report memory exhaustion case from above */
+		if (error != PARSERUTILS_OK)
+			return error;
+	}
+
+	/* Finally, the "normal" case; process all outstanding characters */
+	while (*sourcelen > 0) {
+		error = charset_utf16_codec_read_char(c,
+				source, sourcelen, dest, destlen);
+		if (error != PARSERUTILS_OK) {
+			return error;
+		}
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Clear a utf16 codec's encoding state
+ *
+ * \param codec  The codec to reset
+ * \return PARSERUTILS_OK on success, appropriate error otherwise
+ */
+parserutils_error charset_utf16_codec_reset(parserutils_charset_codec *codec)
+{
+	charset_utf16_codec *c = (charset_utf16_codec *) codec;
+
+	c->inval_buf[0] = '\0';
+	c->inval_len = 0;
+
+	c->read_buf[0] = 0;
+	c->read_len = 0;
+
+	c->write_buf[0] = 0;
+	c->write_len = 0;
+
+	return PARSERUTILS_OK;
+}
+
+
+/**
+ * Read a character from the UTF-16 to UCS4 (big endian)
+ *
+ * \param c          The codec
+ * \param source     Pointer to pointer to source buffer (updated on exit)
+ * \param sourcelen  Pointer to length of source buffer (updated on exit)
+ * \param dest       Pointer to pointer to output buffer (updated on exit)
+ * \param destlen    Pointer to length of output buffer (updated on exit)
+ * \return PARSERUTILS_OK on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read, if the result is _OK or _NOMEM. Any remaining output for the
+ * character will be buffered by the codec for writing on the next call.
+ *
+ * In the case of the result being _INVALID, ::source will point _at_ the 
+ * last input character read; nothing will be written or buffered for the 
+ * failed character. It is up to the client to fix the cause of the failure 
+ * and retry the decoding process.
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ */
+parserutils_error charset_utf16_codec_read_char(charset_utf16_codec *c,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	uint32_t ucs4;
+	size_t sucs4;
+	parserutils_error error;
+
+	/* Convert a single character */
+	error = parserutils_charset_utf16_to_ucs4(*source, *sourcelen, 
+			&ucs4, &sucs4);
+	if (error == PARSERUTILS_OK) {
+		/* Read a character */
+		error = charset_utf16_codec_output_decoded_char(c,
+				ucs4, dest, destlen);
+		if (error == PARSERUTILS_OK || error == PARSERUTILS_NOMEM) {
+			/* output succeeded; update source pointers */
+			*source += sucs4;
+			*sourcelen -= sucs4;
+		}
+
+		/* Clear inval buffer */
+		c->inval_buf[0] = '\0';
+		c->inval_len = 0;
+
+		return error;
+	} else if (error == PARSERUTILS_NEEDDATA) {
+		/* Incomplete input sequence */
+		if (*sourcelen > INVAL_BUFSIZE)
+			abort();
+
+		memmove(c->inval_buf, (char *) *source, *sourcelen);
+		c->inval_buf[*sourcelen] = '\0';
+		c->inval_len = *sourcelen;
+
+		*source += *sourcelen;
+		*sourcelen = 0;
+
+		return PARSERUTILS_OK;
+	} else if (error == PARSERUTILS_INVALID) {
+		/* Illegal input sequence */
+		uint32_t nextchar;
+
+		/* Clear inval buffer */
+		c->inval_buf[0] = '\0';
+		c->inval_len = 0;
+
+		/* Strict errormode; simply flag invalid character */
+		if (c->base.errormode == 
+				PARSERUTILS_CHARSET_CODEC_ERROR_STRICT) {
+			return PARSERUTILS_INVALID;
+		}
+
+		/* Find next valid UTF-16 sequence.
+		 * We're processing client-provided data, so let's
+		 * be paranoid about its validity. */
+		error = parserutils_charset_utf16_next_paranoid(
+				*source, *sourcelen, 0, &nextchar);
+		if (error != PARSERUTILS_OK) {
+			if (error == PARSERUTILS_NEEDDATA) {
+				/* Need more data to be sure */
+				if (*sourcelen > INVAL_BUFSIZE)
+					abort();
+
+				memmove(c->inval_buf, (char *) *source,
+						*sourcelen);
+				c->inval_buf[*sourcelen] = '\0';
+				c->inval_len = *sourcelen;
+
+				*source += *sourcelen;
+				*sourcelen = 0;
+
+				nextchar = 0;
+			} else {
+				return error;
+			}
+		}
+
+		/* output U+FFFD and continue processing. */
+		error = charset_utf16_codec_output_decoded_char(c,
+				0xFFFD, dest, destlen);
+		if (error == PARSERUTILS_OK || error == PARSERUTILS_NOMEM) {
+			/* output succeeded; update source pointers */
+			*source += nextchar;
+			*sourcelen -= nextchar;
+		}
+
+		return error;
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Output a UCS4 character
+ *
+ * \param c        Codec to use
+ * \param ucs4     UCS4 character (host endian)
+ * \param dest     Pointer to pointer to output buffer
+ * \param destlen  Pointer to output buffer length
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ */
+parserutils_error charset_utf16_codec_output_decoded_char(charset_utf16_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen)
+{
+	if (*destlen < 4) {
+		/* Run out of output buffer */
+		c->read_len = 1;
+		c->read_buf[0] = ucs4;
+
+		return PARSERUTILS_NOMEM;
+	}
+
+	*((uint32_t *) (void *) *dest) = htonl(ucs4);
+	*dest += 4;
+	*destlen -= 4;
+
+	return PARSERUTILS_OK;
+}
+
+
+const parserutils_charset_handler charset_utf16_codec_handler = {
+	charset_utf16_codec_handles_charset,
+	charset_utf16_codec_create
+};
diff --git a/src/charset/codecs/codec_utf8.c b/src/charset/codecs/codec_utf8.c
new file mode 100644
index 0000000..838d051
--- /dev/null
+++ b/src/charset/codecs/codec_utf8.c
@@ -0,0 +1,546 @@
+/*
+ * This file is part of LibParserUtils.
+ * Licensed under the MIT License,
+ *                http://www.opensource.org/licenses/mit-license.php
+ * Copyright 2007 John-Mark Bell <jmb@netsurf-browser.org>
+ */
+
+#include <stdlib.h>
+#include <string.h>
+
+/* These two are for htonl / ntohl */
+#include <arpa/inet.h>
+#include <netinet/in.h>
+
+#include <parserutils/charset/mibenum.h>
+
+#include "charset/codecs/codec_impl.h"
+#include "charset/encodings/utf8impl.h"
+#include "utils/utils.h"
+
+/**
+ * UTF-8 charset codec
+ */
+typedef struct charset_utf8_codec {
+	parserutils_charset_codec base;	/**< Base class */
+
+#define INVAL_BUFSIZE (32)
+	uint8_t inval_buf[INVAL_BUFSIZE];	/**< Buffer for fixing up
+						 * incomplete input
+						 * sequences */
+	size_t inval_len;		/*< Byte length of inval_buf **/
+
+#define READ_BUFSIZE (8)
+	uint32_t read_buf[READ_BUFSIZE];	/**< Buffer for partial
+						 * output sequences (decode)
+						 * (host-endian) */
+	size_t read_len;		/**< Character length of read_buf */
+
+#define WRITE_BUFSIZE (8)
+	uint32_t write_buf[WRITE_BUFSIZE];	/**< Buffer for partial
+						 * output sequences (encode)
+						 * (host-endian) */
+	size_t write_len;		/**< Character length of write_buf */
+
+} charset_utf8_codec;
+
+static bool charset_utf8_codec_handles_charset(const char *charset);
+static parserutils_charset_codec *charset_utf8_codec_create(const char *charset,
+		parserutils_alloc alloc, void *pw);
+static void charset_utf8_codec_destroy (parserutils_charset_codec *codec);
+static parserutils_error charset_utf8_codec_encode(
+		parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error charset_utf8_codec_decode(
+		parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static parserutils_error charset_utf8_codec_reset(
+		parserutils_charset_codec *codec);
+static inline parserutils_error charset_utf8_codec_read_char(
+		charset_utf8_codec *c,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen);
+static inline parserutils_error charset_utf8_codec_output_decoded_char(
+		charset_utf8_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen);
+
+/**
+ * Determine whether this codec handles a specific charset
+ *
+ * \param charset  Charset to test
+ * \return true if handleable, false otherwise
+ */
+bool charset_utf8_codec_handles_charset(const char *charset)
+{
+	return parserutils_charset_mibenum_from_name(charset, 
+				strlen(charset)) ==
+			parserutils_charset_mibenum_from_name("UTF-8", 
+				SLEN("UTF-8"));
+}
+
+/**
+ * Create a utf8 codec
+ *
+ * \param charset  The charset to read from / write to
+ * \param alloc    Memory (de)allocation function
+ * \param pw       Pointer to client-specific private data (may be NULL)
+ * \return Pointer to codec, or NULL on failure
+ */
+parserutils_charset_codec *charset_utf8_codec_create(const char *charset,
+		parserutils_alloc alloc, void *pw)
+{
+	charset_utf8_codec *codec;
+
+	UNUSED(charset);
+
+	codec = alloc(NULL, sizeof(charset_utf8_codec), pw);
+	if (codec == NULL)
+		return NULL;
+
+	codec->inval_buf[0] = '\0';
+	codec->inval_len = 0;
+
+	codec->read_buf[0] = 0;
+	codec->read_len = 0;
+
+	codec->write_buf[0] = 0;
+	codec->write_len = 0;
+
+	/* Finally, populate vtable */
+	codec->base.handler.destroy = charset_utf8_codec_destroy;
+	codec->base.handler.encode = charset_utf8_codec_encode;
+	codec->base.handler.decode = charset_utf8_codec_decode;
+	codec->base.handler.reset = charset_utf8_codec_reset;
+
+	return (parserutils_charset_codec *) codec;
+}
+
+/**
+ * Destroy a utf8 codec
+ *
+ * \param codec  The codec to destroy
+ */
+void charset_utf8_codec_destroy (parserutils_charset_codec *codec)
+{
+	UNUSED(codec);
+}
+
+/**
+ * Encode a chunk of UCS4 data into utf8
+ *
+ * \param codec      The codec to use
+ * \param source     Pointer to pointer to source data
+ * \param sourcelen  Pointer to length (in bytes) of source data
+ * \param dest       Pointer to pointer to output buffer
+ * \param destlen    Pointer to length (in bytes) of output buffer
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read. Any remaining output for the character will be buffered by the
+ * codec for writing on the next call.
+ *
+ * Note that, if failure occurs whilst attempting to write any output
+ * buffered by the last call, then ::source and ::sourcelen will remain
+ * unchanged (as nothing more has been read).
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ */
+parserutils_error charset_utf8_codec_encode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	charset_utf8_codec *c = (charset_utf8_codec *) codec;
+	uint32_t ucs4;
+	uint32_t *towrite;
+	size_t towritelen;
+	parserutils_error error;
+
+	/* Process any outstanding characters from the previous call */
+	if (c->write_len > 0) {
+		uint32_t *pwrite = c->write_buf;
+
+		while (c->write_len > 0) {
+			UTF8_FROM_UCS4(pwrite[0], dest, destlen, error);
+			if (error != PARSERUTILS_OK) {
+				if (error != PARSERUTILS_NOMEM)
+					abort();
+
+				/* Insufficient output buffer space */
+				for (uint32_t len = 0; 
+						len < c->write_len; len++) {
+					c->write_buf[len] = pwrite[len];
+				}
+
+				return PARSERUTILS_NOMEM;
+			}
+
+			pwrite++;
+			c->write_len--;
+		}
+	}
+
+	/* Now process the characters for this call */
+	while (*sourcelen > 0) {
+		ucs4 = ntohl(*((uint32_t *) (void *) *source));
+		towrite = &ucs4;
+		towritelen = 1;
+
+		/* Output current characters */
+		while (towritelen > 0) {
+			UTF8_FROM_UCS4(towrite[0], dest, destlen, error);
+			if (error != PARSERUTILS_OK) {
+				if (error != PARSERUTILS_NOMEM)
+					abort();
+
+				/* Insufficient output space */
+				if (towritelen >= WRITE_BUFSIZE)
+					abort();
+
+				c->write_len = towritelen;
+
+				/* Copy pending chars to save area, for
+				 * processing next call. */
+				for (uint32_t len = 0; len < towritelen; len++)
+					c->write_buf[len] = towrite[len];
+
+				/* Claim character we've just buffered,
+				 * so it's not reprocessed */
+				*source += 4;
+				*sourcelen -= 4;
+
+				return PARSERUTILS_NOMEM;
+			}
+
+			towrite++;
+			towritelen--;
+		}
+
+		*source += 4;
+		*sourcelen -= 4;
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Decode a chunk of utf8 data into UCS4
+ *
+ * \param codec      The codec to use
+ * \param source     Pointer to pointer to source data
+ * \param sourcelen  Pointer to length (in bytes) of source data
+ * \param dest       Pointer to pointer to output buffer
+ * \param destlen    Pointer to length (in bytes) of output buffer
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read, if the result is _OK or _NOMEM. Any remaining output for the
+ * character will be buffered by the codec for writing on the next call.
+ *
+ * In the case of the result being _INVALID, ::source will point _at_ the 
+ * last input character read; nothing will be written or buffered for the 
+ * failed character. It is up to the client to fix the cause of the failure 
+ * and retry the decoding process.
+ *
+ * Note that, if failure occurs whilst attempting to write any output
+ * buffered by the last call, then ::source and ::sourcelen will remain
+ * unchanged (as nothing more has been read).
+ *
+ * If STRICT error handling is configured and an illegal sequence is split
+ * over two calls, then _INVALID will be returned from the second call,
+ * but ::source will point mid-way through the invalid sequence (i.e. it
+ * will be unmodified over the second call). In addition, the internal
+ * incomplete-sequence buffer will be emptied, such that subsequent calls
+ * will progress, rather than re-evaluating the same invalid sequence.
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ *
+ * Call this with a source length of 0 to flush the output buffer.
+ */
+parserutils_error charset_utf8_codec_decode(parserutils_charset_codec *codec,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	charset_utf8_codec *c = (charset_utf8_codec *) codec;
+	parserutils_error error;
+
+	if (c->read_len > 0) {
+		/* Output left over from last decode */
+		uint32_t *pread = c->read_buf;
+
+		while (c->read_len > 0 && *destlen >= c->read_len * 4) {
+			*((uint32_t *) (void *) *dest) = htonl(pread[0]);
+
+			*dest += 4;
+			*destlen -= 4;
+
+			pread++;
+			c->read_len--;
+		}
+
+		if (*destlen < c->read_len * 4) {
+			/* Ran out of output buffer */
+			size_t i;
+
+			/* Shuffle remaining output down */
+			for (i = 0; i < c->read_len; i++)
+				c->read_buf[i] = pread[i];
+
+			return PARSERUTILS_NOMEM;
+		}
+	}
+
+	if (c->inval_len > 0) {
+		/* The last decode ended in an incomplete sequence.
+		 * Fill up inval_buf with data from the start of the
+		 * new chunk and process it. */
+		uint8_t *in = c->inval_buf;
+		size_t ol = c->inval_len;
+		size_t l = min(INVAL_BUFSIZE - ol - 1, *sourcelen);
+		size_t orig_l = l;
+
+		memcpy(c->inval_buf + ol, *source, l);
+
+		l += c->inval_len;
+
+		error = charset_utf8_codec_read_char(c,
+				(const uint8_t **) &in, &l, dest, destlen);
+		if (error != PARSERUTILS_OK && error != PARSERUTILS_NOMEM) {
+			return error;
+		}
+
+		/* And now, fix up source pointers */
+		*source += max((signed) (orig_l - l), 0);
+		*sourcelen -= max((signed) (orig_l - l), 0);
+
+		/* Failed to resolve an incomplete character and
+		 * ran out of buffer space. No recovery strategy
+		 * possible, so explode everywhere. */
+		if ((orig_l + ol) - l == 0)
+			abort();
+
+		/* Report memory exhaustion case from above */
+		if (error != PARSERUTILS_OK)
+			return error;
+	}
+
+	/* Finally, the "normal" case; process all outstanding characters */
+	while (*sourcelen > 0) {
+		error = charset_utf8_codec_read_char(c,
+				source, sourcelen, dest, destlen);
+		if (error != PARSERUTILS_OK) {
+			return error;
+		}
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Clear a utf8 codec's encoding state
+ *
+ * \param codec  The codec to reset
+ * \return PARSERUTILS_OK on success, appropriate error otherwise
+ */
+parserutils_error charset_utf8_codec_reset(parserutils_charset_codec *codec)
+{
+	charset_utf8_codec *c = (charset_utf8_codec *) codec;
+
+	c->inval_buf[0] = '\0';
+	c->inval_len = 0;
+
+	c->read_buf[0] = 0;
+	c->read_len = 0;
+
+	c->write_buf[0] = 0;
+	c->write_len = 0;
+
+	return PARSERUTILS_OK;
+}
+
+
+/**
+ * Read a character from the UTF-8 to UCS4 (big endian)
+ *
+ * \param c          The codec
+ * \param source     Pointer to pointer to source buffer (updated on exit)
+ * \param sourcelen  Pointer to length of source buffer (updated on exit)
+ * \param dest       Pointer to pointer to output buffer (updated on exit)
+ * \param destlen    Pointer to length of output buffer (updated on exit)
+ * \return PARSERUTILS_OK on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ *         PARSERUTILS_INVALID     if a character cannot be represented and the
+ *                            codec's error handling mode is set to STRICT,
+ *
+ * On exit, ::source will point immediately _after_ the last input character
+ * read, if the result is _OK or _NOMEM. Any remaining output for the
+ * character will be buffered by the codec for writing on the next call.
+ *
+ * In the case of the result being _INVALID, ::source will point _at_ the 
+ * last input character read; nothing will be written or buffered for the 
+ * failed character. It is up to the client to fix the cause of the failure 
+ * and retry the decoding process.
+ *
+ * ::sourcelen will be reduced appropriately on exit.
+ *
+ * ::dest will point immediately _after_ the last character written.
+ *
+ * ::destlen will be reduced appropriately on exit.
+ */
+parserutils_error charset_utf8_codec_read_char(charset_utf8_codec *c,
+		const uint8_t **source, size_t *sourcelen,
+		uint8_t **dest, size_t *destlen)
+{
+	uint32_t ucs4;
+	size_t sucs4;
+	parserutils_error error;
+
+	/* Convert a single character */
+	{
+		const uint8_t *src = *source;
+		size_t srclen = *sourcelen;
+		uint32_t *uptr = &ucs4;
+		size_t *usptr = &sucs4;
+		UTF8_TO_UCS4(src, srclen, uptr, usptr, error);
+	}
+	if (error == PARSERUTILS_OK) {
+		/* Read a character */
+		error = charset_utf8_codec_output_decoded_char(c,
+				ucs4, dest, destlen);
+		if (error == PARSERUTILS_OK || error == PARSERUTILS_NOMEM) {
+			/* output succeeded; update source pointers */
+			*source += sucs4;
+			*sourcelen -= sucs4;
+		}
+
+		/* Clear inval buffer */
+		c->inval_buf[0] = '\0';
+		c->inval_len = 0;
+
+		return error;
+	} else if (error == PARSERUTILS_NEEDDATA) {
+		/* Incomplete input sequence */
+		if (*sourcelen > INVAL_BUFSIZE)
+			abort();
+
+		memmove(c->inval_buf, (char *) *source, *sourcelen);
+		c->inval_buf[*sourcelen] = '\0';
+		c->inval_len = *sourcelen;
+
+		*source += *sourcelen;
+		*sourcelen = 0;
+
+		return PARSERUTILS_OK;
+	} else if (error == PARSERUTILS_INVALID) {
+		/* Illegal input sequence */
+		uint32_t nextchar;
+	
+		/* Strict errormode; simply flag invalid character */
+		if (c->base.errormode == 
+				PARSERUTILS_CHARSET_CODEC_ERROR_STRICT) {
+			/* Clear inval buffer */
+			c->inval_buf[0] = '\0';
+			c->inval_len = 0;
+
+			return PARSERUTILS_INVALID;
+		}
+
+		/* Find next valid UTF-8 sequence.
+		 * We're processing client-provided data, so let's
+		 * be paranoid about its validity. */
+		{
+			const uint8_t *src = *source;
+			size_t srclen = *sourcelen;
+			uint32_t off = 0;
+			uint32_t *ncptr = &nextchar;
+
+			UTF8_NEXT_PARANOID(src, srclen, off, ncptr, error);
+		}
+		if (error != PARSERUTILS_OK) {
+			if (error == PARSERUTILS_NEEDDATA) {
+				/* Need more data to be sure */
+				if (*sourcelen > INVAL_BUFSIZE)
+					abort();
+
+				memmove(c->inval_buf, (char *) *source,
+						*sourcelen);
+				c->inval_buf[*sourcelen] = '\0';
+				c->inval_len = *sourcelen;
+
+				*source += *sourcelen;
+				*sourcelen = 0;
+
+				nextchar = 0;
+			} else {
+				return error;
+			}
+		}
+
+		/* Clear inval buffer */
+		c->inval_buf[0] = '\0';
+		c->inval_len = 0;
+
+		/* output U+FFFD and continue processing. */
+		error = charset_utf8_codec_output_decoded_char(c,
+				0xFFFD, dest, destlen);
+		if (error == PARSERUTILS_OK || error == PARSERUTILS_NOMEM) {
+			/* output succeeded; update source pointers */
+			*source += nextchar;
+			*sourcelen -= nextchar;
+		}
+
+		return error;
+	}
+
+	return PARSERUTILS_OK;
+}
+
+/**
+ * Output a UCS4 character
+ *
+ * \param c        Codec to use
+ * \param ucs4     UCS4 character (host endian)
+ * \param dest     Pointer to pointer to output buffer
+ * \param destlen  Pointer to output buffer length
+ * \return PARSERUTILS_OK          on success,
+ *         PARSERUTILS_NOMEM       if output buffer is too small,
+ */
+parserutils_error charset_utf8_codec_output_decoded_char(charset_utf8_codec *c,
+		uint32_t ucs4, uint8_t **dest, size_t *destlen)
+{
+	if (*destlen < 4) {
+		/* Run out of output buffer */
+		c->read_len = 1;
+		c->read_buf[0] = ucs4;
+
+		return PARSERUTILS_NOMEM;
+	}
+
+	*((uint32_t *) (void *) *dest) = htonl(ucs4);
+	*dest += 4;
+	*destlen -= 4;
+
+	return PARSERUTILS_OK;
+}
+
+
+const parserutils_charset_handler charset_utf8_codec_handler = {
+	charset_utf8_codec_handles_charset,
+	charset_utf8_codec_create
+};
+
author	John Mark Bell <jmb@netsurf-browser.org>	2008-05-01 16:34:46 +0000
committer	John Mark Bell <jmb@netsurf-browser.org>	2008-05-01 16:34:46 +0000
commit	2777a04ed2ba4fd36138b991d66a32a283361f7e (patch)
tree	b0c3730533c36ca41402b6d0c5b98413f0a57bee /src/charset/codecs
download	libparserutils-2777a04ed2ba4fd36138b991d66a32a283361f7e.tar.gz libparserutils-2777a04ed2ba4fd36138b991d66a32a283361f7e.tar.bz2