CharsetConverter.java

/*
** Module   : CharsetConverter.java
** Abstract : optimized character set conversion routines
**
** Copyright (c) 2008-2017, Golden Code Development Corporation.
**
** -#- -I- --Date-- -T- --JPRM-- ----------------Description-----------------
** 001 GES 20080730 NEW   @39234 Created initial version which provides
**                               optimized character set conversion routines.
*/ 
/*
** This program is free software: you can redistribute it and/or modify
** it under the terms of the GNU Affero General Public License as
** published by the Free Software Foundation, either version 3 of the
** License, or (at your option) any later version.
**
** This program is distributed in the hope that it will be useful,
** but WITHOUT ANY WARRANTY; without even the implied warranty of
** MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
** GNU Affero General Public License for more details.
**
** You may find a copy of the GNU Affero GPL version 3 at the following
** location: https://www.gnu.org/licenses/agpl-3.0.en.html
** 
** Additional terms under GNU Affero GPL version 3 section 7:
** 
**   Under Section 7 of the GNU Affero GPL version 3, the following additional
**   terms apply to the works covered under the License.  These additional terms
**   are non-permissive additional terms allowed under Section 7 of the GNU
**   Affero GPL version 3 and may not be removed by you.
** 
**   0. Attribution Requirement.
** 
**     You must preserve all legal notices or author attributions in the covered
**     work or Appropriate Legal Notices displayed by works containing the covered
**     work.  You may not remove from the covered work any author or developer
**     credit already included within the covered work.
** 
**   1. No License To Use Trademarks.
** 
**     This license does not grant any license or rights to use the trademarks
**     Golden Code, FWD, any Golden Code or FWD logo, or any other trademarks
**     of Golden Code Development Corporation. You are not authorized to use the
**     name Golden Code, FWD, or the names of any author or contributor, for
**     publicity purposes without written authorization.
** 
**   2. No Misrepresentation of Affiliation.
** 
**     You may not represent yourself as Golden Code Development Corporation or FWD.
** 
**     You may not represent yourself for publicity purposes as associated with
**     Golden Code Development Corporation, FWD, or any author or contributor to
**     the covered work, without written authorization.
** 
**   3. No Misrepresentation of Source or Origin.
** 
**     You may not represent the covered work as solely your work.  All modified
**     versions of the covered work must be marked in a reasonable way to make it
**     clear that the modified work is not originating from Golden Code Development
**     Corporation or FWD.  All modified versions must contain the notices of
**     attribution required in this license.
*/

package com.goldencode.p2j.util;

import java.io.*;
import java.util.*;
import java.nio.charset.*;

/**
 * Provides optimized character set conversion routines. Currently this is
 * only useful for single byte character sets.
 * <p>
 * This class trades use of the heap for speed by pre-calculating the
 * conversions and creating conversion "tables". The resulting conversion
 * operation is a simple array de-reference.
 *
 * @author    GES
 */
public class CharsetConverter
{
   /** All possible byte values. */
   private static byte[] allBytes = new byte[256];
   
   /** Simple byte to character conversion mapping. */
   private char[] byteToChar = new char[256];
   
   /** Simple character to byte conversion mapping. */
   private byte[] charToByte = null;
   
   /** Smallest character value of the character to byte mapping range. */
   private int offset = 0;
   
   static
   {
      for (int i = 0; i < 256; i++)
      {
         allBytes[i] = (byte) i;
      }
   }
   
   /**
    * Create an instance that converts between Unicode and the given
    * character set.
    *
    * @param    cset
    *           The character set.
    */
   public CharsetConverter(String cset)
   {
      try
      {
         // prepare our byte to character conversion mapping
         String txt = (cset == null) ? new String(allBytes)
                                     : new String(allBytes, cset);
         
         int min = 64 * 1024;
         int max = 0;
         int len = 0;
                                     
         for (int i = 0; i < 256; i++)
         {
            char ch = txt.charAt(i);
            
            min = Math.min(min, ch);
            max = Math.max(max, ch);
            
            byteToChar[i] = ch;
         }
         
         offset = min;
         len    = max - min + 1;
         charToByte = new byte[len];
         
         Arrays.fill(charToByte, (byte) -1);
         
         for (int i = 0; i < 256; i++)
         {
            char ch = txt.charAt(i);
            
            charToByte[ch - offset] = (byte) i;
         }
      }
      
      catch (UnsupportedEncodingException uee)
      {
         // we can't map this, ensure that an error will be returned
         Arrays.fill(byteToChar, (char) -1);
      }
   }
   
   /**
    * Convert the given Unicode character into the corresponding number in
    * the represented character set.
    *
    * @param    ch
    *           The Unicode character to convert.
    *
    * @return   The number represented or -1 if there is any error.
    */
   public int toInt(char ch)
   {
      return charToByte[ch - offset] & 0x000000FF;
   }
   
   /**
    * Convert the given Unicode character into the corresponding number in
    * the represented character set.
    *
    * @param    ch
    *           The Unicode character to convert.
    *
    * @return   The number represented or -1 if there is any error.
    */
   public byte toByte(char ch)
   {
      return charToByte[ch - offset];
   }
   
   /**
    * Convert the given number into the corresponding Unicode character in
    * the represented character set.
    *
    * @param    i
    *           The number to convert.
    *
    * @return   The Unicode character representing the given number or -1
    *           if there is any error.
    */
   public char toChar(int i)
   {
      return byteToChar[i];
   }
   
   /**
    * Convert the given byte into the corresponding Unicode character in
    * the represented character set.
    *
    * @param    b
    *           The byte to convert.
    *
    * @return   The Unicode character representing the given byte or -1
    *           if there is any error.
    */
   public char toChar(byte b)
   {
      return byteToChar[b];
   }
}