246 lines
8.2 KiB
Java
246 lines
8.2 KiB
Java
/*
|
|
UTF8Char.java
|
|
*
|
|
* You may use and distribute under the terms of either the GNU Lesser
|
|
* General Public License, either version 2 of the license or,
|
|
* at your choice, any later version. Alternatively, you may use and
|
|
* distribute under the terms of the XPL.
|
|
*
|
|
* See the LICENSE.lgpl and LICENSE.xpl files for the specific terms of
|
|
* the licenses.
|
|
*
|
|
* This software is distributed in the hope that it will be useful,
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the README
|
|
* file for more details.
|
|
*
|
|
*/
|
|
/*
|
|
* Written by Antti-Juhani Kaijanaho
|
|
*/
|
|
package org.gzigzag;
|
|
|
|
import java.io.*;
|
|
|
|
public class UTF8Char {
|
|
public static final String rcsid = "$Id: UTF8Char.java,v 1.3 2001/03/21 10:01:21 ajk Exp $";
|
|
|
|
public char c;
|
|
public byte[] b;
|
|
|
|
/* UTF-8 <-> Java char conversion is table-driven. This code is
|
|
* adapted from CatDVI. */
|
|
|
|
private static class ConvOctet {
|
|
/** The number of variable bits, 1-8 */
|
|
public int numbits;
|
|
/** The template for this octet */
|
|
public byte octet;
|
|
|
|
public ConvOctet(int n, byte o) {
|
|
numbits = n;
|
|
octet = o;
|
|
}
|
|
|
|
public ConvOctet(int n, int o) {
|
|
this(n, (byte) o);
|
|
}
|
|
}
|
|
|
|
private static class ConvEntry {
|
|
/** Maximum UCS-4 value encodable using this entry. */
|
|
public char maxval;
|
|
/** Info on each octet. */
|
|
ConvOctet[] template;
|
|
|
|
public ConvEntry(char maxval, ConvOctet[] template) {
|
|
this.maxval = maxval;
|
|
this.template = template;
|
|
}
|
|
|
|
public ConvEntry(int maxval, ConvOctet[] template) {
|
|
this((char) maxval, template);
|
|
}
|
|
|
|
public final int len() { return template.length; }
|
|
}
|
|
|
|
private static final ConvEntry[] convtbl =
|
|
new ConvEntry[] {
|
|
new ConvEntry (0x7f,
|
|
new ConvOctet[] {
|
|
new ConvOctet(7, 0)
|
|
}),
|
|
new ConvEntry (0x7ff,
|
|
new ConvOctet[] {
|
|
new ConvOctet(5, 0xc0),
|
|
new ConvOctet(6, 0x80)
|
|
}),
|
|
new ConvEntry (0xffff,
|
|
new ConvOctet[] {
|
|
new ConvOctet(4, 0xe0),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80)
|
|
}),
|
|
new ConvEntry (0x1fffff,
|
|
new ConvOctet[] {
|
|
new ConvOctet(3, 0xf0),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80)
|
|
}),
|
|
new ConvEntry (0x3ffffff,
|
|
new ConvOctet[] {
|
|
new ConvOctet(2, 0xf4),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80)
|
|
}),
|
|
new ConvEntry (0x7fffffff,
|
|
new ConvOctet[] {
|
|
new ConvOctet(1, 0xfc),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
new ConvOctet(6, 0x80),
|
|
})
|
|
};
|
|
|
|
/** Convert a Java char into an UTF-8 octet stream. */
|
|
public static byte[] charToUTF8(char c) {
|
|
// Stage 1: Determine which ConvEntry to use. We use the
|
|
// maxval attribute for this: the ConvEntry with smallest
|
|
// maxval where the char fits is used.
|
|
ConvEntry e = null;
|
|
for (int i = 0; i < convtbl.length; i++) {
|
|
if (c <= convtbl[i].maxval) {
|
|
e = convtbl[i];
|
|
break;
|
|
}
|
|
}
|
|
if (e == null) throw new ZZError("character has no UTF-8 representation");
|
|
|
|
// Stage 2: Do the conversion from right to left.
|
|
byte[] rv = new byte[e.len()];
|
|
for (int i = rv.length - 1; i >= 0; --i) {
|
|
/* At each iteration we first apply the initial bit
|
|
pattern and then patch the rest with the indicated
|
|
number of lower-order bits in code. When we are done,
|
|
we shift these out. */
|
|
int numbits = e.template[i].numbits;
|
|
rv[i] = e.template[i].octet;
|
|
rv[i] |= ((1 << numbits) - 1) & c;
|
|
c = (char) (c >> numbits);
|
|
}
|
|
return rv;
|
|
}
|
|
|
|
private static class UTF8ToCharRV {
|
|
public char c;
|
|
public int len;
|
|
public UTF8ToCharRV(char c, int len) {
|
|
this.c = c;
|
|
this.len = len;
|
|
}
|
|
}
|
|
|
|
public static class InvalidChar extends ZZError {
|
|
public InvalidChar(String s) { super(s); }
|
|
}
|
|
|
|
/** Determine the ConvEntry to be used. We determine this
|
|
using the first octet's template. */
|
|
private static ConvEntry UTF8ConvEntry(byte o) {
|
|
for (int i = 0; i < convtbl.length; i++) {
|
|
byte b = (byte) (o & ~((1 << convtbl[i].template[0].numbits) - 1));
|
|
if (convtbl[i].template[0].octet == b) {
|
|
return convtbl[i];
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
/** Find out the length of an UTF-8 sequence based on the first
|
|
* octet. */
|
|
public static int UTF8Len(byte o) {
|
|
ConvEntry e = UTF8ConvEntry(o);
|
|
if (e == null) throw new InvalidChar("this is no UTF-8 character");
|
|
return e.len();
|
|
}
|
|
|
|
private static UTF8ToCharRV UTF8ToChar(byte[] array, int start) {
|
|
ConvEntry e = UTF8ConvEntry(array[start]);
|
|
if (e == null) return null;
|
|
|
|
// Decode the octet stram.
|
|
char rv = 0;
|
|
for (int i = 0; i < e.template.length; i++) {
|
|
byte mask = (byte) ((1 << e.template[i].numbits) - 1);
|
|
byte fixed = (byte) (array[start + i] & ~mask);
|
|
byte bits = (byte) (array[start + i] & mask);
|
|
if (fixed != e.template[i].octet) throw new ZZError("Invalid UTF-8 character");
|
|
rv = (char) ((rv << e.template[i].numbits) | bits);
|
|
}
|
|
return new UTF8ToCharRV(rv, e.template.length);
|
|
}
|
|
|
|
private static void init(UTF8Char c, UTF8ToCharRV r, byte[] a, int start) {
|
|
if (r == null) throw new InvalidChar("Invalid UTF-8 character");
|
|
c.c = r.c;
|
|
c.b = new byte[r.len];
|
|
for (int i = 0; i < c.b.length; i++) {
|
|
byte b = a[start + i];
|
|
c.b[i] = b;
|
|
}
|
|
}
|
|
|
|
private UTF8Char(UTF8ToCharRV r, byte[] a, int start) {
|
|
init(this, r, a, start);
|
|
}
|
|
|
|
public UTF8Char tryFromUTF8(byte[] a, int start) {
|
|
UTF8ToCharRV r = UTF8ToChar(a, start);
|
|
if (r == null) return null;
|
|
return new UTF8Char(r, a, start);
|
|
}
|
|
|
|
/** Create an UTF-8 char from the byte stream (bytes after the
|
|
* character are ignored). You can determine how many bytes was
|
|
* used by invoking UTF8Len on the created object. */
|
|
public UTF8Char(byte[] a) { this(a, 0); }
|
|
public UTF8Char(byte[] a, int start) {
|
|
init(this, UTF8ToChar(a, start), a, start);
|
|
}
|
|
|
|
public UTF8Char(char c) {
|
|
this.c = c;
|
|
b = charToUTF8(c);
|
|
}
|
|
|
|
public UTF8Char(Writable w, long offs) {
|
|
long len = (w.length() - offs < 6) ? w.length() - offs : 6;
|
|
byte[] a = w.read(offs, (int)len);
|
|
init(this, UTF8ToChar(a, 0), a, 0);
|
|
}
|
|
|
|
static public void main(String argv[]) {
|
|
FileWritable fw;
|
|
try {
|
|
RandomAccessFile raf = new RandomAccessFile("test.utf8", "rw");
|
|
fw = new FileWritable(raf);
|
|
} catch (IOException e) {
|
|
throw new ZZError("" + e);
|
|
}
|
|
|
|
long offs = 0;
|
|
while (offs < fw.length()) {
|
|
UTF8Char uc = new UTF8Char(fw, offs);
|
|
System.out.println("H: " + (long) uc.c);
|
|
offs += uc.b.length;
|
|
}
|
|
|
|
}
|
|
}
|
|
|