Files
gzz-mirror/Java/media/UTF8Char.java
2026-09-14 20:19:29 -04:00

246 lines
8.2 KiB
Java

/*
UTF8Char.java
*
* You may use and distribute under the terms of either the GNU Lesser
* General Public License, either version 2 of the license or,
* at your choice, any later version. Alternatively, you may use and
* distribute under the terms of the XPL.
*
* See the LICENSE.lgpl and LICENSE.xpl files for the specific terms of
* the licenses.
*
* This software is distributed in the hope that it will be useful,
* but WITHOUT ANY WARRANTY; without even the implied warranty of
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the README
* file for more details.
*
*/
/*
* Written by Antti-Juhani Kaijanaho
*/
package org.gzigzag;
import java.io.*;
public class UTF8Char {
public static final String rcsid = "$Id: UTF8Char.java,v 1.3 2001/03/21 10:01:21 ajk Exp $";
public char c;
public byte[] b;
/* UTF-8 <-> Java char conversion is table-driven. This code is
* adapted from CatDVI. */
private static class ConvOctet {
/** The number of variable bits, 1-8 */
public int numbits;
/** The template for this octet */
public byte octet;
public ConvOctet(int n, byte o) {
numbits = n;
octet = o;
}
public ConvOctet(int n, int o) {
this(n, (byte) o);
}
}
private static class ConvEntry {
/** Maximum UCS-4 value encodable using this entry. */
public char maxval;
/** Info on each octet. */
ConvOctet[] template;
public ConvEntry(char maxval, ConvOctet[] template) {
this.maxval = maxval;
this.template = template;
}
public ConvEntry(int maxval, ConvOctet[] template) {
this((char) maxval, template);
}
public final int len() { return template.length; }
}
private static final ConvEntry[] convtbl =
new ConvEntry[] {
new ConvEntry (0x7f,
new ConvOctet[] {
new ConvOctet(7, 0)
}),
new ConvEntry (0x7ff,
new ConvOctet[] {
new ConvOctet(5, 0xc0),
new ConvOctet(6, 0x80)
}),
new ConvEntry (0xffff,
new ConvOctet[] {
new ConvOctet(4, 0xe0),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80)
}),
new ConvEntry (0x1fffff,
new ConvOctet[] {
new ConvOctet(3, 0xf0),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80)
}),
new ConvEntry (0x3ffffff,
new ConvOctet[] {
new ConvOctet(2, 0xf4),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80)
}),
new ConvEntry (0x7fffffff,
new ConvOctet[] {
new ConvOctet(1, 0xfc),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
new ConvOctet(6, 0x80),
})
};
/** Convert a Java char into an UTF-8 octet stream. */
public static byte[] charToUTF8(char c) {
// Stage 1: Determine which ConvEntry to use. We use the
// maxval attribute for this: the ConvEntry with smallest
// maxval where the char fits is used.
ConvEntry e = null;
for (int i = 0; i < convtbl.length; i++) {
if (c <= convtbl[i].maxval) {
e = convtbl[i];
break;
}
}
if (e == null) throw new ZZError("character has no UTF-8 representation");
// Stage 2: Do the conversion from right to left.
byte[] rv = new byte[e.len()];
for (int i = rv.length - 1; i >= 0; --i) {
/* At each iteration we first apply the initial bit
pattern and then patch the rest with the indicated
number of lower-order bits in code. When we are done,
we shift these out. */
int numbits = e.template[i].numbits;
rv[i] = e.template[i].octet;
rv[i] |= ((1 << numbits) - 1) & c;
c = (char) (c >> numbits);
}
return rv;
}
private static class UTF8ToCharRV {
public char c;
public int len;
public UTF8ToCharRV(char c, int len) {
this.c = c;
this.len = len;
}
}
public static class InvalidChar extends ZZError {
public InvalidChar(String s) { super(s); }
}
/** Determine the ConvEntry to be used. We determine this
using the first octet's template. */
private static ConvEntry UTF8ConvEntry(byte o) {
for (int i = 0; i < convtbl.length; i++) {
byte b = (byte) (o & ~((1 << convtbl[i].template[0].numbits) - 1));
if (convtbl[i].template[0].octet == b) {
return convtbl[i];
}
}
return null;
}
/** Find out the length of an UTF-8 sequence based on the first
* octet. */
public static int UTF8Len(byte o) {
ConvEntry e = UTF8ConvEntry(o);
if (e == null) throw new InvalidChar("this is no UTF-8 character");
return e.len();
}
private static UTF8ToCharRV UTF8ToChar(byte[] array, int start) {
ConvEntry e = UTF8ConvEntry(array[start]);
if (e == null) return null;
// Decode the octet stram.
char rv = 0;
for (int i = 0; i < e.template.length; i++) {
byte mask = (byte) ((1 << e.template[i].numbits) - 1);
byte fixed = (byte) (array[start + i] & ~mask);
byte bits = (byte) (array[start + i] & mask);
if (fixed != e.template[i].octet) throw new ZZError("Invalid UTF-8 character");
rv = (char) ((rv << e.template[i].numbits) | bits);
}
return new UTF8ToCharRV(rv, e.template.length);
}
private static void init(UTF8Char c, UTF8ToCharRV r, byte[] a, int start) {
if (r == null) throw new InvalidChar("Invalid UTF-8 character");
c.c = r.c;
c.b = new byte[r.len];
for (int i = 0; i < c.b.length; i++) {
byte b = a[start + i];
c.b[i] = b;
}
}
private UTF8Char(UTF8ToCharRV r, byte[] a, int start) {
init(this, r, a, start);
}
public UTF8Char tryFromUTF8(byte[] a, int start) {
UTF8ToCharRV r = UTF8ToChar(a, start);
if (r == null) return null;
return new UTF8Char(r, a, start);
}
/** Create an UTF-8 char from the byte stream (bytes after the
* character are ignored). You can determine how many bytes was
* used by invoking UTF8Len on the created object. */
public UTF8Char(byte[] a) { this(a, 0); }
public UTF8Char(byte[] a, int start) {
init(this, UTF8ToChar(a, start), a, start);
}
public UTF8Char(char c) {
this.c = c;
b = charToUTF8(c);
}
public UTF8Char(Writable w, long offs) {
long len = (w.length() - offs < 6) ? w.length() - offs : 6;
byte[] a = w.read(offs, (int)len);
init(this, UTF8ToChar(a, 0), a, 0);
}
static public void main(String argv[]) {
FileWritable fw;
try {
RandomAccessFile raf = new RandomAccessFile("test.utf8", "rw");
fw = new FileWritable(raf);
} catch (IOException e) {
throw new ZZError("" + e);
}
long offs = 0;
while (offs < fw.length()) {
UTF8Char uc = new UTF8Char(fw, offs);
System.out.println("H: " + (long) uc.c);
offs += uc.b.length;
}
}
}