TailoredIndexCodes.java
/*
Copyright (c) 2026 James Ahlborn
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package com.healthmarketscience.jackcess.impl;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStreamReader;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.Comparator;
import java.util.List;
/**
* A collation which orders some characters differently from english, such as
* estonian, where {@code z} sorts between {@code s} and {@code t}. Ms access
* keeps one weight table per family and moves a few characters against it for
* each such collation, so this holds the moves and takes every other
* character from the weight table it tailors.
*
* @author James Ahlborn
*/
public class TailoredIndexCodes extends GeneralLegacyIndexCodes {
/** the weight table this tailors, which supplies every character this
collation leaves where it is */
private final GeneralLegacyIndexCodes _base;
/** the moved characters, sorted, for {@link Arrays#binarySearch} */
private final char[] _chars;
/** the handler of each character in {@link #_chars} */
private final CharHandler[] _handlers;
/** the sequences this weighs as a unit, longest first, or {@code null}
where it weighs none */
private final Contraction[] _contractions;
/** the first character of each sequence, folded and sorted, so a value
which starts no sequence is rejected on one binary search rather than by
walking the sequences */
private final char[] _contractionStarts;
private TailoredIndexCodes(GeneralLegacyIndexCodes base, char[] chars,
CharHandler[] handlers,
Contraction[] contractions,
char[] contractionStarts) {
_base = base;
_chars = chars;
_handlers = handlers;
_contractions = contractions;
_contractionStarts = contractionStarts;
}
@Override
CharHandler getCharHandler(char c)
{
int idx = Arrays.binarySearch(_chars, c);
return ((idx >= 0) ? _handlers[idx] : _base.getCharHandler(c));
}
@Override
Contraction getContraction(String str, int idx)
{
if(_contractions == null) {
return null;
}
if(Arrays.binarySearch(_contractionStarts,
Character.toLowerCase(str.charAt(idx))) < 0) {
return null;
}
// the sequences are longest first, so the first which matches is the
// longest one, which is what ms access takes
for(Contraction contraction : _contractions) {
if(contraction.matches(str, idx)) {
return contraction;
}
}
return null;
}
/**
* Returns the index codes of the collation with the given name, whose moves
* are in the resource file named after it and the family it tailors.
*/
static GeneralLegacyIndexCodes load(String name,
GeneralLegacyIndexCodes base,
String family)
{
String codesFilePath = DatabaseImpl.RESOURCE_PATH +
"index_codes_" + family + "_" + name + ".txt";
List<Character> chars = new ArrayList<>();
List<CharHandler> handlers = new ArrayList<>();
List<String> seqs = new ArrayList<>();
List<CharHandler[]> seqHandlers = new ArrayList<>();
BufferedReader reader = null;
try {
// a sequence can hold a character outside ascii, such as the croatian
// d and z-caron
reader = new BufferedReader(
new InputStreamReader(
DatabaseImpl.getResourceAsStream(codesFilePath), "UTF-8"));
String line = null;
while((line = reader.readLine()) != null) {
if(line.isEmpty() || (line.charAt(0) == '#')) {
continue;
}
// a line is the code point of the character, then the codes for it
// in the format the weight table files use. the file is written in
// code point order, which is the order the search needs. a line
// which starts with '+' holds a sequence of characters rather than a
// code point, and is marked because a sequence such as "aa" is also
// valid hex
int split = line.indexOf(',');
String codes = line.substring(split + 1);
if(line.charAt(0) == '+') {
// the codes of a sequence are those of each unit it writes, since
// hungarian writes a doubled digraph as the digraph twice
String[] units = codes.split("\\|");
CharHandler[] unitHandlers = new CharHandler[units.length];
for(int i = 0; i < units.length; ++i) {
unitHandlers[i] = parseCodes(units[i]);
}
seqs.add(line.substring(1, split));
seqHandlers.add(unitHandlers);
continue;
}
chars.add((char)Integer.parseInt(line.substring(0, split), 16));
handlers.add(parseCodes(codes));
}
} catch(IOException e) {
throw new RuntimeException("failed loading index codes file " +
codesFilePath, e);
} finally {
ByteUtil.closeQuietly(reader);
}
if(chars.isEmpty() && seqs.isEmpty()) {
// the collation moves nothing, so it is the weight table itself
return base;
}
char[] charArr = new char[chars.size()];
for(int i = 0; i < charArr.length; ++i) {
charArr[i] = chars.get(i);
if((i > 0) && (charArr[i] <= charArr[i - 1])) {
throw new IllegalStateException(
"index codes file " + codesFilePath + " is out of order at " +
(int)charArr[i]);
}
}
Contraction[] contractionArr = null;
char[] starts = null;
if(!seqs.isEmpty()) {
List<Contraction> contractions = new ArrayList<>();
for(int i = 0; i < seqs.size(); ++i) {
String seq = seqs.get(i);
contractions.add(new Contraction(seq, seqHandlers.get(i),
(isDoubled(seq, seqs) ? 1 : 0)));
}
// longest first, so a match cannot stop at a shorter sequence
contractions.sort(
Comparator.comparingInt(Contraction::getLength).reversed());
contractionArr = contractions.toArray(new Contraction[0]);
starts = new char[contractionArr.length];
for(int i = 0; i < starts.length; ++i) {
starts[i] = Character.toLowerCase(
contractionArr[i].toString().charAt(0));
}
Arrays.sort(starts);
}
return new TailoredIndexCodes(
base, charArr, handlers.toArray(new CharHandler[0]),
contractionArr, starts);
}
/**
* Whether the sequence is a doubled digraph, such as the hungarian
* {@code ccs}, which is the repeated first letter and then the digraph
* {@code cs}. Ms access reads the case of such a sequence over the digraph
* alone, so {@code cCSa} is weighed as a unit where {@code cHa} is not.
*/
private static boolean isDoubled(String seq, List<String> seqs)
{
if((seq.length() < 3) ||
(Character.toLowerCase(seq.charAt(0)) !=
Character.toLowerCase(seq.charAt(1)))) {
return false;
}
String rest = seq.substring(1);
for(String other : seqs) {
if(other.equalsIgnoreCase(rest)) {
return true;
}
}
return false;
}
}