Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions lucene/CHANGES.txt
Original file line number Diff line number Diff line change
Expand Up @@ -123,6 +123,9 @@ API Changes

New Features
---------------------
* GITHUB#16722: Support user dictionary in ThaiTokenizer and ThaiTokenizerFactory.
(Kamthorn Krairaksa)

* GITHUB#16717: Add ThaiCharFilter, ThaiCharFilterFactory, and ThaiNormalizationFilter for
pre-tokenization character normalization and token-level normalization in ThaiAnalyzer.
(Kamthorn Krairaksa)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -18,14 +18,19 @@

import java.text.BreakIterator;
import java.util.Locale;
import org.apache.lucene.analysis.CharArraySet;
import org.apache.lucene.analysis.tokenattributes.CharTermAttribute;
import org.apache.lucene.analysis.tokenattributes.OffsetAttribute;
import org.apache.lucene.analysis.util.CharArrayIterator;
import org.apache.lucene.analysis.util.SegmentingTokenizerBase;
import org.apache.lucene.util.AttributeFactory;

/**
* Tokenizer that use {@link BreakIterator} to tokenize Thai text.
* Tokenizer that uses {@link BreakIterator} to tokenize Thai text.
*
* <p>Supports an optional user dictionary ({@link CharArraySet}) for custom or domain-specific
* words. When a user dictionary word matches, it takes precedence over default segmentation
* boundaries.
*
* <p>WARNING: this tokenizer may not be supported by all JREs. It is known to work with Sun/Oracle
* and Harmony JREs. If your application needs to be fully portable, consider using ICUTokenizer
Expand Down Expand Up @@ -53,6 +58,10 @@ public class ThaiTokenizer extends SegmentingTokenizerBase {
private final BreakIterator wordBreaker;
private final CharArrayIterator wrapper = CharArrayIterator.newWordInstance();

private final CharArraySet userDictionary;
private final int minDictWordLen;
private final int maxDictWordLen;

int sentenceStart;
int sentenceEnd;

Expand All @@ -61,17 +70,51 @@ public class ThaiTokenizer extends SegmentingTokenizerBase {

/** Creates a new ThaiTokenizer */
public ThaiTokenizer() {
this(DEFAULT_TOKEN_ATTRIBUTE_FACTORY);
this(DEFAULT_TOKEN_ATTRIBUTE_FACTORY, null);
}

/**
* Creates a new ThaiTokenizer with a user dictionary.
*
* @param userDictionary custom dictionary of words, or null for default dictionary only
*/
public ThaiTokenizer(CharArraySet userDictionary) {
this(DEFAULT_TOKEN_ATTRIBUTE_FACTORY, userDictionary);
}

/** Creates a new ThaiTokenizer, supplying the AttributeFactory */
public ThaiTokenizer(AttributeFactory factory) {
this(factory, null);
}

/**
* Creates a new ThaiTokenizer, supplying the AttributeFactory and user dictionary.
*
* @param factory AttributeFactory to use
* @param userDictionary custom dictionary of words, or null for default dictionary only
*/
public ThaiTokenizer(AttributeFactory factory, CharArraySet userDictionary) {
super(factory, (BreakIterator) sentenceProto.clone());
if (!DBBI_AVAILABLE) {
throw new UnsupportedOperationException(
"This JRE does not have support for Thai segmentation");
}
wordBreaker = (BreakIterator) proto.clone();
this.userDictionary = userDictionary;
if (userDictionary != null && !userDictionary.isEmpty()) {
int minLen = Integer.MAX_VALUE;
int maxLen = 0;
for (Object obj : userDictionary) {
int length = ((char[]) obj).length;
minLen = Math.min(minLen, length);
maxLen = Math.max(maxLen, length);
}
this.minDictWordLen = minLen;
this.maxDictWordLen = maxLen;
} else {
this.minDictWordLen = 0;
this.maxDictWordLen = 0;
}
}

@Override
Expand Down Expand Up @@ -102,10 +145,39 @@ protected boolean incrementWord() {
return false; // BreakIterator exhausted
}

boolean reanchor = false;
if (userDictionary != null && !userDictionary.isEmpty()) {
int remaining = sentenceEnd - (sentenceStart + start);
int maxLen = Math.min(maxDictWordLen, remaining);
for (int l = maxLen; l >= minDictWordLen; l--) {
if (userDictionary.contains(buffer, sentenceStart + start, l)) {
int customEnd = start + l;
if (customEnd != end) {
end = customEnd;
if (end >= sentenceEnd - sentenceStart) {
wordBreaker.last();
} else if (wordBreaker.isBoundary(end)) {
wordBreaker.following(end - 1);
} else {
reanchor = true;
}
}
break;
}
}
}

clearAttributes();
termAtt.copyBuffer(buffer, sentenceStart + start, end - start);
offsetAtt.setOffset(
correctOffset(offset + sentenceStart + start), correctOffset(offset + sentenceStart + end));

if (reanchor) {
sentenceStart += end;
wrapper.setText(buffer, sentenceStart, sentenceEnd - sentenceStart);
wordBreaker.setText(wrapper);
}

return true;
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -16,32 +16,40 @@
*/
package org.apache.lucene.analysis.th;

import java.io.IOException;
import java.util.Map;
import org.apache.lucene.analysis.CharArraySet;
import org.apache.lucene.analysis.Tokenizer;
import org.apache.lucene.analysis.TokenizerFactory;
import org.apache.lucene.util.AttributeFactory;
import org.apache.lucene.util.ResourceLoader;
import org.apache.lucene.util.ResourceLoaderAware;

/**
* Factory for {@link ThaiTokenizer}.
*
* <pre><code class="language-xml">
* &lt;fieldType name="text_thai" class="solr.TextField" positionIncrementGap="100"&gt;
* &lt;analyzer&gt;
* &lt;tokenizer class="solr.ThaiTokenizerFactory"/&gt;
* &lt;tokenizer class="solr.ThaiTokenizerFactory" dictionary="custom_words.txt"/&gt;
* &lt;/analyzer&gt;
* &lt;/fieldType&gt;</code></pre>
*
* @since 4.10.0
* @lucene.spi {@value #NAME}
*/
public class ThaiTokenizerFactory extends TokenizerFactory {
public class ThaiTokenizerFactory extends TokenizerFactory implements ResourceLoaderAware {

/** SPI name */
public static final String NAME = "thai";

private final String dictFile;
private CharArraySet dictionary;

/** Creates a new ThaiTokenizerFactory */
public ThaiTokenizerFactory(Map<String, String> args) {
super(args);
dictFile = get(args, "dictionary");
if (!args.isEmpty()) {
throw new IllegalArgumentException("Unknown parameters: " + args);
}
Expand All @@ -52,8 +60,15 @@ public ThaiTokenizerFactory() {
throw defaultCtorException();
}

@Override
public void inform(ResourceLoader loader) throws IOException {
if (dictFile != null) {
dictionary = getWordSet(loader, dictFile, false);
}
}

@Override
public Tokenizer create(AttributeFactory factory) {
return new ThaiTokenizer(factory);
return new ThaiTokenizer(factory, dictionary);
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed with
* this work for additional information regarding copyright ownership.
* The ASF licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package org.apache.lucene.analysis.th;

import java.io.IOException;
import java.io.StringReader;
import java.util.List;
import org.apache.lucene.analysis.CharArraySet;
import org.apache.lucene.analysis.Tokenizer;
import org.apache.lucene.tests.analysis.BaseTokenStreamTestCase;

/** Test case for {@link ThaiTokenizer}. */
public class TestThaiTokenizer extends BaseTokenStreamTestCase {

@Override
public void setUp() throws Exception {
super.setUp();
assumeTrue(
"JRE does not support Thai dictionary-based BreakIterator", ThaiTokenizer.DBBI_AVAILABLE);
}

public void testDefaultSegmentation() throws IOException {
Tokenizer tokenizer = new ThaiTokenizer();
tokenizer.setReader(new StringReader("ภาษาไทย"));
assertTokenStreamContents(tokenizer, new String[] {"ภาษา", "ไทย"});
}

public void testUserDictionary() throws IOException {
CharArraySet userDict = new CharArraySet(List.of("พารากอน", "คนขับรถ"), false);
Tokenizer tokenizer = new ThaiTokenizer(userDict);
tokenizer.setReader(new StringReader("ไปพารากอนกัน"));
assertTokenStreamContents(tokenizer, new String[] {"ไป", "พารากอน", "กัน"});
}

public void testMultipleUserDictionaryTermsAndOffsets() throws IOException {
CharArraySet userDict = new CharArraySet(List.of("พารากอน", "คนขับรถ"), false);
Tokenizer tokenizer = new ThaiTokenizer(userDict);
tokenizer.setReader(new StringReader("เขาเป็นคนขับรถไปพารากอน"));
assertTokenStreamContents(
tokenizer,
new String[] {"เขา", "เป็น", "คนขับรถ", "ไป", "พารากอน"},
new int[] {0, 3, 7, 14, 16},
new int[] {3, 7, 14, 16, 23});
}

public void testEmptyUserDictionary() throws IOException {
CharArraySet userDict = new CharArraySet(0, false);
Tokenizer tokenizer = new ThaiTokenizer(userDict);
tokenizer.setReader(new StringReader("ภาษาไทย"));
assertTokenStreamContents(tokenizer, new String[] {"ภาษา", "ไทย"});
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,16 @@ public void testWordBreak() throws Exception {
tokenizer, new String[] {"การ", "ที่", "ได้", "ต้อง", "แสดง", "ว่า", "งาน", "ดี"});
}

public void testCustomDictionary() throws Exception {
assumeTrue(
"JRE does not support Thai dictionary-based BreakIterator", ThaiTokenizer.DBBI_AVAILABLE);
Tokenizer tokenizer =
tokenizerFactory("Thai", "dictionary", "customThaiDictionary.txt")
.create(newAttributeFactory());
tokenizer.setReader(new StringReader("ไปพารากอน"));
assertTokenStreamContents(tokenizer, new String[] {"ไป", "พารากอน"});
}

/** Test that bogus arguments result in exception */
public void testBogusArguments() throws Exception {
assumeTrue(
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
# Custom Thai Dictionary
พารากอน
คนขับรถ
Loading