001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017
018package org.apache.commons.codec.language.bm;
019
020import org.apache.commons.codec.EncoderException;
021import org.apache.commons.codec.StringEncoder;
022
023/**
024 * Encodes strings into their Beider-Morse phonetic encoding.
025 * <p>
026 * Beider-Morse phonetic encodings are optimized for family names. However, they may be useful for a wide range of words.
027 * </p>
028 * <p>
029 * This encoder is intentionally mutable to allow dynamic configuration through bean properties. As such, it is mutable, and may not be thread-safe. If you
030 * require a guaranteed thread-safe encoding then use {@link PhoneticEngine} directly.
031 * </p>
032 * <h2>Encoding overview</h2>
033 * <p>
034 * Beider-Morse phonetic encodings is a multi-step process. Firstly, a table of rules is consulted to guess what language the word comes from. For example, if
035 * it ends in "{@code ault}" then it infers that the word is French. Next, the word is translated into a phonetic representation using a language-specific
036 * phonetics table. Some runs of letters can be pronounced in multiple ways, and a single run of letters may be potentially broken up into phonemes at different
037 * places, so this stage results in a set of possible language-specific phonetic representations. Lastly, this language-specific phonetic representation is
038 * processed by a table of rules that re-writes it phonetically taking into account systematic pronunciation differences between languages, to move it towards a
039 * pan-indo-european phonetic representation. Again, sometimes there are multiple ways this could be done and sometimes things that can be pronounced in several
040 * ways in the source language have only one way to represent them in this average phonetic language, so the result is again a set of phonetic spellings.
041 * </p>
042 * <p>
043 * Some names are treated as having multiple parts. This can be due to two things. Firstly, they may be hyphenated. In this case, each individual hyphenated
044 * word is encoded, and then these are combined end-to-end for the final encoding. Secondly, some names have standard prefixes, for example, "{@code Mac/Mc}" in
045 * Scottish (English) names. As sometimes it is ambiguous whether the prefix is intended or is an accident of the spelling, the word is encoded once with the
046 * prefix and once without it. The resulting encoding contains one and then the other result.
047 * </p>
048 * <h2>Encoding format</h2>
049 * <p>
050 * Individual phonetic spellings of an input word are represented in upper- and lower-case roman characters. Where there are multiple possible phonetic
051 * representations, these are joined with a pipe ({@code |}) character. If multiple hyphenated words where found, or if the word may contain a name prefix, each
052 * encoded word is placed in ellipses and these blocks are then joined with hyphens. For example, "{@code d'ortley}" has a possible prefix. The form without
053 * prefix encodes to "{@code ortlaj|ortlej}", while the form with prefix encodes to " {@code dortlaj|dortlej}". Thus, the full, combined encoding is
054 * "{@code (ortlaj|ortlej)-(dortlaj|dortlej)}".
055 * </p>
056 * <p>
057 * The encoded forms are often quite a bit longer than the input strings. This is because a single input may have many potential phonetic interpretations. For
058 * example, "{@code Renault}" encodes to " {@code rYnDlt|rYnalt|rYnult|rinDlt|rinalt|rinult}". The {@code APPROX} rules will tend to produce larger encodings as
059 * they consider a wider range of possible, approximate phonetic interpretations of the original word. Down-stream applications may wish to further process the
060 * encoding for indexing or lookup purposes, for example, by splitting on pipe ({@code |}) and indexing under each of these alternatives.
061 * </p>
062 * <p>
063 * <strong>Note</strong>: this version of the Beider-Morse encoding is equivalent with v3.4 of the reference implementation.
064 * </p>
065 * <p>
066 * This class is Not ThreadSafe.
067 * </p>
068 *
069 * @see <a href="https://stevemorse.org/phonetics/bmpm.htm">Beider-Morse Phonetic Matching</a>
070 * @see <a href="https://stevemorse.org/phoneticinfo.htm">Reference implementation</a>
071 * @since 1.6
072 */
073public class BeiderMorseEncoder implements StringEncoder {
074
075    /**
076     * Creates a new builder for a Beider-Morse encoder.
077     *
078     * @since 1.23.0
079     */
080    public static final class Builder {
081
082        private PhoneticEngine engine = PhoneticEngine.builder().get();
083
084        private Builder() {
085            // empty.
086        }
087
088        /**
089         * Gets a new Beider-Morse encoder with the current configuration.
090         *
091         * @return a new Beider-Morse encoder with the current configuration.
092         */
093        public BeiderMorseEncoder get() {
094            return new BeiderMorseEncoder(this);
095        }
096
097        /**
098         * Sets the phonetic engine to use.
099         *
100         * @param engine the phonetic engine to use.
101         * @return this builder, for chaining.
102         */
103        public Builder setPhoneticEngine(final PhoneticEngine engine) {
104            this.engine = engine != null ? engine : PhoneticEngine.builder().get();
105            return this;
106        }
107    }
108
109    /**
110     * Creates a new builder for a BeiderMorseEncoder.
111     *
112     * @return a new builder for a BeiderMorseEncoder.
113     * @since 1.23.0
114     */
115    public static Builder builder() {
116        return new Builder();
117    }
118
119    /** A cached object. */
120    private PhoneticEngine engine = PhoneticEngine.builder().get();
121
122    /**
123     * Constructs a new instance.
124     *
125     * @deprecated Use {@link #builder()} to create a new instance.
126     */
127    @Deprecated
128    public BeiderMorseEncoder() {
129        // empty
130    }
131
132    private BeiderMorseEncoder(final Builder builder) {
133        engine = builder.engine;
134    }
135
136    @Override
137    public Object encode(final Object source) throws EncoderException {
138        if (!(source instanceof String)) {
139            throw new EncoderException("BeiderMorseEncoder encode parameter is not of type String");
140        }
141        return encode((String) source);
142    }
143
144    @Override
145    public String encode(final String source) throws EncoderException {
146        return source != null ? engine.encode(source) : null;
147    }
148
149    /**
150     * Gets the name type currently in operation.
151     *
152     * @return The NameType currently being used.
153     */
154    public NameType getNameType() {
155        return engine.getNameType();
156    }
157
158    /**
159     * Gets the rule type currently in operation.
160     *
161     * @return The RuleType currently being used.
162     */
163    public RuleType getRuleType() {
164        return engine.getRuleType();
165    }
166
167    /**
168     * Tests whether multiple phonetic encodings are concatenated or just the first one is kept.
169     *
170     * @return true if multiple encodings are concatenated, false if just the first one is returned.
171     */
172    public boolean isConcat() {
173        return engine.isConcat();
174    }
175
176    /**
177     * Sets how multiple possible phonetic encodings are combined.
178     *
179     * @param concat true if multiple encodings are to be combined with a '|', false if just the first one is to be considered.
180     */
181    public void setConcat(final boolean concat) {
182        engine = PhoneticEngine.builder().setAll(engine).setConcat(concat).get();
183    }
184
185    /**
186     * Sets the number of maximum of phonemes that shall be considered by the engine.
187     * <p>
188     * A value less than 0 will reset the maximum number of phonemes to the default {@value PhoneticEngine.Builder#MAX_PHONEMES}.
189     * </p>
190     *
191     * @param maxPhonemes the maximum number of phonemes returned by the engine.
192     * @since 1.7
193     */
194    public void setMaxPhonemes(final int maxPhonemes) {
195        engine = PhoneticEngine.builder().setAll(engine).setMaxPhonemes(maxPhonemes).get();
196    }
197
198    /**
199     * Sets the type of name. Use {@link NameType#GENERIC} unless you specifically want phonetic encodings optimized for Ashkenazi or Sephardic Jewish family
200     * names.
201     * <p>
202     * A null value will reset the name type to the default of {@link NameType#GENERIC}.
203     * </p>
204     *
205     * @param nameType the NameType in use.
206     */
207    public void setNameType(final NameType nameType) {
208        engine = PhoneticEngine.builder().setAll(engine).setNameType(nameType).get();
209    }
210
211    /**
212     * Sets the rule type to apply. This will widen or narrow the range of phonetic encodings considered.
213     * <p>
214     * A null value will reset the rule type to the default of {@link RuleType#APPROX}.
215     * </p>
216     *
217     * @param ruleType {@link RuleType#APPROX} or {@link RuleType#EXACT} for approximate or exact phonetic matches.
218     */
219    public void setRuleType(final RuleType ruleType) {
220        engine = PhoneticEngine.builder().setAll(engine).setRuleType(ruleType).get();
221    }
222}