@artur_mkrtchyannnn/metaphone
Version:
Metaphone implementation
387 lines (343 loc) • 7.75 kB
JavaScript
var sh = 'X'
var th = '0'
/**
* Get the phonetics according to the original Metaphone algorithm from a value
*
* @param {string} value
* @returns {string}
*/
function metaphone(value) {
var phonized = ''
var index = 0
/** @type {number} */
var skip
/** @type {function(): string} */
var next
/** @type {function(): string} */
var current
/** @type {function(): string} */
var previous
/**
* Add `characters` to `phonized`
*
* @param {string} characters
*/
function phonize(characters) {
phonized += characters
}
/**
* Get the character offset by `offset` from the current character
*
* @param {number} offset
*/
function at(offset) {
return value.charAt(index + offset).toUpperCase()
}
/**
* Create an `at` function with a bound `offset`
*
* @param {number} offset
*/
function atFactory(offset) {
return function () {
return at(offset)
}
}
value = String(value || '')
if (!value) {
return ''
}
next = atFactory(1)
current = atFactory(0)
previous = atFactory(-1)
// Find our first letter
while (!alpha(current())) {
if (!current()) {
return ''
}
index++
}
switch (current()) {
case 'A':
// AE becomes E
if (next() === 'E') {
phonize('E')
index += 2
} else {
// Remember, preserve vowels at the beginning
phonize('A')
index++
}
break
// [GKP]N becomes N
case 'G':
case 'K':
case 'P':
if (next() === 'N') {
phonize('N')
index += 2
}
break
// WH becomes H, WR becomes R, W if followed by a vowel
case 'W':
if (next() === 'R') {
phonize(next())
index += 2
} else if (next() === 'H') {
phonize(current())
index += 2
} else if (vowel(next())) {
phonize('W')
index += 2
}
// Ignore
break
// X becomes S
case 'X':
phonize('S')
index++
break
// Vowels are kept (we did A already)
case 'E':
case 'I':
case 'O':
case 'U':
phonize(current())
index++
break
default:
// Ignore
break
}
// On to the metaphoning
while (current()) {
// How many letters to skip because an eariler encoding handled multiple
// letters
skip = 1
// Ignore non-alphas
if (!alpha(current()) || (current() === previous() && current() !== 'C')) {
index += skip
continue
}
// eslint-disable-next-line default-case
switch (current()) {
// B -> B unless in MB
case 'B':
if (previous() !== 'M') {
phonize('B')
}
break
// 'sh' if -CIA- or -CH, but not SCH, except SCHW (SCHW is handled in S)
// S if -CI-, -CE- or -CY- dropped if -SCI-, SCE-, -SCY- (handed in S)
// else K
case 'C':
if (soft(next())) {
// C[IEY]
if (next() === 'I' && at(2) === 'A') {
// CIA
phonize(sh)
} else if (previous() !== 'S') {
phonize('S')
}
} else if (next() === 'H') {
phonize(sh)
skip++
} else {
// C
phonize('K')
}
break
// J if in -DGE-, -DGI- or -DGY-, else T
case 'D':
if (next() === 'G' && soft(at(2))) {
phonize('J')
skip++
} else {
phonize('T')
}
break
// F if in -GH and not B--GH, D--GH, -H--GH, -H---GH
// else dropped if -GNED, -GN,
// else dropped if -DGE-, -DGI- or -DGY- (handled in D)
// else J if in -GE-, -GI, -GY and not GG
// else K
case 'G':
if (next() === 'H') {
if (!(noGhToF(at(-3)) || at(-4) === 'H')) {
phonize('F')
skip++
}
} else if (next() === 'N') {
if (!(!alpha(at(2)) || (at(2) === 'E' && at(3) === 'D'))) {
phonize('K')
}
} else if (soft(next()) && previous() !== 'G') {
phonize('J')
} else {
phonize('K')
}
break
// H if before a vowel and not after C,G,P,S,T
case 'H':
if (vowel(next()) && !dipthongH(previous())) {
phonize('H')
}
break
// Dropped if after C, else K
case 'K':
if (previous() !== 'C') {
phonize('K')
}
break
// F if before H, else P
case 'P':
if (next() === 'H') {
phonize('F')
} else {
phonize('P')
}
break
// K
case 'Q':
phonize('K')
break
// 'sh' in -SH-, -SIO- or -SIA- or -SCHW-, else S
case 'S':
if (next() === 'I' && (at(2) === 'O' || at(2) === 'A')) {
phonize(sh)
} else if (next() === 'H') {
phonize(sh)
skip++
} else {
phonize('S')
}
break
// 'sh' in -TIA- or -TIO-, else 'th' before H, else T
case 'T':
if (next() === 'I' && (at(2) === 'O' || at(2) === 'A')) {
phonize(sh)
} else if (next() === 'H') {
phonize(th)
skip++
} else if (!(next() === 'C' && at(2) === 'H')) {
phonize('T')
}
break
// F
case 'V':
phonize('F')
break
case 'W':
if (vowel(next())) {
phonize('W')
}
break
// KS
case 'X':
phonize('KS')
break
// Y if followed by a vowel
case 'Y':
if (vowel(next())) {
phonize('Y')
}
break
// S
case 'Z':
phonize('S')
break
// No transformation
case 'F':
case 'J':
case 'L':
case 'M':
case 'N':
case 'R':
phonize(current())
break
}
index += skip
}
return phonized
}
/**
* Check whether `character` would make `'GH'` an `'F'`
*
* @param {string} character
* @returns {boolean}
*/
function noGhToF(character) {
character = char(character)
return character === 'B' || character === 'D' || character === 'H'
}
/**
* Check whether `character` would make a `'C'` or `'G'` soft
*
* @param {string} character
* @returns {boolean}
*/
function soft(character) {
character = char(character)
return character === 'E' || character === 'I' || character === 'Y'
}
/**
* Check whether `character` is a vowel
*
* @param {string} character
* @returns {boolean}
*/
function vowel(character) {
character = char(character)
return (
character === 'A' ||
character === 'E' ||
character === 'I' ||
character === 'O' ||
character === 'U'
)
}
/**
* Check whether `character` forms a dipthong when preceding H
*
* @param {string} character
* @returns {boolean}
*/
function dipthongH(character) {
character = char(character)
return (
character === 'C' ||
character === 'G' ||
character === 'P' ||
character === 'S' ||
character === 'T'
)
}
/**
* Check whether `character` is in the alphabet
*
* @param {string} character
* @returns {boolean}
*/
function alpha(character) {
var code = charCode(character)
return code >= 65 && code <= 90
}
/**
* Get the upper-case character code of the first character in `character`
*
* @param {string} character
* @returns {number}
*/
function charCode(character) {
return char(character).charCodeAt(0)
}
/**
* Turn `character` into a single, upper-case character
*
* @param {string} character
* @returns {string}
*/
function char(character) {
return String(character).charAt(0).toUpperCase()
}
module.exports = {metaphone}