@predictive-text-studio/lexical-model-compiler
Version:
Keyman Developer lexical model compiler
294 lines • 10.7 kB
JavaScript
;
Object.defineProperty(exports, "__esModule", { value: true });
const parse_wordlist_1 = require("../parse-wordlist");
// Supports LF or CRLF line terminators.
exports.NEWLINE_SEPARATOR = /\u000d?\u000a/;
/**
* Returns a data structure that can be loaded by the TrieModel.
*
* It implements a **weighted** trie, whose indices (paths down the trie) are
* generated by a search key, and not concrete wordforms themselves.
*
* @param wordlist a complete mapping of words to frequencies
*/
function compileTrieFromWordlist(wordlist, searchTermToKey) {
let trie = Trie.buildTrie(wordlist, searchTermToKey);
return JSON.stringify(trie);
}
exports.compileTrieFromWordlist = compileTrieFromWordlist;
/**
* Parses a word list from a string. The string should have multiple lines
* with LF or CRLF line terminators.
*
* @param wordlist word list to merge entries into (may have existing entries)
* @param filename filename of the word list
*/
function parseWordListFromContents(wordlist, contents) {
parse_wordlist_1.parseWordList(wordlist, new WordListFromMemory(contents));
}
exports.parseWordListFromContents = parseWordListFromContents;
class WordListFromMemory {
constructor(contents) {
this.name = '<memory>';
this._contents = contents;
}
*lines() {
yield* enumerateLines(this._contents.split(exports.NEWLINE_SEPARATOR));
}
}
/**
* Yields pairs of [lineno, line], given an Array of lines.
*/
function* enumerateLines(lines) {
let i = 1;
for (let line of lines) {
yield [i, line];
i++;
}
}
exports.enumerateLines = enumerateLines;
var Trie;
(function (Trie_1) {
/**
* A sentinel value for when an internal node has contents and requires an
* "internal" leaf. That is, this internal node has content. Instead of placing
* entries as children in an internal node, a "fake" leaf is created, and its
* key is this special internal value.
*
* The value is a valid Unicode BMP code point, but it is a "non-character".
* Unicode will never assign semantics to these characters, as they are
* intended to be used internally as sentinel values.
*/
const INTERNAL_VALUE = '\uFDD0';
/**
* Builds a trie from a word list.
*
* @param wordlist The wordlist with non-negative weights.
* @param keyFunction Function that converts word forms into indexed search keys
* @returns A JSON-serialiable object that can be given to the TrieModel constructor.
*/
function buildTrie(wordlist, keyFunction) {
let root = new Trie(keyFunction).buildFromWordList(wordlist).root;
return {
totalWeight: sumWeights(root),
root: root
};
}
Trie_1.buildTrie = buildTrie;
/**
* Wrapper class for the trie and its nodes and wordform to search
*/
class Trie {
constructor(wordform2key) {
this.root = createRootNode();
this.toKey = wordform2key;
}
/**
* Populates the trie with the contents of an entire wordlist.
* @param words a list of word and count pairs.
*/
buildFromWordList(words) {
for (let [wordform, weight] of Object.entries(words)) {
let key = this.toKey(wordform);
addUnsorted(this.root, { key, weight, content: wordform }, 0);
}
sortTrie(this.root);
return this;
}
}
// "Constructors"
function createRootNode() {
return {
type: 'leaf',
weight: 0,
entries: []
};
}
// Implement Trie creation.
/**
* Adds an entry to the trie.
*
* Note that the trie will likely be unsorted after the add occurs. Before
* performing a lookup on the trie, use call sortTrie() on the root note!
*
* @param node Which node should the entry be added to?
* @param entry the wordform/weight/key to add to the trie
* @param index the index in the key and also the trie depth. Should be set to
* zero when adding onto the root node of the trie.
*/
function addUnsorted(node, entry, index = 0) {
// Each node stores the MAXIMUM weight out of all of its decesdents, to
// enable a greedy search through the trie.
node.weight = Math.max(node.weight, entry.weight);
// When should a leaf become an interior node?
// When it already has a value, but the key of the current value is longer
// than the prefix.
if (node.type === 'leaf' && index < entry.key.length && node.entries.length >= 1) {
convertLeafToInternalNode(node, index);
}
if (node.type === 'leaf') {
// The key matches this leaf node, so add yet another entry.
addItemToLeaf(node, entry);
}
else {
// Push the node down to a lower node.
addItemToInternalNode(node, entry, index);
}
node.unsorted = true;
}
/**
* Adds an item to the internal node at a given depth.
* @param item
* @param index
*/
function addItemToInternalNode(node, item, index) {
let char = item.key[index];
if (!node.children[char]) {
node.children[char] = createRootNode();
node.values.push(char);
}
addUnsorted(node.children[char], item, index + 1);
}
function addItemToLeaf(leaf, item) {
leaf.entries.push(item);
}
/**
* Mutates the given Leaf to turn it into an InternalNode.
*
* NOTE: the node passed in will be DESTRUCTIVELY CHANGED into a different
* type when passed into this function!
*
* @param depth depth of the trie at this level.
*/
function convertLeafToInternalNode(leaf, depth) {
let entries = leaf.entries;
// Alias the current node, as the desired type.
let internal = leaf;
internal.type = 'internal';
delete leaf.entries;
internal.values = [];
internal.children = {};
// Convert the old values array into the format for interior nodes.
for (let item of entries) {
let char;
if (depth < item.key.length) {
char = item.key[depth];
}
else {
char = INTERNAL_VALUE;
}
if (!internal.children[char]) {
internal.children[char] = createRootNode();
internal.values.push(char);
}
addUnsorted(internal.children[char], item, depth + 1);
}
internal.unsorted = true;
}
/**
* Recursively sort the trie, in descending order of weight.
* @param node any node in the trie
*/
function sortTrie(node) {
if (node.type === 'leaf') {
if (!node.unsorted) {
return;
}
node.entries.sort(function (a, b) { return b.weight - a.weight; });
}
else {
// We MUST recurse and sort children before returning.
for (let char of node.values) {
sortTrie(node.children[char]);
}
if (!node.unsorted) {
return;
}
node.values.sort((a, b) => {
return node.children[b].weight - node.children[a].weight;
});
}
delete node.unsorted;
}
/**
* O(n) recursive traversal to sum the total weight of all leaves in the
* trie, starting at the provided node.
*
* @param node The node to start summing weights.
*/
function sumWeights(node) {
if (node.type === 'leaf') {
return node.entries
.map(entry => entry.weight)
.reduce((acc, count) => acc + count, 0);
}
else {
return Object.keys(node.children)
.map((key) => sumWeights(node.children[key]))
.reduce((acc, count) => acc + count, 0);
}
}
})(Trie || (Trie = {}));
/**
* Converts wordforms into an indexable form. It does this by
* normalizing the letter case of characters INDIVIDUALLY (to disregard
* context-sensitive case transformations), normalizing to NFKD form,
* and removing common diacritical marks.
*
* This is a very speculative implementation, that might work with
* your language. We don't guarantee that this will be perfect for your
* language, but it's a start.
*
* This uses String.prototype.normalize() to convert normalize into NFKD.
* NFKD neutralizes some funky distinctions, e.g., ꬲ, e, e should all be the
* same character; plus, it's an easy way to separate a Latin character from
* its diacritics; Even then, orthographies regularly use code points
* that, under NFKD normalization, do NOT decompose appropriately for your
* language (e.g., SENĆOŦEN, Plains Cree in syllabics).
*
* Use this in early iterations of the model. For a production lexical model,
* you will probably write/generate your own key function, tailored to your
* language. There is a chance the default will work properly out of the box.
*/
function defaultSearchTermToKey(wordform) {
return Array.from(wordform)
.map(c => c.toLowerCase())
.join('')
.normalize('NFKD')
// Remove any combining diacritics (if input is in NFKD)
.replace(/[\u0300-\u036F]/g, '');
}
exports.defaultSearchTermToKey = defaultSearchTermToKey;
/**
* Detects the encoding of a text file.
*
* Supported encodings are:
*
* - UTF-8, with or without BOM
* - UTF-16, little endian, with BOM
*
* UTF-16 in big endian is explicitly NOT supported! The reason is two-fold:
* 1) Node does not support it without resorting to an external library (or
* swapping every byte in the file!); and 2) I'm not sure anything actually
* outputs in this format anyway!
*
* @param filename filename of the file to detect encoding
*/
function detectEncodingFromBuffer(buffer) {
// Note: BOM is U+FEFF
// In little endian, this is 0xFF 0xFE
if (buffer[0] == 0xFF && buffer[1] == 0xFE) {
return 'utf16le';
}
else if (buffer[0] == 0xFE && buffer[1] == 0xFF) {
// Big Endian, is NOT supported because Node does not support it (???)
// See: https://stackoverflow.com/a/14551669/6626414
throw new Error('UTF-16BE is unsupported');
}
else {
// Assume its in UTF-8, with or without a BOM.
return 'utf8';
}
}
exports.detectEncodingFromBuffer = detectEncodingFromBuffer;
//# sourceMappingURL=index.js.map