UNPKG

@predictive-text-studio/lexical-model-compiler

Version:
294 lines 10.7 kB
"use strict"; Object.defineProperty(exports, "__esModule", { value: true }); const parse_wordlist_1 = require("../parse-wordlist"); // Supports LF or CRLF line terminators. exports.NEWLINE_SEPARATOR = /\u000d?\u000a/; /** * Returns a data structure that can be loaded by the TrieModel. * * It implements a **weighted** trie, whose indices (paths down the trie) are * generated by a search key, and not concrete wordforms themselves. * * @param wordlist a complete mapping of words to frequencies */ function compileTrieFromWordlist(wordlist, searchTermToKey) { let trie = Trie.buildTrie(wordlist, searchTermToKey); return JSON.stringify(trie); } exports.compileTrieFromWordlist = compileTrieFromWordlist; /** * Parses a word list from a string. The string should have multiple lines * with LF or CRLF line terminators. * * @param wordlist word list to merge entries into (may have existing entries) * @param filename filename of the word list */ function parseWordListFromContents(wordlist, contents) { parse_wordlist_1.parseWordList(wordlist, new WordListFromMemory(contents)); } exports.parseWordListFromContents = parseWordListFromContents; class WordListFromMemory { constructor(contents) { this.name = '<memory>'; this._contents = contents; } *lines() { yield* enumerateLines(this._contents.split(exports.NEWLINE_SEPARATOR)); } } /** * Yields pairs of [lineno, line], given an Array of lines. */ function* enumerateLines(lines) { let i = 1; for (let line of lines) { yield [i, line]; i++; } } exports.enumerateLines = enumerateLines; var Trie; (function (Trie_1) { /** * A sentinel value for when an internal node has contents and requires an * "internal" leaf. That is, this internal node has content. Instead of placing * entries as children in an internal node, a "fake" leaf is created, and its * key is this special internal value. * * The value is a valid Unicode BMP code point, but it is a "non-character". * Unicode will never assign semantics to these characters, as they are * intended to be used internally as sentinel values. */ const INTERNAL_VALUE = '\uFDD0'; /** * Builds a trie from a word list. * * @param wordlist The wordlist with non-negative weights. * @param keyFunction Function that converts word forms into indexed search keys * @returns A JSON-serialiable object that can be given to the TrieModel constructor. */ function buildTrie(wordlist, keyFunction) { let root = new Trie(keyFunction).buildFromWordList(wordlist).root; return { totalWeight: sumWeights(root), root: root }; } Trie_1.buildTrie = buildTrie; /** * Wrapper class for the trie and its nodes and wordform to search */ class Trie { constructor(wordform2key) { this.root = createRootNode(); this.toKey = wordform2key; } /** * Populates the trie with the contents of an entire wordlist. * @param words a list of word and count pairs. */ buildFromWordList(words) { for (let [wordform, weight] of Object.entries(words)) { let key = this.toKey(wordform); addUnsorted(this.root, { key, weight, content: wordform }, 0); } sortTrie(this.root); return this; } } // "Constructors" function createRootNode() { return { type: 'leaf', weight: 0, entries: [] }; } // Implement Trie creation. /** * Adds an entry to the trie. * * Note that the trie will likely be unsorted after the add occurs. Before * performing a lookup on the trie, use call sortTrie() on the root note! * * @param node Which node should the entry be added to? * @param entry the wordform/weight/key to add to the trie * @param index the index in the key and also the trie depth. Should be set to * zero when adding onto the root node of the trie. */ function addUnsorted(node, entry, index = 0) { // Each node stores the MAXIMUM weight out of all of its decesdents, to // enable a greedy search through the trie. node.weight = Math.max(node.weight, entry.weight); // When should a leaf become an interior node? // When it already has a value, but the key of the current value is longer // than the prefix. if (node.type === 'leaf' && index < entry.key.length && node.entries.length >= 1) { convertLeafToInternalNode(node, index); } if (node.type === 'leaf') { // The key matches this leaf node, so add yet another entry. addItemToLeaf(node, entry); } else { // Push the node down to a lower node. addItemToInternalNode(node, entry, index); } node.unsorted = true; } /** * Adds an item to the internal node at a given depth. * @param item * @param index */ function addItemToInternalNode(node, item, index) { let char = item.key[index]; if (!node.children[char]) { node.children[char] = createRootNode(); node.values.push(char); } addUnsorted(node.children[char], item, index + 1); } function addItemToLeaf(leaf, item) { leaf.entries.push(item); } /** * Mutates the given Leaf to turn it into an InternalNode. * * NOTE: the node passed in will be DESTRUCTIVELY CHANGED into a different * type when passed into this function! * * @param depth depth of the trie at this level. */ function convertLeafToInternalNode(leaf, depth) { let entries = leaf.entries; // Alias the current node, as the desired type. let internal = leaf; internal.type = 'internal'; delete leaf.entries; internal.values = []; internal.children = {}; // Convert the old values array into the format for interior nodes. for (let item of entries) { let char; if (depth < item.key.length) { char = item.key[depth]; } else { char = INTERNAL_VALUE; } if (!internal.children[char]) { internal.children[char] = createRootNode(); internal.values.push(char); } addUnsorted(internal.children[char], item, depth + 1); } internal.unsorted = true; } /** * Recursively sort the trie, in descending order of weight. * @param node any node in the trie */ function sortTrie(node) { if (node.type === 'leaf') { if (!node.unsorted) { return; } node.entries.sort(function (a, b) { return b.weight - a.weight; }); } else { // We MUST recurse and sort children before returning. for (let char of node.values) { sortTrie(node.children[char]); } if (!node.unsorted) { return; } node.values.sort((a, b) => { return node.children[b].weight - node.children[a].weight; }); } delete node.unsorted; } /** * O(n) recursive traversal to sum the total weight of all leaves in the * trie, starting at the provided node. * * @param node The node to start summing weights. */ function sumWeights(node) { if (node.type === 'leaf') { return node.entries .map(entry => entry.weight) .reduce((acc, count) => acc + count, 0); } else { return Object.keys(node.children) .map((key) => sumWeights(node.children[key])) .reduce((acc, count) => acc + count, 0); } } })(Trie || (Trie = {})); /** * Converts wordforms into an indexable form. It does this by * normalizing the letter case of characters INDIVIDUALLY (to disregard * context-sensitive case transformations), normalizing to NFKD form, * and removing common diacritical marks. * * This is a very speculative implementation, that might work with * your language. We don't guarantee that this will be perfect for your * language, but it's a start. * * This uses String.prototype.normalize() to convert normalize into NFKD. * NFKD neutralizes some funky distinctions, e.g., ꬲ, e, e should all be the * same character; plus, it's an easy way to separate a Latin character from * its diacritics; Even then, orthographies regularly use code points * that, under NFKD normalization, do NOT decompose appropriately for your * language (e.g., SENĆOŦEN, Plains Cree in syllabics). * * Use this in early iterations of the model. For a production lexical model, * you will probably write/generate your own key function, tailored to your * language. There is a chance the default will work properly out of the box. */ function defaultSearchTermToKey(wordform) { return Array.from(wordform) .map(c => c.toLowerCase()) .join('') .normalize('NFKD') // Remove any combining diacritics (if input is in NFKD) .replace(/[\u0300-\u036F]/g, ''); } exports.defaultSearchTermToKey = defaultSearchTermToKey; /** * Detects the encoding of a text file. * * Supported encodings are: * * - UTF-8, with or without BOM * - UTF-16, little endian, with BOM * * UTF-16 in big endian is explicitly NOT supported! The reason is two-fold: * 1) Node does not support it without resorting to an external library (or * swapping every byte in the file!); and 2) I'm not sure anything actually * outputs in this format anyway! * * @param filename filename of the file to detect encoding */ function detectEncodingFromBuffer(buffer) { // Note: BOM is U+FEFF // In little endian, this is 0xFF 0xFE if (buffer[0] == 0xFF && buffer[1] == 0xFE) { return 'utf16le'; } else if (buffer[0] == 0xFE && buffer[1] == 0xFF) { // Big Endian, is NOT supported because Node does not support it (???) // See: https://stackoverflow.com/a/14551669/6626414 throw new Error('UTF-16BE is unsupported'); } else { // Assume its in UTF-8, with or without a BOM. return 'utf8'; } } exports.detectEncodingFromBuffer = detectEncodingFromBuffer; //# sourceMappingURL=index.js.map