@trackr/effects-gain
Version:
A subpackage of trackr that contains gain effects.
195 lines (194 loc) • 8.7 kB
JavaScript
var __awaiter = (this && this.__awaiter) || function (thisArg, _arguments, P, generator) {
function adopt(value) { return value instanceof P ? value : new P(function (resolve) { resolve(value); }); }
return new (P || (P = Promise))(function (resolve, reject) {
function fulfilled(value) { try { step(generator.next(value)); } catch (e) { reject(e); } }
function rejected(value) { try { step(generator["throw"](value)); } catch (e) { reject(e); } }
function step(result) { result.done ? resolve(result.value) : adopt(result.value).then(fulfilled, rejected); }
step((generator = generator.apply(thisArg, _arguments || [])).next());
});
};
import { Normalizer, Score, Vector, Configuration } from ".";
import { inspectText } from "./inspect-text";
import { intersect } from "./utilities";
export class TextHill {
constructor(s, normalizer = new Normalizer(), configuration = new Configuration()) {
this.s = s;
this.normalizer = normalizer;
this.configuration = configuration;
this._N = 0;
}
feedDoc(key, unstructuredDoc, options) {
var _a;
return __awaiter(this, void 0, void 0, function* () {
var ignoreProps = (_a = options === null || options === void 0 ? void 0 : options.ignoreProps) !== null && _a !== void 0 ? _a : [];
let text = inspectText(unstructuredDoc, ignoreProps);
// lookup if doc already exist
const [docs_map, docIds_map, index, tf, latestDocId] = yield Promise.all([this.s.getItem("docs", {}),
this.s.getItem("docIds", {}),
this.s.getItem("index", {}),
this.s.getItem("tf", {}),
this.s.getItem(TextHill.LATEST_DOCID)]);
return yield this._feedDocBy(key, text, docs_map, docIds_map, index, tf, latestDocId);
});
}
_feedDocBy(key, unstructuredDoc, docs_map, docIds_map, index, tf, latestDocId) {
return __awaiter(this, void 0, void 0, function* () {
var docInfo = docs_map[key];
let docId;
if (docInfo == null) {
// put docId info into persistence
docId = this._latestDocId(latestDocId);
docs_map[key] = docId;
yield this.s.setItem("docs", docs_map);
docIds_map[`${docId}`] = key;
yield this.s.setItem("docIds", docIds_map);
}
else {
docId = docInfo;
this.removeDocIdFromIndex(index, tf, docId);
}
const words = unstructuredDoc.split(" ");
for (let word of words) {
word = this.normalizer.normalize(word);
if (!this.configuration.skipWord(word)) {
let wordSet = index[word];
if (wordSet == null) {
wordSet = [];
}
if (wordSet.indexOf(docId) === -1) {
wordSet.push(docId);
index[word] = wordSet;
}
tf = this._setTfInStore(tf, `${docId}`, word);
}
}
yield this.s.setItem("tf", tf);
yield this.s.setItem("index", index);
return docId;
});
}
removeDocIdFromIndex(index, tf, docId) {
// docId already exist so clear the document in the index before re-indexing the new document
let removals = [];
Object.keys(index).forEach((key) => {
const value = index[key];
if (Array.isArray(value)) {
const postings = value;
postings.splice(postings.indexOf(docId), 1);
if (postings.length === 0) {
removals.push(key);
}
}
});
removals.forEach((o) => delete index[o]);
removals = [];
Object.keys(tf).forEach((key) => {
const value = tf[key];
if (Array.isArray(value)) {
const mapWithDocId = value;
mapWithDocId.slice(mapWithDocId.indexOf(`${docId}`), 1);
if (mapWithDocId.length == 0) {
removals.push(key);
}
}
});
removals.forEach((o) => delete tf[o]);
return { index, tf };
}
removeDoc(key) {
return __awaiter(this, void 0, void 0, function* () {
// lookup if doc already exist
const [docs_map, docIds_map, index, tf, latestDocId] = yield Promise.all([this.s.getItem("docs", {}),
this.s.getItem("docIds", {}),
this.s.getItem("index", {}),
this.s.getItem("tf", {}),
this.s.getItem(TextHill.LATEST_DOCID)]);
return yield this._removeDocBy(key, docs_map, docIds_map, index, tf, latestDocId);
});
}
_removeDocBy(key, docs_map, docIds_map, index, tf, latestDocId) {
return __awaiter(this, void 0, void 0, function* () {
var docId = docs_map[key];
if (docId != null) {
delete docs_map[key];
yield this.s.setItem("docs", docs_map);
delete docIds_map[`${key}`];
yield this.s.setItem("docIds", docIds_map);
this.removeDocIdFromIndex(index, tf, docId);
yield this.s.setItem("tf", tf);
yield this.s.setItem("index", index);
}
});
}
search(sentence) {
return __awaiter(this, void 0, void 0, function* () {
let findDocs = [];
const index = yield this.s.getItem('index');
if (index != null) {
let docIdsRetrieval;
for (let term of sentence.split(" ")) {
term = this.normalizer.normalize(term);
if (index[term] != null && !this.configuration.skipWord(term)) {
if (docIdsRetrieval == null) {
docIdsRetrieval = new Set(index[term]);
}
else {
docIdsRetrieval = intersect(docIdsRetrieval, new Set(index[term]));
}
}
}
// calculate scores for every document
const docIds = yield this.s.getItem('docIds');
const tf = yield this.s.getItem('tf');
const N = Object.keys(docIds).length;
if (docIdsRetrieval != null) {
for (const docId of docIdsRetrieval) {
let scorings = new Vector();
let terms = sentence.split(" ");
for (let term of sentence.split(" ")) {
term = this.normalizer.normalize(term);
let tf_value = tf[term] != null ? tf[term][`${docId}`] : 0;
const postings = new Set(index[term]);
let df = postings != null ? postings.size : 0;
const score = (1 + Math.log(tf_value)) * Math.log(N / df);
scorings.add(score);
}
// only normalize it when you have more then one terms
if (terms.length > 1) {
scorings = scorings.normalize();
}
const totalScore = scorings.avg();
findDocs.push(new Score(totalScore, docId, docIds[`${docId}`]));
}
}
}
// sort the bounties on score
findDocs.sort((a, b) => a.compareTo(b));
return findDocs;
});
}
// set a term frequency in a certain document
_setTfInStore(tf, docId, word) {
let tf_map = tf[word];
if (tf_map == null) {
tf_map = new Map();
}
if (tf_map[docId] == null) {
tf_map[docId] = 0;
}
tf_map[docId]++;
tf[word] = tf_map;
return tf;
}
_latestDocId(latestDocId) {
if (latestDocId == null) {
this.s.setItem(TextHill.LATEST_DOCID, 1);
latestDocId = 0;
}
else {
this.s.setItem(TextHill.LATEST_DOCID, latestDocId + 1);
}
return latestDocId;
}
}
TextHill.LATEST_DOCID = "latest_docId";