UNPKG

langextract

Version:

A TypeScript library for extracting structured and grounded information from text using LLMs

36 lines 1.32 kB
/** * Copyright 2025 kmbro. * * This is a TypeScript translation of the original Python LangExtract library * by Google LLC (https://github.com/google/langextract). * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. * You may obtain a copy of the License at * * http://www.apache.org/licenses/LICENSE-2.0 * * Unless required by applicable law or agreed to in writing, software * distributed under the License is distributed on an "AS IS" BASIS, * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. * See the License for the specific language governing permissions and * limitations under the License. */ /** * Simple tokenization utilities. */ import { TokenizedText } from "./types"; /** * Tokenizes text into words and tracks character positions. * This is a simplified tokenizer that splits on whitespace and punctuation. */ export declare function tokenize(text: string): TokenizedText; /** * Normalizes a token for comparison (lowercase, trim). */ export declare function normalizeToken(token: string): string; /** * Tokenizes text with lowercase normalization. */ export declare function tokenizeWithLowercase(text: string): string[]; //# sourceMappingURL=tokenizer.d.ts.map