langextract
Version:
A TypeScript library for extracting structured and grounded information from text using LLMs
36 lines • 1.32 kB
TypeScript
/**
* Copyright 2025 kmbro.
*
* This is a TypeScript translation of the original Python LangExtract library
* by Google LLC (https://github.com/google/langextract).
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
/**
* Simple tokenization utilities.
*/
import { TokenizedText } from "./types";
/**
* Tokenizes text into words and tracks character positions.
* This is a simplified tokenizer that splits on whitespace and punctuation.
*/
export declare function tokenize(text: string): TokenizedText;
/**
* Normalizes a token for comparison (lowercase, trim).
*/
export declare function normalizeToken(token: string): string;
/**
* Tokenizes text with lowercase normalization.
*/
export declare function tokenizeWithLowercase(text: string): string[];
//# sourceMappingURL=tokenizer.d.ts.map