@n8n/n8n-nodes-langchain
Version:

153 lines (148 loc) • 5.92 kB
text/typescript
/**
* Ollama Node - Version 1
* Discriminator: resource=text, operation=message
*/
interface Credentials {
ollamaApi: CredentialReference;
}
/** Send a message to Ollama model */
export type LcOllamaV1TextMessageParams = {
resource: 'text';
operation: 'message';
/**
* Model
* @searchListMethod modelSearch
* @default {"mode":"list","value":""}
*/
modelId?: { __rl: true; mode: 'list' | 'id'; value: string; cachedResultName?: string };
/**
* Messages
* @default {"values":[{"content":"","role":"user"}]}
*/
messages?: {
/** Values
*/
values?: Array<{
/** The content of the message to be sent
*/
content?: string | Expression<string> | PlaceholderValue;
/** The role of this message in the conversation
* @default user
*/
role?: 'user' | 'assistant' | Expression<string>;
}>;
};
/**
* Whether to simplify the response or not
* @default true
*/
simplify?: boolean | Expression<boolean>;
/**
* Options
* @default {}
*/
options?: {
/** System message to set the context for the conversation
*/
system?: string | Expression<string> | PlaceholderValue;
/** Controls randomness in responses. Lower values make output more focused.
* @default 0.8
*/
temperature?: number | Expression<number>;
/** The maximum cumulative probability of tokens to consider when sampling
* @default 0.7
*/
top_p?: number | Expression<number>;
/** Controls diversity by limiting the number of top tokens to consider
* @default 40
*/
top_k?: number | Expression<number>;
/** Maximum number of tokens to generate in the completion
* @default 1024
*/
num_predict?: number | Expression<number>;
/** Adjusts the penalty for tokens that have already appeared in the generated text. Higher values discourage repetition.
* @default 0
*/
frequency_penalty?: number | Expression<number>;
/** Adjusts the penalty for tokens based on their presence in the generated text so far. Positive values penalize tokens that have already appeared, encouraging diversity.
* @default 0
*/
presence_penalty?: number | Expression<number>;
/** Sets how strongly to penalize repetitions. A higher value (e.g., 1.5) will penalize repetitions more strongly, while a lower value (e.g., 0.9) will be more lenient.
* @default 1.1
*/
repeat_penalty?: number | Expression<number>;
/** Sets the size of the context window used to generate the next token
* @default 4096
*/
num_ctx?: number | Expression<number>;
/** Sets how far back for the model to look back to prevent repetition. (0 = disabled, -1 = num_ctx).
* @default 64
*/
repeat_last_n?: number | Expression<number>;
/** Alternative to the top_p, and aims to ensure a balance of quality and variety. The parameter p represents the minimum probability for a token to be considered, relative to the probability of the most likely token.
* @default 0
*/
min_p?: number | Expression<number>;
/** Sets the random number seed to use for generation. Setting this to a specific number will make the model generate the same text for the same prompt.
* @default 0
*/
seed?: number | Expression<number>;
/** Sets the stop sequences to use. When this pattern is encountered the LLM will stop generating text and return. Separate multiple patterns with commas
*/
stop?: string | Expression<string> | PlaceholderValue;
/** Specifies the duration to keep the loaded model in memory after use. Format: 1h30m (1 hour 30 minutes).
* @default 5m
*/
keep_alive?: string | Expression<string> | PlaceholderValue;
/** Whether to activate low VRAM mode, which reduces memory usage at the cost of slower generation speed. Useful for GPUs with limited memory.
* @default false
*/
low_vram?: boolean | Expression<boolean>;
/** Specifies the ID of the GPU to use for the main computation. Only change this if you have multiple GPUs.
* @default 0
*/
main_gpu?: number | Expression<number>;
/** Sets the batch size for prompt processing. Larger batch sizes may improve generation speed but increase memory usage.
* @default 512
*/
num_batch?: number | Expression<number>;
/** Specifies the number of GPUs to use for parallel processing. Set to -1 for auto-detection.
* @default -1
*/
num_gpu?: number | Expression<number>;
/** Specifies the number of CPU threads to use for processing. Set to 0 for auto-detection.
* @default 0
*/
num_thread?: number | Expression<number>;
/** Whether the model will be less likely to generate newline characters, encouraging longer continuous sequences of text
* @default true
*/
penalize_newline?: boolean | Expression<boolean>;
/** Whether to lock the model in memory to prevent swapping. This can improve performance but requires sufficient available memory.
* @default false
*/
use_mlock?: boolean | Expression<boolean>;
/** Whether to use memory mapping for loading the model. This can reduce memory usage but may impact performance.
* @default true
*/
use_mmap?: boolean | Expression<boolean>;
/** Whether to only load the model vocabulary without the weights. Useful for quickly testing tokenization.
* @default false
*/
vocab_only?: boolean | Expression<boolean>;
/** Specifies the format of the API response
*/
format?: '' | 'json' | Expression<string>;
};
};
export interface LcOllamaV1TextMessageSubnodeConfig {
tools?: ToolInstance[];
}
export type LcOllamaV1TextMessageNode = {
type: '@n8n/n8n-nodes-langchain.ollama';
version: 1;
credentials?: Credentials;
config: NodeConfig<LcOllamaV1TextMessageParams> & { subnodes?: LcOllamaV1TextMessageSubnodeConfig };
};