UNPKG

sanity-plugin-taxonomy-manager

Version:

Create and manage SKOS compliant taxonomies, thesauri, and classification schemes in Sanity Studio.

128 lines (117 loc) 5.67 kB
import {getPublishedId, type DocumentId} from '@sanity/id-utils' import type {ConceptRecommendation} from '../types' /** * #### Semantic Recommendations (pure core) * The decision-bearing logic for the semantic term-recommendation seam: the GROQ * query, the query-text assembly + validation, result mapping, and error * messaging. The Studio adapter (`useSemanticRecommendations`) supplies the form * values, runs the query, and owns React state; everything here is pure so it can * be unit-tested without Studio or the network. * * Replaces the deprecated Embeddings Index API (`/embeddings-index/query/…`, * `text::embedding()`) with dataset embeddings queried through * `text::semanticSimilarity()` — no named index, scored in GROQ. */ /** * GROQ for the recommendations query. Scores the `skosConcept`s belonging to the * scheme identified by `$schemeId` — those referenced by that scheme's `topConcepts` * or `concepts` (concepts carry no scheme back-reference, so membership runs through * the scheme) — by semantic similarity to `$searchQuery`, returning the top * `$maxResults` as `{conceptId, score}`, most-relevant first. Scoping to the field's * scheme keeps recommendations within the vocabulary the field draws from rather than * the whole dataset. `text::semanticSimilarity()` is only valid inside `score()`, and * the dataset must have embeddings enabled or the query errors (surfaced via * {@link recommendationsErrorMessage}). * * Scoping is scheme-level: a `branchFilter` field still scores the whole scheme, not * just the displayed branch. Branch-level scoping is tracked in #93. */ export const recommendationsQuery = (): string => `*[_type == "skosConcept" && _id in *[_type == "skosConceptScheme" && schemeId == $schemeId][0]{ "ids": coalesce(topConcepts[]._ref, []) + coalesce(concepts[]._ref, []) }.ids] | score(text::semanticSimilarity($searchQuery)) | order(_score desc) [0...$maxResults] { "conceptId": _id, "score": _score }` /** A document field name paired with its current value, read from the edited document. */ export interface RecommendationField { name: string value: unknown } /** * Concatenate the referenced fields' values into the semantic-search query text. * Every field must hold a non-empty string; otherwise this throws with a message * naming the empty field(s), which the adapter surfaces as `recsError` rather than * searching against partial input. The messages are preserved verbatim from the * pre-migration embeddings hook. */ export function assembleQueryText(fields: RecommendationField[]): string { const values: string[] = [] const emptyFields: string[] = [] for (const {name, value} of fields) { if (typeof value === 'string' && value.trim() !== '') { values.push(value) } else { emptyFields.push(name) } } if (emptyFields.length === 1) { throw new Error(`Please fill out the ${emptyFields[0]} field to enable match scores.`) } if (emptyFields.length > 1) { throw new Error( `The following fields must be filled out to enable match scores: ${emptyFields.join(', ')}` ) } return values.join(' ') } /** A raw row from {@link recommendationsQuery}, before id normalization. */ export interface RecommendationRow { conceptId: string score: number } /** * Map raw query rows to {@link ConceptRecommendation}s, normalizing each concept id * to its published form so they line up with `annotateRanks`, which keys the tree by * published id. Rows missing a string id or numeric score are dropped. */ export function toConceptRecommendations( rows: readonly RecommendationRow[] | null | undefined ): ConceptRecommendation[] { if (!rows) return [] const recs: ConceptRecommendation[] = [] for (const row of rows) { if (!row || typeof row.conceptId !== 'string' || row.conceptId === '') continue if (typeof row.score !== 'number') continue recs.push({conceptId: getPublishedId(row.conceptId as DocumentId), score: row.score}) } return recs } /** * The set of published concept ids that are semantic recommendations. The raw * `text::semanticSimilarity()` score is opaque and unitless (meaningful only for * relative ordering within one query), so we keep only set membership — a concept * is either recommended or it isn't — and let the query's `maxResults` bound the * size. De-duplicated by construction. */ export function recommendedConceptIds(recs: readonly ConceptRecommendation[]): Set<string> { return new Set(recs.map((rec) => rec.conceptId)) } /** Friendly message shown when a dataset has no embeddings enabled. */ const EMBEDDINGS_DISABLED_MESSAGE = "Semantic recommendations aren't enabled for this dataset." /** Fallback for any other recommendations failure. */ const GENERIC_RECS_ERROR_MESSAGE = 'Unable to load semantic recommendations.' /** * Map a failed recommendations fetch to a friendly `recsError` message. A dataset * without embeddings is the expected failure — `text::semanticSimilarity()` errors * there — so an embeddings-related error becomes a clear setup hint; anything else * falls back to a generic message. Either way the tree still renders, only without * scores. * * The embeddings-disabled detection matches on the error text; confirm it against * the live API error in a `pnpm dev` pass and widen the match if the wording differs. */ export function recommendationsErrorMessage(error: unknown): string { const text = error instanceof Error ? error.message : typeof error === 'string' ? error : '' return /embedding/i.test(text) ? EMBEDDINGS_DISABLED_MESSAGE : GENERIC_RECS_ERROR_MESSAGE }