sanity-plugin-taxonomy-manager
Version:
Create and manage SKOS compliant taxonomies, thesauri, and classification schemes in Sanity Studio.
128 lines (117 loc) • 5.67 kB
text/typescript
import {getPublishedId, type DocumentId} from '@sanity/id-utils'
import type {ConceptRecommendation} from '../types'
/**
* #### Semantic Recommendations (pure core)
* The decision-bearing logic for the semantic term-recommendation seam: the GROQ
* query, the query-text assembly + validation, result mapping, and error
* messaging. The Studio adapter (`useSemanticRecommendations`) supplies the form
* values, runs the query, and owns React state; everything here is pure so it can
* be unit-tested without Studio or the network.
*
* Replaces the deprecated Embeddings Index API (`/embeddings-index/query/…`,
* `text::embedding()`) with dataset embeddings queried through
* `text::semanticSimilarity()` — no named index, scored in GROQ.
*/
/**
* GROQ for the recommendations query. Scores the `skosConcept`s belonging to the
* scheme identified by `$schemeId` — those referenced by that scheme's `topConcepts`
* or `concepts` (concepts carry no scheme back-reference, so membership runs through
* the scheme) — by semantic similarity to `$searchQuery`, returning the top
* `$maxResults` as `{conceptId, score}`, most-relevant first. Scoping to the field's
* scheme keeps recommendations within the vocabulary the field draws from rather than
* the whole dataset. `text::semanticSimilarity()` is only valid inside `score()`, and
* the dataset must have embeddings enabled or the query errors (surfaced via
* {@link recommendationsErrorMessage}).
*
* Scoping is scheme-level: a `branchFilter` field still scores the whole scheme, not
* just the displayed branch. Branch-level scoping is tracked in #93.
*/
export const recommendationsQuery = (): string =>
`*[_type == "skosConcept" && _id in *[_type == "skosConceptScheme" && schemeId == $schemeId][0]{
"ids": coalesce(topConcepts[]._ref, []) + coalesce(concepts[]._ref, [])
}.ids] | score(text::semanticSimilarity($searchQuery)) | order(_score desc) [0...$maxResults] {
"conceptId": _id,
"score": _score
}`
/** A document field name paired with its current value, read from the edited document. */
export interface RecommendationField {
name: string
value: unknown
}
/**
* Concatenate the referenced fields' values into the semantic-search query text.
* Every field must hold a non-empty string; otherwise this throws with a message
* naming the empty field(s), which the adapter surfaces as `recsError` rather than
* searching against partial input. The messages are preserved verbatim from the
* pre-migration embeddings hook.
*/
export function assembleQueryText(fields: RecommendationField[]): string {
const values: string[] = []
const emptyFields: string[] = []
for (const {name, value} of fields) {
if (typeof value === 'string' && value.trim() !== '') {
values.push(value)
} else {
emptyFields.push(name)
}
}
if (emptyFields.length === 1) {
throw new Error(`Please fill out the ${emptyFields[0]} field to enable match scores.`)
}
if (emptyFields.length > 1) {
throw new Error(
`The following fields must be filled out to enable match scores: ${emptyFields.join(', ')}`
)
}
return values.join(' ')
}
/** A raw row from {@link recommendationsQuery}, before id normalization. */
export interface RecommendationRow {
conceptId: string
score: number
}
/**
* Map raw query rows to {@link ConceptRecommendation}s, normalizing each concept id
* to its published form so they line up with `annotateRanks`, which keys the tree by
* published id. Rows missing a string id or numeric score are dropped.
*/
export function toConceptRecommendations(
rows: readonly RecommendationRow[] | null | undefined
): ConceptRecommendation[] {
if (!rows) return []
const recs: ConceptRecommendation[] = []
for (const row of rows) {
if (!row || typeof row.conceptId !== 'string' || row.conceptId === '') continue
if (typeof row.score !== 'number') continue
recs.push({conceptId: getPublishedId(row.conceptId as DocumentId), score: row.score})
}
return recs
}
/**
* The set of published concept ids that are semantic recommendations. The raw
* `text::semanticSimilarity()` score is opaque and unitless (meaningful only for
* relative ordering within one query), so we keep only set membership — a concept
* is either recommended or it isn't — and let the query's `maxResults` bound the
* size. De-duplicated by construction.
*/
export function recommendedConceptIds(recs: readonly ConceptRecommendation[]): Set<string> {
return new Set(recs.map((rec) => rec.conceptId))
}
/** Friendly message shown when a dataset has no embeddings enabled. */
const EMBEDDINGS_DISABLED_MESSAGE = "Semantic recommendations aren't enabled for this dataset."
/** Fallback for any other recommendations failure. */
const GENERIC_RECS_ERROR_MESSAGE = 'Unable to load semantic recommendations.'
/**
* Map a failed recommendations fetch to a friendly `recsError` message. A dataset
* without embeddings is the expected failure — `text::semanticSimilarity()` errors
* there — so an embeddings-related error becomes a clear setup hint; anything else
* falls back to a generic message. Either way the tree still renders, only without
* scores.
*
* The embeddings-disabled detection matches on the error text; confirm it against
* the live API error in a `pnpm dev` pass and widen the match if the wording differs.
*/
export function recommendationsErrorMessage(error: unknown): string {
const text = error instanceof Error ? error.message : typeof error === 'string' ? error : ''
return /embedding/i.test(text) ? EMBEDDINGS_DISABLED_MESSAGE : GENERIC_RECS_ERROR_MESSAGE
}