---
title: "Text embeddings"
canonical: https://docs.qvac.tether.io/sdk/v0.20/ai-capabilities/text-embeddings/
collection: "SDK"
package: "@qvac/sdk"
line: v0.20
current_line: false
---

# Text embeddings (/sdk/v0.20/ai-capabilities/text-embeddings)



## Overview

Text embeddings uses [`qvac-fabric-llm.cpp`](https://github.com/tetherto/qvac-fabric-llm.cpp) as inference engine. Load any supported model using `modelType: "embeddings"`. Then, provide text input as `text` where the value is either a single `string` or an array of strings.

`embed()` returns a single embedding vector (`number[]`) for single text input, or an array of embedding vectors (`number[][]`) for batch input.

## Functions

Use the following sequence of function calls:

1. [`loadModel()`](/sdk/v0.20/reference/api#loadmodel)
2. [`embed()`](/sdk/v0.20/reference/api#embed)
3. [`unloadModel()`](/sdk/v0.20/reference/api#unloadmodel)

For how to use each function, see [SDK — API reference](/sdk/v0.20/reference/api/).

## Models

You can load any [`llama.cpp`](https://github.com/ggml-org/llama.cpp)-compatible embeddings model. Model file format: `*.gguf`.

* If the model is sharded across multiple files (a multi-file bundle), see [Sharded models](/sdk/v0.20/models/sharded-models).
* For models available as constants, see [SDK — Models](/sdk/v0.20/#models).

## Example

The following script shows an example of embedding:

<Tabs>
  <Tab value="js" label="JavaScript" default>
    <WrapCode>
      ```js file=<rootDir>/packages/sdk/dist/examples/embed-p2p.js title="text-embeddings.js" lineNumbers
      import { embed, GTE_LARGE_FP16, loadModel, unloadModel } from '@qvac/sdk';
      function cosineSimilarity(vecA, vecB) {
          let dotProduct = 0;
          for (let i = 0; i < vecA.length; i++) {
              dotProduct += vecA[i] * vecB[i];
          }
          return dotProduct;
      }
      try {
          const modelId = await loadModel({
              modelSrc: GTE_LARGE_FP16,
              onProgress: (p) => {
                  const mb = (n) => (n / 1e6).toFixed(1);
                  const line = `▸ Downloading ${p.percentage.toFixed(0)}% (${mb(p.downloaded)}/${mb(p.total)} MB)`;
                  process.stderr.write(process.stderr.isTTY ? `\r${line}` : `${line}\n`);
                  if (p.percentage >= 100)
                      process.stderr.write('\n');
              },
              modelConfig: {
                  gpuLayers: 99,
                  device: 'gpu'
              }
          });
          console.log('\n▸ Example 1: Single Text Embedding');
          console.log('='.repeat(50));
          const { embedding: singleEmbedding } = await embed({
              modelId,
              text: 'Hello, world!'
          });
          console.log("Input: 'Hello, world!'");
          console.log('Embedding dimensions:', singleEmbedding.length);
          console.log('First 10 values:', singleEmbedding.slice(0, 10));
          console.log('\n▸ Example 2: Batch Text Embeddings');
          console.log('='.repeat(50));
          const texts = [
              'The quick brown fox jumps over the lazy dog',
              'A fast auburn fox leaps over a sleepy canine',
              'Python is a programming language'
          ];
          const { embedding: batchEmbeddings } = await embed({ modelId, text: texts });
          console.log('Input: Array of', texts.length, 'texts');
          console.log('Output: Array of', batchEmbeddings.length, 'embeddings');
          const [emb1, emb2, emb3] = batchEmbeddings;
          if (!emb1 || !emb2 || !emb3) {
              throw new Error('Expected 3 embeddings');
          }
          console.log('Each embedding dimensions:', emb1.length);
          console.log('\n▸ Similarity Analysis');
          console.log('='.repeat(50));
          const similarity1 = cosineSimilarity(emb1, emb2);
          const similarity2 = cosineSimilarity(emb1, emb3);
          console.log('Similarity between texts 1 and 2 (similar meaning):', similarity1.toFixed(4));
          console.log('Similarity between texts 1 and 3 (different topics):', similarity2.toFixed(4));
          console.log('\n▸ Higher values indicate more similar meanings');
          await unloadModel({ modelId, clearStorage: false });
      }
      catch (error) {
          console.error('✖', error);
          process.exit(1);
      }
      ```
    </WrapCode>
  </Tab>

  <Tab value="ts" label="TypeScript">
    <WrapCode>
      ```ts file=<rootDir>/packages/sdk/examples/embed-p2p.ts title="text-embeddings.ts" lineNumbers
      import { embed, GTE_LARGE_FP16, loadModel, unloadModel } from '@qvac/sdk'

      function cosineSimilarity(vecA: number[], vecB: number[]) {
        let dotProduct = 0
        for (let i = 0; i < vecA.length; i++) {
          dotProduct += vecA[i]! * vecB[i]!
        }
        return dotProduct
      }

      try {
        const modelId = await loadModel({
          modelSrc: GTE_LARGE_FP16,
          onProgress: (p) => {
            const mb = (n: number) => (n / 1e6).toFixed(1)
            const line = `▸ Downloading ${p.percentage.toFixed(0)}% (${mb(p.downloaded)}/${mb(p.total)} MB)`
            process.stderr.write(process.stderr.isTTY ? `\r${line}` : `${line}\n`)
            if (p.percentage >= 100) process.stderr.write('\n')
          },
          modelConfig: {
            gpuLayers: 99,
            device: 'gpu'
          }
        })

        console.log('\n▸ Example 1: Single Text Embedding')
        console.log('='.repeat(50))

        const { embedding: singleEmbedding } = await embed({
          modelId,
          text: 'Hello, world!'
        })

        console.log("Input: 'Hello, world!'")
        console.log('Embedding dimensions:', singleEmbedding.length)
        console.log('First 10 values:', singleEmbedding.slice(0, 10))

        console.log('\n▸ Example 2: Batch Text Embeddings')
        console.log('='.repeat(50))

        const texts = [
          'The quick brown fox jumps over the lazy dog',
          'A fast auburn fox leaps over a sleepy canine',
          'Python is a programming language'
        ]

        const { embedding: batchEmbeddings } = await embed({ modelId, text: texts })

        console.log('Input: Array of', texts.length, 'texts')
        console.log('Output: Array of', batchEmbeddings.length, 'embeddings')

        const [emb1, emb2, emb3] = batchEmbeddings

        if (!emb1 || !emb2 || !emb3) {
          throw new Error('Expected 3 embeddings')
        }

        console.log('Each embedding dimensions:', emb1.length)

        console.log('\n▸ Similarity Analysis')
        console.log('='.repeat(50))

        const similarity1 = cosineSimilarity(emb1, emb2)
        const similarity2 = cosineSimilarity(emb1, emb3)

        console.log('Similarity between texts 1 and 2 (similar meaning):', similarity1.toFixed(4))
        console.log('Similarity between texts 1 and 3 (different topics):', similarity2.toFixed(4))
        console.log('\n▸ Higher values indicate more similar meanings')

        await unloadModel({ modelId, clearStorage: false })
      } catch (error) {
        console.error('✖', error)
        process.exit(1)
      }
      ```
    </WrapCode>
  </Tab>

  <Tab value="python" label="Python">
    <WrapCode>
      ```python file=<rootDir>/packages/sdk-python/examples/embeddings.py title="embeddings.py" lineNumbers
      """Python port of packages/sdk/examples/embed-p2p.ts.

      Single + batch text embeddings and a cosine-similarity comparison.

      `embed` is a generated request-reply method: it takes a validated
      `EmbedRequest` and returns an `EmbedResponse` (there's no flat `embed(text=)`
      convenience wrapper yet). Request models are re-exported from the flat surface,
      so `from tetherto.qvac_sdk import EmbedRequest`.

      RUN: python examples/embeddings.py
      """

      from __future__ import annotations

      import asyncio
      import math
      import sys

      from tetherto.qvac_sdk import Client, EmbedRequest, embed, load_model, unload_model
      from tetherto.qvac_sdk.models import EMBEDDINGGEMMA_300M_Q4_0


      def print_progress(p) -> None:
          """Print model download progress; pass as `on_progress=` to `load_model`."""
          line = (
              f"▸ Downloading {p.percentage:.0f}% "
              f"({p.downloaded / 1e6:.1f}/{p.total / 1e6:.1f} MB)"
          )
          print(line, end="\r" if sys.stderr.isatty() else "\n", file=sys.stderr)
          if p.percentage >= 100:
              print(file=sys.stderr)


      def cosine_similarity(a, b) -> float:
          dot = sum(x * y for x, y in zip(a, b))
          na = math.sqrt(sum(x * x for x in a))
          nb = math.sqrt(sum(y * y for y in b))
          return dot / (na * nb)


      async def embed_text(t, model_id, text):
          response = await embed(
              t,
              EmbedRequest.model_validate(
                  {"type": "embed", "modelId": model_id, "text": text}
              ),
          )
          if not response.success:
              raise RuntimeError(response.error)
          return response.embedding


      async def main() -> int:
          async with Client() as client:
              t = client.transport
              try:
                  model_id = await load_model(
                      t, model_src=EMBEDDINGGEMMA_300M_Q4_0, on_progress=print_progress
                  )

                  print("\n▸ Example 1: Single Text Embedding")
                  print("=" * 50)
                  single = await embed_text(t, model_id, "Hello, world!")
                  print("Input: 'Hello, world!'")
                  print("Embedding dimensions:", len(single))
                  print("First 10 values:", single[:10])

                  print("\n▸ Example 2: Batch Text Embeddings")
                  print("=" * 50)
                  texts = [
                      "The quick brown fox jumps over the lazy dog",
                      "A fast auburn fox leaps over a sleepy canine",
                      "Python is a programming language",
                  ]
                  batch = await embed_text(t, model_id, texts)
                  print("Input: Array of", len(texts), "texts")
                  print("Output: Array of", len(batch), "embeddings")
                  emb1, emb2, emb3 = batch
                  print("Each embedding dimensions:", len(emb1))

                  print("\n▸ Similarity Analysis")
                  print("=" * 50)
                  print(
                      "Similarity between texts 1 and 2 (similar meaning):",
                      f"{cosine_similarity(emb1, emb2):.4f}",
                  )
                  print(
                      "Similarity between texts 1 and 3 (different topics):",
                      f"{cosine_similarity(emb1, emb3):.4f}",
                  )
                  print("\n▸ Higher values indicate more similar meanings")

                  await unload_model(t, model_id)
              except Exception as error:
                  print(f"✖ {error}", file=sys.stderr)
                  return 1
          return 0


      if __name__ == "__main__":
          sys.exit(asyncio.run(main()))
      ```
    </WrapCode>
  </Tab>
</Tabs>

<Callout type="success">
  **Tip:** all examples throughout this documentation are self-contained and runnable. For instructions on how to run them, see the [JS/TS quickstart](/sdk/v0.20/js-ts-sdk#quickstart) or the [Python quickstart](/sdk/v0.20/python-sdk#quickstart).
</Callout>
