commit bdc03a5d592e8dc8d9e907a5e8b36ea803e8e40f Author: estorozhenko Date: Sun Sep 6 13:51:06 2026 +0000 Initial commit: Hermes skill rag-pipeline-docker diff --git a/SKILL.md b/SKILL.md new file mode 100644 index 0000000..377994e --- /dev/null +++ b/SKILL.md @@ -0,0 +1,308 @@ +--- +name: rag-pipeline-docker +title: RAG Pipeline in Docker Compose +category: devops +description: Deploying a RAG pipeline with Docker Compose — Qdrant (vector DB), Redis (job queue), ARQ worker (chunking + embedding), connected to an external Ollama instance for embeddings and LLM inference. Covers env var management, cross-network connectivity, cron-based sync+ingest, and troubleshooting. +triggers: + - qdrant docker + - arq worker + - rag pipeline deploy + - memory-os setup + - vector db docker compose + - ollama + qdrant integration + - wiki ingest pipeline + - obsidian sync qdrant +--- + +# RAG Pipeline in Docker Compose + +## Architecture + +``` +Obsidian vault (host) + ↓ sync_obsidian_to_wiki.py (cron: every 10m) +Wiki path (host) + ↓ ARQ queue → worker container + ↓ chunk_text() + get_embedding() + get_sparse_embedding() +Qdrant (localhost:6333 / collection: knowledge_base) + ↓ dense: nomic-embed-text (768d, Cosine) + ↓ sparse: BM25 (on_disk) +``` + +## Quick Start + +### 1. Docker Compose Stack + +```yaml +services: + redis: + image: redis:7-alpine + restart: unless-stopped + # password via REDIS_PASSWORD env + healthcheck: [CMD-SHELL, "redis-cli ${REDIS_PASSWORD:+-a $REDIS_PASSWORD} ping"] + + qdrant: + image: qdrant/qdrant:v1.17.1 + restart: unless-stopped + ports: ["127.0.0.1:6333:6333"] + volumes: [qdrant_data:/qdrant/storage] + healthcheck: [CMD, sh, -c, "grep -q ':18BD' /proc/net/tcp"] + + worker: + build: ./worker + restart: unless-stopped + depends_on: [qdrant, redis] + # see env section below + volumes: + - wiki_path:/wiki:ro + - hermes_home:/hermes:rw +``` + +### 2. Environment Variables + +**LLM for reflection / reasoning (inside worker):** +```env +OLLAMA_BASE_URL=http://ollama:11434 +OLLAMA_MODEL=qwen3-8b-64k +``` + +**Embedding (inside worker):** +```env +EMBEDDING_API_BASE=http://ollama:11434/v1 +EMBEDDING_MODEL=nomic-embed-text:latest +EMBEDDING_DIMS=768 +EMBEDDING_API_KEY= +``` + +**Redis:** +```env +REDIS_PASSWORD= +REDIS_HOST=redis +REDIS_PORT=6379 +``` + +**Qdrant:** +```env +QDRANT_HOST=qdrant +QDRANT_PORT=6333 +COLLECTION_NAME=knowledge_base +``` + +### 3. Cross-Stack Network + +If Qdrant/Redis/worker are in one compose stack and Ollama is in another, the worker needs access to both networks: + +```yaml +services: + worker: + networks: + - default # memory-os_default — for Redis + Qdrant + - ollama_default # external — for Ollama DNS + +networks: + default: + name: memory-os_default + ollama_default: + external: true +``` + +> **Critical:** `host.docker.internal` does NOT work on Linux (Docker Desktop only). Use `ollama:11434` (via shared network) or `172.17.0.1:11434` (host gateway) instead. + +### 4. Verify Connectivity + +```bash +# DNS resolution +docker exec getent hosts ollama + +# Ollama API +docker exec python3 -c " +import urllib.request, json +req = urllib.request.Request('http://ollama:11434/api/tags') +resp = urllib.request.urlopen(req, timeout=10) +data = json.loads(resp.read()) +print(f'Models: {len(data[\"models\"])}') +" + +# Qdrant collection +curl -s http://127.0.0.1:6333/collections/knowledge_base | python3 -c " +import sys,json; d=json.load(sys.stdin) +print(f'points: {d[\"result\"][\"points_count\"]}') +" +``` + +## Search API (FastAPI) + +Add a search API layer that accepts text queries and returns results from Qdrant: + +### Docker Compose Service + +```yaml +search-api: + build: + context: ../search_api # relative to docker/ directory + dockerfile: Dockerfile + restart: unless-stopped + depends_on: + qdrant: + condition: service_healthy + networks: + - default + - ollama_default + environment: + OLLAMA_URL: http://ollama:11434 + OLLAMA_EMBEDDING_MODEL: nomic-embed-text:latest + QDRANT_URL: http://qdrant:6333 + COLLECTION_NAME: ${COLLECTION_NAME:-knowledge_base} + ports: + - "127.0.0.1:8000:8000" + healthcheck: + test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:8000/health', timeout=5)"] + interval: 15s + timeout: 5s + retries: 5 + start_period: 10s +``` + +### FastAPI App Structure + +``` +search_api/ +├── Dockerfile +├── requirements.txt # fastapi, uvicorn, httpx, pydantic +└── main.py +``` + +### Endpoints + +- `GET /health` — returns `{"status": "ok", "qdrant": true, "ollama": true}` +- `POST /search` — accepts `{"query": "...", "top_k": 5}`, returns `{"query": "...", "results": [...], "total": N}` + +### Flow + +1. Receive text query → POST to Ollama `/api/embeddings` (nomic-embed-text) +2. Use returned dense vector → POST to Qdrant `/collections/{name}/points/search` +3. Return results with score, text, source + +### Verify + +```bash +# Health +curl http://127.0.0.1:8000/health + +# Search +curl -X POST http://127.0.0.1:8000/search \ + -H 'Content-Type: application/json' \ + -d '{"query":"your search text","top_k":3}' +``` + +## Periodic Tasks + +### Sync + Ingest (every 10 min) + +Set up a cron job that runs every 10 minutes: + +1. **Sync script** — copies new/changed `.md` files from Obsidian vault to wiki path, tracking state via JSON file +2. **Ingest script** — detects new/modified files, enqueues them to ARQ worker for chunking + embedding + +```bash +# Manual run +python3 /path/to/scripts/sync_obsidian_to_wiki.py +python3 /path/to/scripts/wiki_continuous_ingest.py +``` + +Via Hermes cronjob (LLM-driven — uses `no_agent: false`): +``` +hermes cron create \ + --name "memory-os sync+ingest" \ + --schedule "every 10m" \ + --prompt "Run: python3 /path/to/sync_obsidian_to_wiki.py" +``` + +### Micro-Reflection Trigger (every 5 min, silent) + +An ARQ worker can have a `process_micro_reflection` function that runs idle-time reflection. To trigger it on a schedule **without LLM overhead**, use a `no_agent: true` watchdog cronjob that runs a script inside the worker container. + +**Pre-requisite:** Mount the scripts directory into the worker container: + +```yaml +services: + worker: + volumes: + - ../scripts:/app/scripts:ro # relative to docker/ directory +``` + +**Script** (`reflection_trigger.py`): checks if the ARQ worker is idle (no pending/executing jobs), respects a per-hour budget, and enqueues `process_micro_reflection` via Redis. + +**Cronjob (no_agent, silent, local):** +``` +hermes cron create \ + --name "memory-os micro-reflection" \ + --schedule "*/5 * * * *" \ + --script "docker exec python3 /app/scripts/reflection_trigger.py" \ + --no-agent +hermes cron update \ + --job-id \ + --deliver local +``` + +Key points: +- `no_agent: true` — no LLM tokens consumed, just runs the script and delivers stdout verbatim +- `deliver: local` — suppresses Telegram/Discord notifications; the job runs silently +- Empty stdout = silent (no message sent), error output = alert delivered +- The script must be on the host filesystem AND mounted into the container via `volumes:` + +## Checking Worker Health + +```bash +# Container status +docker ps --filter name=worker + +# Worker logs +docker logs --tail 50 + +# Check for errors +docker logs 2>&1 | grep -i "error\|traceback\|exception" | head -10 + +# ARQ stats (from worker logs) +docker logs 2>&1 | grep "j_complete\|j_failed" +``` + +## Pitfalls + +### `host.docker.internal` on Linux +`host.docker.internal` is a Docker Desktop feature (macOS/Windows). On Linux, it does not resolve. Use one of: +- Container name on shared network: `http://ollama:11434` +- Host gateway: `http://172.17.0.1:11434` + +### Env vars not propagated to container +Variables defined in `.env` are NOT automatically available inside containers — they must be explicitly listed in `docker-compose.yml` under `services.worker.environment`. `docker compose config` can verify the effective config. + +### Redis password mismatch +If the worker uses `redis.asyncio` or `arq.connections.RedisSettings`, ensure the password matches what's in `redis.conf`. Test with `redis-cli -a $PASSWORD ping`. + +### Network detachment on recreate +When a container is recreated via `docker compose up -d --force-recreate`, it may lose connections to external networks. The fix is to declare the network in `docker-compose.yml` with `external: true` and add it to the service's `networks:` list. + +### Qdrant healthcheck on custom port +The default Qdrant healthcheck greps `/proc/net/tcp` for `:18BD` (port 6333 in hex). If using a non-standard port, update the healthcheck. + +### ARQ worker timeout +The `ollama_chat` function in reflection tasks may timeout if the model is large or generating long responses. Set `ARQ_JOB_TIMEOUT` high enough (e.g., 300s) and ensure `httpx.AsyncClient(timeout=120)` matches. + +### `no_agent` cron script must be on host filesystem +A `no_agent: true` cronjob's `--script` runs on the host, not inside the container. If the script only exists inside the container (e.g., at `/app/scripts/`), the cronjob will fail. Mount the scripts directory into the container AND keep the script accessible on the host, or use `docker exec` to run it inside the container: + +``` +--script "docker exec python3 /app/scripts/script.py" +``` + +### `reflection_trigger.py` paths hardcoded to old project +The `reflection_trigger.py` script was originally written for a different project (`~/.ai-stack/`). The `.env` path and log paths must be updated to match the new project layout before the script works after a copy. Search for `Path.home() / "ai-stack"` or similar hardcoded paths and update them to the new project root. + +### Volume paths in docker-compose are relative to compose file +When adding a `volumes:` mount like `- ../scripts:/app/scripts:ro`, the path is relative to the `docker-compose.yml` file's directory, not the project root. If the compose file is in `docker/`, then `../scripts` resolves to `project/scripts/`. + +## Support Files + +- **`references/memory-os-session.md`** — session-specific details from the Memory OS deployment (env files, state files, error transcripts, search API code) +- **`scripts/test_qdrant_search.py`** — standalone test script: gets embedding from Ollama, searches Qdrant, prints top-5 results \ No newline at end of file diff --git a/references/memory-os-session.md b/references/memory-os-session.md new file mode 100644 index 0000000..074637e --- /dev/null +++ b/references/memory-os-session.md @@ -0,0 +1,108 @@ +# Memory OS Deployment — Session Details + +## Environment + +- Host: Linux, no Docker Desktop +- Ollama: Docker container on `ollama_default` network, port 11434 +- Worker: Docker Compose stack with Qdrant + Redis + ARQ worker +- Obsidian vault: `/opt/hermes/obsidian-vault/` +- Wiki path: `/opt/hermes/vault/wiki/raw/obsidian/` +- State file: `~/.hermes/wiki_ingest_state.json` (tracks ingested files by hash) +- Sync state: `~/.hermes/obsidian_sync_state.json` + +## Docker Compose Path + +`/opt/hermes/memory-os/docker/docker-compose.yml` + +## .env File + +`/opt/hermes/memory-os/docker/.env` — contains: +- `OLLAMA_BASE_URL=http://ollama:11434` +- `OLLAMA_MODEL=qwen3-8b-64k` +- `OLLAMA_EMBEDDING_MODEL=nomic-embed-text:latest` +- `EMBEDDING_API_BASE=http://ollama:11434/v1` +- `EMBEDDING_DIMS=768` +- `REDIS_PASSWORD`, `QDRANT_API_KEY`, `OPENROUTER_API_KEY` + +## Error: Reflection Failing + +**Symptom:** `j_failed=90` on worker, all from `cron:process_reflection` + +**Error:** +``` +httpx.ConnectError: [Errno -2] Name or service not known +``` +The worker was trying `http://host.docker.internal:11434/api/generate` + +**Root cause:** `OLLAMA_BASE_URL` defaulted to `http://host.docker.internal:11434` in `services/llm.py`: +```python +OLLAMA_BASE_URL = os.environ.get("OLLAMA_BASE_URL", "http://host.docker.internal:11434") +``` +This variable was not listed in `docker-compose.yml` under `worker.environment`, so the container fell back to the hardcoded default. + +**Fix:** +1. Added `OLLAMA_BASE_URL: ${OLLAMA_BASE_URL:-http://ollama:11434}` and `OLLAMA_MODEL: ${OLLAMA_MODEL:-qwen3-8b-64k}` to `docker-compose.yml` +2. Added `OLLAMA_MODEL=qwen3-8b-64k` to `.env` +3. Connected worker to `ollama_default` external network in compose +4. `docker compose up -d --force-recreate worker` to rebuild + +## Network Layout + +``` +worker (memory-os_default: 172.24.0.x, ollama_default: 172.18.0.3) + → ollama (ollama_default: 172.18.0.5) via DNS + → qdrant (memory-os_default) via DNS + → redis (memory-os_default) via DNS +search-api (memory-os_default: 172.24.0.x, ollama_default: 172.18.0.x) + → ollama (ollama_default) via DNS + → qdrant (memory-os_default) via DNS +``` + +## Qdrant Collection + +- Name: `knowledge_base` +- Points: 683 +- Dense vector: 768 dims, Cosine distance +- Sparse vector: BM25 (on_disk=True) +- Embedding model: `nomic-embed-text:latest` + +## Search API + +**Path:** `/opt/hermes/memory-os/search_api/main.py` +**Port:** localhost:8000 +**Docker image:** `docker-search-api` (built from `search_api/Dockerfile`) + +**Files:** +``` +search_api/ +├── Dockerfile +├── requirements.txt # fastapi, uvicorn, httpx, pydantic +└── main.py # FastAPI app with POST /search and GET /health +``` + +**Endpoints:** +- `GET /health` → `{"status": "ok", "qdrant": true, "ollama": true}` +- `POST /search` → body: `{"query": "search text", "top_k": 3}` → `{"query": "...", "results": [...], "total": N}` + +## Cron Jobs + +### sync+ingest (LLM-driven, every 10m) +- Name: `memory-os obsidian sync + ingest` +- Schedule: every 10 minutes +- Prompt: Runs sync script + ingest pipeline +- Tools: terminal only + +### micro-reflection trigger (no_agent, every 5m, silent) +- Name: `memory-os micro-reflection trigger` +- Schedule: `*/5 * * * *` +- Script: `docker exec docker-worker-1 python3 /app/scripts/reflection_trigger.py` +- `no_agent: true` — no LLM tokens, just runs the script +- `deliver: local` — silent, no Telegram notifications +- Script path must be mounted into container: `- ../scripts:/app/scripts:ro` in docker-compose.yml + +## Scripts + +- `scripts/test_qdrant_search.py` — standalone test: takes --query, gets embedding, searches Qdrant, prints top-5 +- `scripts/sync_obsidian_to_wiki.py` — copies .md from Obsidian vault to wiki path, tracks state +- `scripts/wiki_continuous_ingest.py` — detects new/changed files, enqueues to ARQ worker +- `scripts/reflection_trigger.py` — checks idle, respects budget, enqueues micro-reflection \ No newline at end of file diff --git a/references/mtproto-alexbers-caddy.md b/references/mtproto-alexbers-caddy.md new file mode 100644 index 0000000..0c39020 --- /dev/null +++ b/references/mtproto-alexbers-caddy.md @@ -0,0 +1,71 @@ +# MTProto Proxy via alexbers/mtprotoproxy + Caddy TLS + +## Отличие от seriyps/mtproto-proxy + +Существующий скилл `mtproto-proxy` описывает образ `seriyps/mtproto-proxy` с Fake TLS и переменными `MTP_*`. Это **другой** образ. `alexbers/mtprotoproxy` использует: + +- **Python-конфиг** (`config.py`) вместо переменных окружения +- **Реальные TLS-сертификаты** (через nginx/Caddy reverse proxy) вместо Fake TLS +- `network_mode: host` вместо bridge + +## Docker Compose (с Caddy) + +```yaml +services: + mtproto: + image: alexbers/mtprotoproxy + restart: always + network_mode: host + volumes: + # Сертификаты от Caddy (LetsEncrypt) + - /opt/caddy/caddy_data/caddy/certificates/acme-v02.api.letsencrypt.org-directory/domain.ru:/certs:ro + # Конфиг + - ./mtproto:/config + command: python3 mtprotoproxy.py /config/config.py +``` + +## config.py + +```python +# MTProto proxy config +PORT = 443 # or whatever port Caddy forwards TLS to +USERS = { + "tg": "ee" + "32-hex-chars-secret" +} +# Optional: stats reporting +# SECRET = 123456 # for stats (unsafe, optional) +``` + +## TLS via Caddy + +Caddy reverse proxy ставится перед MTProto: + +```yaml +labels: + caddy: domain.ru + caddy.reverse_proxy: / "{{upstreams 443}}" + caddy.reverse_proxy.transport: http + caddy.reverse_proxy.transport.tls: "insecure_skip_verify" +``` + +Caddy получает LetsEncrypt сертификаты и пробрасывает HTTPS-трафик на MTProto (который внутри слушает без TLS). + +## Ссылка для подключения + +``` +https://t.me/proxy?server=domain.ru&port=443&secret=eexxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx +``` + +Секрет с префиксом `ee` — Telegram на клиенте сам определяет что это Fake TLS / реальный TLS. + +## Медленное подключение (3+ минуты) — возможные причины + +1. **DNS resolver на сервере** — MTProto прокси использует DNS для проверки Telegram API. Если DNS медленный или блокируется, задержка большая. Лечение: проверить `/etc/resolv.conf`, поставить `1.1.1.1` / `8.8.8.8`. + +2. **Caddy появляется раньше MTProto** — если Caddy стартует быстрее, он выдаёт ошибку вместо прокси. Telegram клиент пытается переподключаться, что добавляет задержку. Лечение: настроить `depends_on` или restart политику. + +3. **TCP keepalive** — MTProto держит соединения. Если между клиентом и сервером есть NAT с таймаутом меньше 3 минут, соединение разрывается и восстанавливается. Это может выглядеть как "3 минуты подключается". + +4. **Медленная загрузка сертификатов** — если сертификаты лежат на WebDAV/Yandex Disk, mount может тормозить. Проверить `mount` и права доступа. + +5. **Проверка:** внутри контейнера `docker exec mtproto cat /config/config.py`, снаружи `ss -tlnp | grep mtproto`. \ No newline at end of file diff --git a/references/qdrant-versions-multicollection.md b/references/qdrant-versions-multicollection.md new file mode 100644 index 0000000..62c09b6 --- /dev/null +++ b/references/qdrant-versions-multicollection.md @@ -0,0 +1,81 @@ +# Qdrant: version matching + multiple collections (verified 2026-09-04) + +## qdrant-client must match the server minor version + +Symptom chain with client 1.19.0 vs server 1.17.1 (`qdrant/qdrant:v1.17.1`): +- `Qdrant client version 1.19.0 is incompatible with server version 1.17.1` warning, +- `create_collection` with a bare `models.VectorParams(size=1024, ...)` silently + creates an ANONYMOUS vector (name `""`), NOT `dense`, +- the subsequent `upsert` fails: `400 ... Not existing vector name error: dense`. + +Fix — pin the client to the server minor: +```bash +pip install "qdrant-client==1.17.1" # match qdrant/qdrant:v1.17.1 +``` + +Always pass NAMED vectors so the config works regardless of client version: +```python +from qdrant_client import QdrantClient, models +c = QdrantClient("http://localhost:6333") +c.create_collection( + collection_name=NAME, + vectors_config={ + "dense": models.VectorParams(size=1024, distance=models.Distance.COSINE), + }, + sparse_vectors_config={ + "sparse": models.SparseVectorParams(index=models.SparseIndexParams(on_disk=True)), + }, +) +``` + +## Query API in qdrant-client 1.17 + +- `client.search(...)` does NOT exist in 1.17. +- `query_points(..., query_vector=...)` → `AssertionError: Unknown arguments: ['query_vector']`. +- Working call: +```python +res = c.query_points( + collection_name=NAME, + query=, # list[float] from Ollama /api/embeddings + using="dense", # named-vector selector + limit=5, + with_payload=True, +) +for pt in res.points: + print(pt.score, pt.payload.get("text")) +``` + +## One collection per project/domain (multi-collection design) + +For a distinct document set (batch of PDF protocol/files), create a SEPARATE +collection with the SAME schema (`dense` 1024d COSINE + sparse `sparse` BM25) and the +SAME embedder (bge-m3). Keep search query embeddings compatible by using the same +embedder for all collections. +Benefits: independent re-index, per-domain context search, no pollution of the +general KB. Example: `skc_vinny_gorod` alongside `knowledge_base`. + +## Do NOT retarget context_enhancer to a second collection + +`context_enhancer.py` binds `COLLECTION = os.environ.get("QDRANT_COLLECTION", +"knowledge_base")` at IMPORT time (module level). Swapping the env var at runtime +does NOT retarget it — the module-level constant is already fixed. + +To search a second collection, write a STANDALONE REST search: +1. `POST http://ollama:11434/api/embeddings` `{"model": "bge-m3", "prompt": text}` → `embedding` (1024d). +2. `POST http://qdrant:6333/collections//points/query` with `{"vector": emb, "limit": N, "with_payload": true, "using": "dense"}`. +3. Read `result.points[*].payload` + `.score`. + +This same pattern is the foundation for a per-project context-injector hook +(e.g. NetBox project context) — hit the secondary collection directly, don't go +through the shared KB search. + +## Scanned PDFs: pymupdf returns empty text (no text layer) + +`page.get_text("text")` returns `""` for image-only pages — verified on a real +51-page government PDF (0 text on every page). It is a genuine scan, not a glitch. +Mark scanned pages `[SCANNED_PAGE]` and route to OCR (marker-pdf / vision); +do NOT report "no content" or fabricate text. pymupdf's built-in Type1 fonts +(times-roman, helv, cour, tiro) do NOT contain the Cyrillic glyph map — inserting +Cyrillic with them renders as dots and re-extracts as dots. Real PDFs (generated +from Word/CAD) embed proper fonts and extract Cyrillic fine; the byte test above is +only a pymupdf-font artifact, not a real-PDF problem. \ No newline at end of file diff --git a/references/webui-session-cache-reset.md b/references/webui-session-cache-reset.md new file mode 100644 index 0000000..ed8f557 --- /dev/null +++ b/references/webui-session-cache-reset.md @@ -0,0 +1,28 @@ +# WebUI Session Cache Reset (Post-Agent-Update) + +## Симптомы + +После обновления Hermes agent (особенно 0.19.x → 0.20.x+): + +- WebUI бесконечно показывает "Loading conversation..." +- Создание нового чата не помогает +- Даже ответ в существующем чате не рендерится +- Формат сессий на диске изменился, старый фронтенд не может его отрендерить + +## Фикс + +```bash +# 1. Сбросить кэш сессий внутри контейнера WebUI +docker exec hermes-webui sh -c 'rm -rf /home/hermeswebui/.hermes/webui/sessions/*' + +# 2. Перезапустить WebUI +cd /opt/hermes/docker && docker compose restart hermes-webui + +# 3. Открыть в браузере приватное/инкогнито окно (чтобы не было клиентского кэша) +``` + +## Почему это происходит + +Hermes WebUI — отдельный контейнер со своим образом. Когда агент обновляется (через `!hermes update`), сессии на диске могут изменить формат (новые поля, другая структура context_messages). WebUI-образ не обновляется автоматически — он остаётся старым и не умеет парсить новые сессии. + +Радикальное решение: сбросить кэш старых сессий, WebUI начнёт с чистого листа. \ No newline at end of file diff --git a/scripts/test_qdrant_search.py b/scripts/test_qdrant_search.py new file mode 100644 index 0000000..b034d9b --- /dev/null +++ b/scripts/test_qdrant_search.py @@ -0,0 +1,60 @@ +#!/usr/bin/env python3 +""" +test_qdrant_search.py — Тестовый скрипт поиска по Qdrant. +Берёт текст, получает эмбеддинг через Ollama, ищет в Qdrant. + +Использование: python3 scripts/test_qdrant_search.py --query "текст" +""" +import argparse +import json +import sys +import urllib.request +from pathlib import Path + +OLLAMA_URL = "http://localhost:11434/api/embeddings" +EMBEDDING_MODEL = "nomic-embed-text:latest" +QDRANT_SEARCH_URL = "http://localhost:6333/collections/knowledge_base/points/search" +TOP_K = 5 + + +def get_embedding(text: str) -> list[float]: + payload = json.dumps({"model": EMBEDDING_MODEL, "prompt": text}).encode() + req = urllib.request.Request(OLLAMA_URL, data=payload, + headers={"Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=30) as resp: + data = json.loads(resp.read()) + return data.get("embedding") + + +def search_qdrant(vector: list[float], top_k: int = TOP_K) -> list[dict]: + payload = json.dumps({ + "vector": {"name": "dense", "vector": vector}, + "limit": top_k, + "with_payload": True, + "with_vector": False, + }).encode() + req = urllib.request.Request(QDRANT_SEARCH_URL, data=payload, + headers={"Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=15) as resp: + data = json.loads(resp.read()) + return data.get("result", []) + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--query", "-q", required=True) + args = parser.parse_args() + + vector = get_embedding(args.query) + print(f"Embedding: {len(vector)} dims") + + results = search_qdrant(vector) + print(f"Results: {len(results)}") + for i, r in enumerate(results, 1): + payload = r.get("payload", {}) + text = (payload.get("text") or payload.get("content", ""))[:200] + print(f" #{i} score={r['score']:.4f} | {payload.get('source','?')} | {text}...") + + +if __name__ == "__main__": + main() \ No newline at end of file