mirror of
https://gitverse.ru/kpa39l/rag-pipeline-docker.git
synced 2026-09-29 09:15:11 +00:00
Initial commit: Hermes skill rag-pipeline-docker
This commit is contained in:
@@ -0,0 +1,308 @@
|
|||||||
|
---
|
||||||
|
name: rag-pipeline-docker
|
||||||
|
title: RAG Pipeline in Docker Compose
|
||||||
|
category: devops
|
||||||
|
description: Deploying a RAG pipeline with Docker Compose — Qdrant (vector DB), Redis (job queue), ARQ worker (chunking + embedding), connected to an external Ollama instance for embeddings and LLM inference. Covers env var management, cross-network connectivity, cron-based sync+ingest, and troubleshooting.
|
||||||
|
triggers:
|
||||||
|
- qdrant docker
|
||||||
|
- arq worker
|
||||||
|
- rag pipeline deploy
|
||||||
|
- memory-os setup
|
||||||
|
- vector db docker compose
|
||||||
|
- ollama + qdrant integration
|
||||||
|
- wiki ingest pipeline
|
||||||
|
- obsidian sync qdrant
|
||||||
|
---
|
||||||
|
|
||||||
|
# RAG Pipeline in Docker Compose
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
```
|
||||||
|
Obsidian vault (host)
|
||||||
|
↓ sync_obsidian_to_wiki.py (cron: every 10m)
|
||||||
|
Wiki path (host)
|
||||||
|
↓ ARQ queue → worker container
|
||||||
|
↓ chunk_text() + get_embedding() + get_sparse_embedding()
|
||||||
|
Qdrant (localhost:6333 / collection: knowledge_base)
|
||||||
|
↓ dense: nomic-embed-text (768d, Cosine)
|
||||||
|
↓ sparse: BM25 (on_disk)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Quick Start
|
||||||
|
|
||||||
|
### 1. Docker Compose Stack
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
redis:
|
||||||
|
image: redis:7-alpine
|
||||||
|
restart: unless-stopped
|
||||||
|
# password via REDIS_PASSWORD env
|
||||||
|
healthcheck: [CMD-SHELL, "redis-cli ${REDIS_PASSWORD:+-a $REDIS_PASSWORD} ping"]
|
||||||
|
|
||||||
|
qdrant:
|
||||||
|
image: qdrant/qdrant:v1.17.1
|
||||||
|
restart: unless-stopped
|
||||||
|
ports: ["127.0.0.1:6333:6333"]
|
||||||
|
volumes: [qdrant_data:/qdrant/storage]
|
||||||
|
healthcheck: [CMD, sh, -c, "grep -q ':18BD' /proc/net/tcp"]
|
||||||
|
|
||||||
|
worker:
|
||||||
|
build: ./worker
|
||||||
|
restart: unless-stopped
|
||||||
|
depends_on: [qdrant, redis]
|
||||||
|
# see env section below
|
||||||
|
volumes:
|
||||||
|
- wiki_path:/wiki:ro
|
||||||
|
- hermes_home:/hermes:rw
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2. Environment Variables
|
||||||
|
|
||||||
|
**LLM for reflection / reasoning (inside worker):**
|
||||||
|
```env
|
||||||
|
OLLAMA_BASE_URL=http://ollama:11434
|
||||||
|
OLLAMA_MODEL=qwen3-8b-64k
|
||||||
|
```
|
||||||
|
|
||||||
|
**Embedding (inside worker):**
|
||||||
|
```env
|
||||||
|
EMBEDDING_API_BASE=http://ollama:11434/v1
|
||||||
|
EMBEDDING_MODEL=nomic-embed-text:latest
|
||||||
|
EMBEDDING_DIMS=768
|
||||||
|
EMBEDDING_API_KEY=
|
||||||
|
```
|
||||||
|
|
||||||
|
**Redis:**
|
||||||
|
```env
|
||||||
|
REDIS_PASSWORD=<your-password>
|
||||||
|
REDIS_HOST=redis
|
||||||
|
REDIS_PORT=6379
|
||||||
|
```
|
||||||
|
|
||||||
|
**Qdrant:**
|
||||||
|
```env
|
||||||
|
QDRANT_HOST=qdrant
|
||||||
|
QDRANT_PORT=6333
|
||||||
|
COLLECTION_NAME=knowledge_base
|
||||||
|
```
|
||||||
|
|
||||||
|
### 3. Cross-Stack Network
|
||||||
|
|
||||||
|
If Qdrant/Redis/worker are in one compose stack and Ollama is in another, the worker needs access to both networks:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
worker:
|
||||||
|
networks:
|
||||||
|
- default # memory-os_default — for Redis + Qdrant
|
||||||
|
- ollama_default # external — for Ollama DNS
|
||||||
|
|
||||||
|
networks:
|
||||||
|
default:
|
||||||
|
name: memory-os_default
|
||||||
|
ollama_default:
|
||||||
|
external: true
|
||||||
|
```
|
||||||
|
|
||||||
|
> **Critical:** `host.docker.internal` does NOT work on Linux (Docker Desktop only). Use `ollama:11434` (via shared network) or `172.17.0.1:11434` (host gateway) instead.
|
||||||
|
|
||||||
|
### 4. Verify Connectivity
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# DNS resolution
|
||||||
|
docker exec <worker> getent hosts ollama
|
||||||
|
|
||||||
|
# Ollama API
|
||||||
|
docker exec <worker> python3 -c "
|
||||||
|
import urllib.request, json
|
||||||
|
req = urllib.request.Request('http://ollama:11434/api/tags')
|
||||||
|
resp = urllib.request.urlopen(req, timeout=10)
|
||||||
|
data = json.loads(resp.read())
|
||||||
|
print(f'Models: {len(data[\"models\"])}')
|
||||||
|
"
|
||||||
|
|
||||||
|
# Qdrant collection
|
||||||
|
curl -s http://127.0.0.1:6333/collections/knowledge_base | python3 -c "
|
||||||
|
import sys,json; d=json.load(sys.stdin)
|
||||||
|
print(f'points: {d[\"result\"][\"points_count\"]}')
|
||||||
|
"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Search API (FastAPI)
|
||||||
|
|
||||||
|
Add a search API layer that accepts text queries and returns results from Qdrant:
|
||||||
|
|
||||||
|
### Docker Compose Service
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
search-api:
|
||||||
|
build:
|
||||||
|
context: ../search_api # relative to docker/ directory
|
||||||
|
dockerfile: Dockerfile
|
||||||
|
restart: unless-stopped
|
||||||
|
depends_on:
|
||||||
|
qdrant:
|
||||||
|
condition: service_healthy
|
||||||
|
networks:
|
||||||
|
- default
|
||||||
|
- ollama_default
|
||||||
|
environment:
|
||||||
|
OLLAMA_URL: http://ollama:11434
|
||||||
|
OLLAMA_EMBEDDING_MODEL: nomic-embed-text:latest
|
||||||
|
QDRANT_URL: http://qdrant:6333
|
||||||
|
COLLECTION_NAME: ${COLLECTION_NAME:-knowledge_base}
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8000:8000"
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD", "python3", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:8000/health', timeout=5)"]
|
||||||
|
interval: 15s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 5
|
||||||
|
start_period: 10s
|
||||||
|
```
|
||||||
|
|
||||||
|
### FastAPI App Structure
|
||||||
|
|
||||||
|
```
|
||||||
|
search_api/
|
||||||
|
├── Dockerfile
|
||||||
|
├── requirements.txt # fastapi, uvicorn, httpx, pydantic
|
||||||
|
└── main.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### Endpoints
|
||||||
|
|
||||||
|
- `GET /health` — returns `{"status": "ok", "qdrant": true, "ollama": true}`
|
||||||
|
- `POST /search` — accepts `{"query": "...", "top_k": 5}`, returns `{"query": "...", "results": [...], "total": N}`
|
||||||
|
|
||||||
|
### Flow
|
||||||
|
|
||||||
|
1. Receive text query → POST to Ollama `/api/embeddings` (nomic-embed-text)
|
||||||
|
2. Use returned dense vector → POST to Qdrant `/collections/{name}/points/search`
|
||||||
|
3. Return results with score, text, source
|
||||||
|
|
||||||
|
### Verify
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Health
|
||||||
|
curl http://127.0.0.1:8000/health
|
||||||
|
|
||||||
|
# Search
|
||||||
|
curl -X POST http://127.0.0.1:8000/search \
|
||||||
|
-H 'Content-Type: application/json' \
|
||||||
|
-d '{"query":"your search text","top_k":3}'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Periodic Tasks
|
||||||
|
|
||||||
|
### Sync + Ingest (every 10 min)
|
||||||
|
|
||||||
|
Set up a cron job that runs every 10 minutes:
|
||||||
|
|
||||||
|
1. **Sync script** — copies new/changed `.md` files from Obsidian vault to wiki path, tracking state via JSON file
|
||||||
|
2. **Ingest script** — detects new/modified files, enqueues them to ARQ worker for chunking + embedding
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Manual run
|
||||||
|
python3 /path/to/scripts/sync_obsidian_to_wiki.py
|
||||||
|
python3 /path/to/scripts/wiki_continuous_ingest.py
|
||||||
|
```
|
||||||
|
|
||||||
|
Via Hermes cronjob (LLM-driven — uses `no_agent: false`):
|
||||||
|
```
|
||||||
|
hermes cron create \
|
||||||
|
--name "memory-os sync+ingest" \
|
||||||
|
--schedule "every 10m" \
|
||||||
|
--prompt "Run: python3 /path/to/sync_obsidian_to_wiki.py"
|
||||||
|
```
|
||||||
|
|
||||||
|
### Micro-Reflection Trigger (every 5 min, silent)
|
||||||
|
|
||||||
|
An ARQ worker can have a `process_micro_reflection` function that runs idle-time reflection. To trigger it on a schedule **without LLM overhead**, use a `no_agent: true` watchdog cronjob that runs a script inside the worker container.
|
||||||
|
|
||||||
|
**Pre-requisite:** Mount the scripts directory into the worker container:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
worker:
|
||||||
|
volumes:
|
||||||
|
- ../scripts:/app/scripts:ro # relative to docker/ directory
|
||||||
|
```
|
||||||
|
|
||||||
|
**Script** (`reflection_trigger.py`): checks if the ARQ worker is idle (no pending/executing jobs), respects a per-hour budget, and enqueues `process_micro_reflection` via Redis.
|
||||||
|
|
||||||
|
**Cronjob (no_agent, silent, local):**
|
||||||
|
```
|
||||||
|
hermes cron create \
|
||||||
|
--name "memory-os micro-reflection" \
|
||||||
|
--schedule "*/5 * * * *" \
|
||||||
|
--script "docker exec <worker> python3 /app/scripts/reflection_trigger.py" \
|
||||||
|
--no-agent
|
||||||
|
hermes cron update \
|
||||||
|
--job-id <id> \
|
||||||
|
--deliver local
|
||||||
|
```
|
||||||
|
|
||||||
|
Key points:
|
||||||
|
- `no_agent: true` — no LLM tokens consumed, just runs the script and delivers stdout verbatim
|
||||||
|
- `deliver: local` — suppresses Telegram/Discord notifications; the job runs silently
|
||||||
|
- Empty stdout = silent (no message sent), error output = alert delivered
|
||||||
|
- The script must be on the host filesystem AND mounted into the container via `volumes:`
|
||||||
|
|
||||||
|
## Checking Worker Health
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Container status
|
||||||
|
docker ps --filter name=worker
|
||||||
|
|
||||||
|
# Worker logs
|
||||||
|
docker logs <worker> --tail 50
|
||||||
|
|
||||||
|
# Check for errors
|
||||||
|
docker logs <worker> 2>&1 | grep -i "error\|traceback\|exception" | head -10
|
||||||
|
|
||||||
|
# ARQ stats (from worker logs)
|
||||||
|
docker logs <worker> 2>&1 | grep "j_complete\|j_failed"
|
||||||
|
```
|
||||||
|
|
||||||
|
## Pitfalls
|
||||||
|
|
||||||
|
### `host.docker.internal` on Linux
|
||||||
|
`host.docker.internal` is a Docker Desktop feature (macOS/Windows). On Linux, it does not resolve. Use one of:
|
||||||
|
- Container name on shared network: `http://ollama:11434`
|
||||||
|
- Host gateway: `http://172.17.0.1:11434`
|
||||||
|
|
||||||
|
### Env vars not propagated to container
|
||||||
|
Variables defined in `.env` are NOT automatically available inside containers — they must be explicitly listed in `docker-compose.yml` under `services.worker.environment`. `docker compose config` can verify the effective config.
|
||||||
|
|
||||||
|
### Redis password mismatch
|
||||||
|
If the worker uses `redis.asyncio` or `arq.connections.RedisSettings`, ensure the password matches what's in `redis.conf`. Test with `redis-cli -a $PASSWORD ping`.
|
||||||
|
|
||||||
|
### Network detachment on recreate
|
||||||
|
When a container is recreated via `docker compose up -d --force-recreate`, it may lose connections to external networks. The fix is to declare the network in `docker-compose.yml` with `external: true` and add it to the service's `networks:` list.
|
||||||
|
|
||||||
|
### Qdrant healthcheck on custom port
|
||||||
|
The default Qdrant healthcheck greps `/proc/net/tcp` for `:18BD` (port 6333 in hex). If using a non-standard port, update the healthcheck.
|
||||||
|
|
||||||
|
### ARQ worker timeout
|
||||||
|
The `ollama_chat` function in reflection tasks may timeout if the model is large or generating long responses. Set `ARQ_JOB_TIMEOUT` high enough (e.g., 300s) and ensure `httpx.AsyncClient(timeout=120)` matches.
|
||||||
|
|
||||||
|
### `no_agent` cron script must be on host filesystem
|
||||||
|
A `no_agent: true` cronjob's `--script` runs on the host, not inside the container. If the script only exists inside the container (e.g., at `/app/scripts/`), the cronjob will fail. Mount the scripts directory into the container AND keep the script accessible on the host, or use `docker exec` to run it inside the container:
|
||||||
|
|
||||||
|
```
|
||||||
|
--script "docker exec <container> python3 /app/scripts/script.py"
|
||||||
|
```
|
||||||
|
|
||||||
|
### `reflection_trigger.py` paths hardcoded to old project
|
||||||
|
The `reflection_trigger.py` script was originally written for a different project (`~/.ai-stack/`). The `.env` path and log paths must be updated to match the new project layout before the script works after a copy. Search for `Path.home() / "ai-stack"` or similar hardcoded paths and update them to the new project root.
|
||||||
|
|
||||||
|
### Volume paths in docker-compose are relative to compose file
|
||||||
|
When adding a `volumes:` mount like `- ../scripts:/app/scripts:ro`, the path is relative to the `docker-compose.yml` file's directory, not the project root. If the compose file is in `docker/`, then `../scripts` resolves to `project/scripts/`.
|
||||||
|
|
||||||
|
## Support Files
|
||||||
|
|
||||||
|
- **`references/memory-os-session.md`** — session-specific details from the Memory OS deployment (env files, state files, error transcripts, search API code)
|
||||||
|
- **`scripts/test_qdrant_search.py`** — standalone test script: gets embedding from Ollama, searches Qdrant, prints top-5 results
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
# Memory OS Deployment — Session Details
|
||||||
|
|
||||||
|
## Environment
|
||||||
|
|
||||||
|
- Host: Linux, no Docker Desktop
|
||||||
|
- Ollama: Docker container on `ollama_default` network, port 11434
|
||||||
|
- Worker: Docker Compose stack with Qdrant + Redis + ARQ worker
|
||||||
|
- Obsidian vault: `/opt/hermes/obsidian-vault/`
|
||||||
|
- Wiki path: `/opt/hermes/vault/wiki/raw/obsidian/`
|
||||||
|
- State file: `~/.hermes/wiki_ingest_state.json` (tracks ingested files by hash)
|
||||||
|
- Sync state: `~/.hermes/obsidian_sync_state.json`
|
||||||
|
|
||||||
|
## Docker Compose Path
|
||||||
|
|
||||||
|
`/opt/hermes/memory-os/docker/docker-compose.yml`
|
||||||
|
|
||||||
|
## .env File
|
||||||
|
|
||||||
|
`/opt/hermes/memory-os/docker/.env` — contains:
|
||||||
|
- `OLLAMA_BASE_URL=http://ollama:11434`
|
||||||
|
- `OLLAMA_MODEL=qwen3-8b-64k`
|
||||||
|
- `OLLAMA_EMBEDDING_MODEL=nomic-embed-text:latest`
|
||||||
|
- `EMBEDDING_API_BASE=http://ollama:11434/v1`
|
||||||
|
- `EMBEDDING_DIMS=768`
|
||||||
|
- `REDIS_PASSWORD`, `QDRANT_API_KEY`, `OPENROUTER_API_KEY`
|
||||||
|
|
||||||
|
## Error: Reflection Failing
|
||||||
|
|
||||||
|
**Symptom:** `j_failed=90` on worker, all from `cron:process_reflection`
|
||||||
|
|
||||||
|
**Error:**
|
||||||
|
```
|
||||||
|
httpx.ConnectError: [Errno -2] Name or service not known
|
||||||
|
```
|
||||||
|
The worker was trying `http://host.docker.internal:11434/api/generate`
|
||||||
|
|
||||||
|
**Root cause:** `OLLAMA_BASE_URL` defaulted to `http://host.docker.internal:11434` in `services/llm.py`:
|
||||||
|
```python
|
||||||
|
OLLAMA_BASE_URL = os.environ.get("OLLAMA_BASE_URL", "http://host.docker.internal:11434")
|
||||||
|
```
|
||||||
|
This variable was not listed in `docker-compose.yml` under `worker.environment`, so the container fell back to the hardcoded default.
|
||||||
|
|
||||||
|
**Fix:**
|
||||||
|
1. Added `OLLAMA_BASE_URL: ${OLLAMA_BASE_URL:-http://ollama:11434}` and `OLLAMA_MODEL: ${OLLAMA_MODEL:-qwen3-8b-64k}` to `docker-compose.yml`
|
||||||
|
2. Added `OLLAMA_MODEL=qwen3-8b-64k` to `.env`
|
||||||
|
3. Connected worker to `ollama_default` external network in compose
|
||||||
|
4. `docker compose up -d --force-recreate worker` to rebuild
|
||||||
|
|
||||||
|
## Network Layout
|
||||||
|
|
||||||
|
```
|
||||||
|
worker (memory-os_default: 172.24.0.x, ollama_default: 172.18.0.3)
|
||||||
|
→ ollama (ollama_default: 172.18.0.5) via DNS
|
||||||
|
→ qdrant (memory-os_default) via DNS
|
||||||
|
→ redis (memory-os_default) via DNS
|
||||||
|
search-api (memory-os_default: 172.24.0.x, ollama_default: 172.18.0.x)
|
||||||
|
→ ollama (ollama_default) via DNS
|
||||||
|
→ qdrant (memory-os_default) via DNS
|
||||||
|
```
|
||||||
|
|
||||||
|
## Qdrant Collection
|
||||||
|
|
||||||
|
- Name: `knowledge_base`
|
||||||
|
- Points: 683
|
||||||
|
- Dense vector: 768 dims, Cosine distance
|
||||||
|
- Sparse vector: BM25 (on_disk=True)
|
||||||
|
- Embedding model: `nomic-embed-text:latest`
|
||||||
|
|
||||||
|
## Search API
|
||||||
|
|
||||||
|
**Path:** `/opt/hermes/memory-os/search_api/main.py`
|
||||||
|
**Port:** localhost:8000
|
||||||
|
**Docker image:** `docker-search-api` (built from `search_api/Dockerfile`)
|
||||||
|
|
||||||
|
**Files:**
|
||||||
|
```
|
||||||
|
search_api/
|
||||||
|
├── Dockerfile
|
||||||
|
├── requirements.txt # fastapi, uvicorn, httpx, pydantic
|
||||||
|
└── main.py # FastAPI app with POST /search and GET /health
|
||||||
|
```
|
||||||
|
|
||||||
|
**Endpoints:**
|
||||||
|
- `GET /health` → `{"status": "ok", "qdrant": true, "ollama": true}`
|
||||||
|
- `POST /search` → body: `{"query": "search text", "top_k": 3}` → `{"query": "...", "results": [...], "total": N}`
|
||||||
|
|
||||||
|
## Cron Jobs
|
||||||
|
|
||||||
|
### sync+ingest (LLM-driven, every 10m)
|
||||||
|
- Name: `memory-os obsidian sync + ingest`
|
||||||
|
- Schedule: every 10 minutes
|
||||||
|
- Prompt: Runs sync script + ingest pipeline
|
||||||
|
- Tools: terminal only
|
||||||
|
|
||||||
|
### micro-reflection trigger (no_agent, every 5m, silent)
|
||||||
|
- Name: `memory-os micro-reflection trigger`
|
||||||
|
- Schedule: `*/5 * * * *`
|
||||||
|
- Script: `docker exec docker-worker-1 python3 /app/scripts/reflection_trigger.py`
|
||||||
|
- `no_agent: true` — no LLM tokens, just runs the script
|
||||||
|
- `deliver: local` — silent, no Telegram notifications
|
||||||
|
- Script path must be mounted into container: `- ../scripts:/app/scripts:ro` in docker-compose.yml
|
||||||
|
|
||||||
|
## Scripts
|
||||||
|
|
||||||
|
- `scripts/test_qdrant_search.py` — standalone test: takes --query, gets embedding, searches Qdrant, prints top-5
|
||||||
|
- `scripts/sync_obsidian_to_wiki.py` — copies .md from Obsidian vault to wiki path, tracks state
|
||||||
|
- `scripts/wiki_continuous_ingest.py` — detects new/changed files, enqueues to ARQ worker
|
||||||
|
- `scripts/reflection_trigger.py` — checks idle, respects budget, enqueues micro-reflection
|
||||||
@@ -0,0 +1,71 @@
|
|||||||
|
# MTProto Proxy via alexbers/mtprotoproxy + Caddy TLS
|
||||||
|
|
||||||
|
## Отличие от seriyps/mtproto-proxy
|
||||||
|
|
||||||
|
Существующий скилл `mtproto-proxy` описывает образ `seriyps/mtproto-proxy` с Fake TLS и переменными `MTP_*`. Это **другой** образ. `alexbers/mtprotoproxy` использует:
|
||||||
|
|
||||||
|
- **Python-конфиг** (`config.py`) вместо переменных окружения
|
||||||
|
- **Реальные TLS-сертификаты** (через nginx/Caddy reverse proxy) вместо Fake TLS
|
||||||
|
- `network_mode: host` вместо bridge
|
||||||
|
|
||||||
|
## Docker Compose (с Caddy)
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
services:
|
||||||
|
mtproto:
|
||||||
|
image: alexbers/mtprotoproxy
|
||||||
|
restart: always
|
||||||
|
network_mode: host
|
||||||
|
volumes:
|
||||||
|
# Сертификаты от Caddy (LetsEncrypt)
|
||||||
|
- /opt/caddy/caddy_data/caddy/certificates/acme-v02.api.letsencrypt.org-directory/domain.ru:/certs:ro
|
||||||
|
# Конфиг
|
||||||
|
- ./mtproto:/config
|
||||||
|
command: python3 mtprotoproxy.py /config/config.py
|
||||||
|
```
|
||||||
|
|
||||||
|
## config.py
|
||||||
|
|
||||||
|
```python
|
||||||
|
# MTProto proxy config
|
||||||
|
PORT = 443 # or whatever port Caddy forwards TLS to
|
||||||
|
USERS = {
|
||||||
|
"tg": "ee" + "32-hex-chars-secret"
|
||||||
|
}
|
||||||
|
# Optional: stats reporting
|
||||||
|
# SECRET = 123456 # for stats (unsafe, optional)
|
||||||
|
```
|
||||||
|
|
||||||
|
## TLS via Caddy
|
||||||
|
|
||||||
|
Caddy reverse proxy ставится перед MTProto:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
labels:
|
||||||
|
caddy: domain.ru
|
||||||
|
caddy.reverse_proxy: / "{{upstreams 443}}"
|
||||||
|
caddy.reverse_proxy.transport: http
|
||||||
|
caddy.reverse_proxy.transport.tls: "insecure_skip_verify"
|
||||||
|
```
|
||||||
|
|
||||||
|
Caddy получает LetsEncrypt сертификаты и пробрасывает HTTPS-трафик на MTProto (который внутри слушает без TLS).
|
||||||
|
|
||||||
|
## Ссылка для подключения
|
||||||
|
|
||||||
|
```
|
||||||
|
https://t.me/proxy?server=domain.ru&port=443&secret=eexxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx
|
||||||
|
```
|
||||||
|
|
||||||
|
Секрет с префиксом `ee` — Telegram на клиенте сам определяет что это Fake TLS / реальный TLS.
|
||||||
|
|
||||||
|
## Медленное подключение (3+ минуты) — возможные причины
|
||||||
|
|
||||||
|
1. **DNS resolver на сервере** — MTProto прокси использует DNS для проверки Telegram API. Если DNS медленный или блокируется, задержка большая. Лечение: проверить `/etc/resolv.conf`, поставить `1.1.1.1` / `8.8.8.8`.
|
||||||
|
|
||||||
|
2. **Caddy появляется раньше MTProto** — если Caddy стартует быстрее, он выдаёт ошибку вместо прокси. Telegram клиент пытается переподключаться, что добавляет задержку. Лечение: настроить `depends_on` или restart политику.
|
||||||
|
|
||||||
|
3. **TCP keepalive** — MTProto держит соединения. Если между клиентом и сервером есть NAT с таймаутом меньше 3 минут, соединение разрывается и восстанавливается. Это может выглядеть как "3 минуты подключается".
|
||||||
|
|
||||||
|
4. **Медленная загрузка сертификатов** — если сертификаты лежат на WebDAV/Yandex Disk, mount может тормозить. Проверить `mount` и права доступа.
|
||||||
|
|
||||||
|
5. **Проверка:** внутри контейнера `docker exec mtproto cat /config/config.py`, снаружи `ss -tlnp | grep mtproto`.
|
||||||
@@ -0,0 +1,81 @@
|
|||||||
|
# Qdrant: version matching + multiple collections (verified 2026-09-04)
|
||||||
|
|
||||||
|
## qdrant-client must match the server minor version
|
||||||
|
|
||||||
|
Symptom chain with client 1.19.0 vs server 1.17.1 (`qdrant/qdrant:v1.17.1`):
|
||||||
|
- `Qdrant client version 1.19.0 is incompatible with server version 1.17.1` warning,
|
||||||
|
- `create_collection` with a bare `models.VectorParams(size=1024, ...)` silently
|
||||||
|
creates an ANONYMOUS vector (name `""`), NOT `dense`,
|
||||||
|
- the subsequent `upsert` fails: `400 ... Not existing vector name error: dense`.
|
||||||
|
|
||||||
|
Fix — pin the client to the server minor:
|
||||||
|
```bash
|
||||||
|
pip install "qdrant-client==1.17.1" # match qdrant/qdrant:v1.17.1
|
||||||
|
```
|
||||||
|
|
||||||
|
Always pass NAMED vectors so the config works regardless of client version:
|
||||||
|
```python
|
||||||
|
from qdrant_client import QdrantClient, models
|
||||||
|
c = QdrantClient("http://localhost:6333")
|
||||||
|
c.create_collection(
|
||||||
|
collection_name=NAME,
|
||||||
|
vectors_config={
|
||||||
|
"dense": models.VectorParams(size=1024, distance=models.Distance.COSINE),
|
||||||
|
},
|
||||||
|
sparse_vectors_config={
|
||||||
|
"sparse": models.SparseVectorParams(index=models.SparseIndexParams(on_disk=True)),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Query API in qdrant-client 1.17
|
||||||
|
|
||||||
|
- `client.search(...)` does NOT exist in 1.17.
|
||||||
|
- `query_points(..., query_vector=...)` → `AssertionError: Unknown arguments: ['query_vector']`.
|
||||||
|
- Working call:
|
||||||
|
```python
|
||||||
|
res = c.query_points(
|
||||||
|
collection_name=NAME,
|
||||||
|
query=<dense_embedding_list>, # list[float] from Ollama /api/embeddings
|
||||||
|
using="dense", # named-vector selector
|
||||||
|
limit=5,
|
||||||
|
with_payload=True,
|
||||||
|
)
|
||||||
|
for pt in res.points:
|
||||||
|
print(pt.score, pt.payload.get("text"))
|
||||||
|
```
|
||||||
|
|
||||||
|
## One collection per project/domain (multi-collection design)
|
||||||
|
|
||||||
|
For a distinct document set (batch of PDF protocol/files), create a SEPARATE
|
||||||
|
collection with the SAME schema (`dense` 1024d COSINE + sparse `sparse` BM25) and the
|
||||||
|
SAME embedder (bge-m3). Keep search query embeddings compatible by using the same
|
||||||
|
embedder for all collections.
|
||||||
|
Benefits: independent re-index, per-domain context search, no pollution of the
|
||||||
|
general KB. Example: `skc_vinny_gorod` alongside `knowledge_base`.
|
||||||
|
|
||||||
|
## Do NOT retarget context_enhancer to a second collection
|
||||||
|
|
||||||
|
`context_enhancer.py` binds `COLLECTION = os.environ.get("QDRANT_COLLECTION",
|
||||||
|
"knowledge_base")` at IMPORT time (module level). Swapping the env var at runtime
|
||||||
|
does NOT retarget it — the module-level constant is already fixed.
|
||||||
|
|
||||||
|
To search a second collection, write a STANDALONE REST search:
|
||||||
|
1. `POST http://ollama:11434/api/embeddings` `{"model": "bge-m3", "prompt": text}` → `embedding` (1024d).
|
||||||
|
2. `POST http://qdrant:6333/collections/<NAME>/points/query` with `{"vector": emb, "limit": N, "with_payload": true, "using": "dense"}`.
|
||||||
|
3. Read `result.points[*].payload` + `.score`.
|
||||||
|
|
||||||
|
This same pattern is the foundation for a per-project context-injector hook
|
||||||
|
(e.g. NetBox project context) — hit the secondary collection directly, don't go
|
||||||
|
through the shared KB search.
|
||||||
|
|
||||||
|
## Scanned PDFs: pymupdf returns empty text (no text layer)
|
||||||
|
|
||||||
|
`page.get_text("text")` returns `""` for image-only pages — verified on a real
|
||||||
|
51-page government PDF (0 text on every page). It is a genuine scan, not a glitch.
|
||||||
|
Mark scanned pages `[SCANNED_PAGE]` and route to OCR (marker-pdf / vision);
|
||||||
|
do NOT report "no content" or fabricate text. pymupdf's built-in Type1 fonts
|
||||||
|
(times-roman, helv, cour, tiro) do NOT contain the Cyrillic glyph map — inserting
|
||||||
|
Cyrillic with them renders as dots and re-extracts as dots. Real PDFs (generated
|
||||||
|
from Word/CAD) embed proper fonts and extract Cyrillic fine; the byte test above is
|
||||||
|
only a pymupdf-font artifact, not a real-PDF problem.
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
# WebUI Session Cache Reset (Post-Agent-Update)
|
||||||
|
|
||||||
|
## Симптомы
|
||||||
|
|
||||||
|
После обновления Hermes agent (особенно 0.19.x → 0.20.x+):
|
||||||
|
|
||||||
|
- WebUI бесконечно показывает "Loading conversation..."
|
||||||
|
- Создание нового чата не помогает
|
||||||
|
- Даже ответ в существующем чате не рендерится
|
||||||
|
- Формат сессий на диске изменился, старый фронтенд не может его отрендерить
|
||||||
|
|
||||||
|
## Фикс
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 1. Сбросить кэш сессий внутри контейнера WebUI
|
||||||
|
docker exec hermes-webui sh -c 'rm -rf /home/hermeswebui/.hermes/webui/sessions/*'
|
||||||
|
|
||||||
|
# 2. Перезапустить WebUI
|
||||||
|
cd /opt/hermes/docker && docker compose restart hermes-webui
|
||||||
|
|
||||||
|
# 3. Открыть в браузере приватное/инкогнито окно (чтобы не было клиентского кэша)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Почему это происходит
|
||||||
|
|
||||||
|
Hermes WebUI — отдельный контейнер со своим образом. Когда агент обновляется (через `!hermes update`), сессии на диске могут изменить формат (новые поля, другая структура context_messages). WebUI-образ не обновляется автоматически — он остаётся старым и не умеет парсить новые сессии.
|
||||||
|
|
||||||
|
Радикальное решение: сбросить кэш старых сессий, WebUI начнёт с чистого листа.
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
test_qdrant_search.py — Тестовый скрипт поиска по Qdrant.
|
||||||
|
Берёт текст, получает эмбеддинг через Ollama, ищет в Qdrant.
|
||||||
|
|
||||||
|
Использование: python3 scripts/test_qdrant_search.py --query "текст"
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
import urllib.request
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
OLLAMA_URL = "http://localhost:11434/api/embeddings"
|
||||||
|
EMBEDDING_MODEL = "nomic-embed-text:latest"
|
||||||
|
QDRANT_SEARCH_URL = "http://localhost:6333/collections/knowledge_base/points/search"
|
||||||
|
TOP_K = 5
|
||||||
|
|
||||||
|
|
||||||
|
def get_embedding(text: str) -> list[float]:
|
||||||
|
payload = json.dumps({"model": EMBEDDING_MODEL, "prompt": text}).encode()
|
||||||
|
req = urllib.request.Request(OLLAMA_URL, data=payload,
|
||||||
|
headers={"Content-Type": "application/json"})
|
||||||
|
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||||
|
data = json.loads(resp.read())
|
||||||
|
return data.get("embedding")
|
||||||
|
|
||||||
|
|
||||||
|
def search_qdrant(vector: list[float], top_k: int = TOP_K) -> list[dict]:
|
||||||
|
payload = json.dumps({
|
||||||
|
"vector": {"name": "dense", "vector": vector},
|
||||||
|
"limit": top_k,
|
||||||
|
"with_payload": True,
|
||||||
|
"with_vector": False,
|
||||||
|
}).encode()
|
||||||
|
req = urllib.request.Request(QDRANT_SEARCH_URL, data=payload,
|
||||||
|
headers={"Content-Type": "application/json"})
|
||||||
|
with urllib.request.urlopen(req, timeout=15) as resp:
|
||||||
|
data = json.loads(resp.read())
|
||||||
|
return data.get("result", [])
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--query", "-q", required=True)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
vector = get_embedding(args.query)
|
||||||
|
print(f"Embedding: {len(vector)} dims")
|
||||||
|
|
||||||
|
results = search_qdrant(vector)
|
||||||
|
print(f"Results: {len(results)}")
|
||||||
|
for i, r in enumerate(results, 1):
|
||||||
|
payload = r.get("payload", {})
|
||||||
|
text = (payload.get("text") or payload.get("content", ""))[:200]
|
||||||
|
print(f" #{i} score={r['score']:.4f} | {payload.get('source','?')} | {text}...")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Reference in New Issue
Block a user