diff --git a/AGENTS.md b/AGENTS.md index 0563859..f9a35ec 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,24 +1,35 @@ -# VESTI — новостной конвейер +# Ctrl-C: VESTI — новостной конвейер -Конвейер новостей: краулер источников → классификатор (LLM) → банк статей → веб-интерфейс отбора → публикация в Telegram (@dedinit_vesti) и Fediverse (GoToSocial @vesti@dedinit.ru). +## Что это -## Документация (читать при старте сессии) +Конвейер новостей: краулер источников → классификатор (LLM) → банк статей → +веб-интерфейс отбора (модерация) → публикация в Telegram (@dedinit_vesti) +и Fediverse (GoToSocial @vesti@dedinit.ru). -- **STATUS.md** — точка входа: текущее состояние, архитектура, доступы. Читать первым. -- PRD.md — продуктовое описание. -- WALKTHROUGH.md — разборы этапов, команды, засады. -- TODO.md — задачи. +## Стек -## Расположение и доступ - -- Проект: /opt/vesti (docker compose в корне, services/publisher/) -- Веб-интерфейс: https://vesti.nixg.ru (uvicorn :8400, systemd vesti-web.service) -- Publisher: docker-контейнер vesti-publisher, 127.0.0.1:8410 -- БД: SQLite в проекте; медиа в media/ -- Бэкап: backup.sh → backups/ + Яндекс.Диск (root cron 2:45) -- Cron Hermes: vesti-crawler-all-sources (*/30), vesti-watchdog (*/15) +FastAPI + Jinja2 (vesti-web, :8400), publisher-service (Docker, :8410), +SQLite db/vesti.db, локальная LLM — Ollama qwen3:8b-nothink. +Фронтенд — локальный Bootstrap 5.3 (без CDN), без htmx/JS-фреймворков. ## Правила -- При старте сессии из этой папки сначала прочитай STATUS.md (и PRD.md при необходимости), потом действуй. -- Логи почтовых/публикационных сервисов — при наличии смотри перед диагнозом. \ No newline at end of file +- **ВСЕ изменения — через OpenSpec** (change в openspec/changes/), перед каждым + планируемым изменением — **grill-with-docs** (обязательно, AGENT.MD). +- НЕ начинать правки кода, пока change не создан и `openspec validate` не чисто. +- После правок — STATUS.md/TODO.md + бэкап (автокрон, вручную не запускать). +- Дата в UI — фильтр `| dt` (ЧЧ:ММ ДД.ММ.ГГГГ), markdown — XSS-safe фильтр. +- Секреты — только в .env, не коммитить; .env.example актуализировать. +- Ссылки на ресурсы и доступы — фиксировать в файлах проекта. + +## Запуск / проверка + +```bash +cd /opt/vesti +sudo systemctl restart vesti-web # веб :8400 +docker compose -f services/publisher/docker-compose.yml up -d --build # publisher :8410 +curl -s http://127.0.0.1:8410/healthz +openspec validate +``` + +Внешний доступ: https://vesti.nixg.ru (Caddy на VPS02, reverse_proxy 10.8.0.2:8400). \ No newline at end of file diff --git a/CONTEXT.md b/CONTEXT.md new file mode 100644 index 0000000..ed89e33 --- /dev/null +++ b/CONTEXT.md @@ -0,0 +1,24 @@ +# CONTEXT.md — Глоссарий домена VESTI + +Запись терминов по мере их уточнения в дизайн-сессиях. Только глоссарий: +никаких деталей реализации, скетчей спецификаций или ADR. + +## Термины + +### Архив +Страница опубликованных постов на сайте, организованная по времени публикации. +Представляет собой **не** плоский список, а **дерево навигации**: годы → месяцы. + +### Дерево навигации архива +Навигационный элемент (правая колонка страницы архива), перечисляющий годы и, +внутри каждого года, месяцы. Строится **только из фактически существующих** +периодов (постов): если в периоде нет постов — он не появляется в дереве. +Клик по месяцу показывает посты за этот месяц, клик по году — все посты за год. + +### Период без даты +Группа постов, у которых не заполнена дата публикации (`published_at`). +Показывается в дереве навигации отдельной группой «Без даты». + +### Текущий период (активный) +Период (год/месяц или «все посты»), выбранный сейчас в дереве навигации; +подсвечивается, чтобы модератор видел, где он находится. \ No newline at end of file diff --git a/backup.sh b/backup.sh index aed82e9..52400ec 100755 --- a/backup.sh +++ b/backup.sh @@ -15,15 +15,15 @@ mkdir -p ${BACKUP_DIR} echo "🔄 [$(date)] Начинаем резервное копирование VESTI..." -# Проверяем, примонтирован ли Яндекс.Диск -if ! mountpoint -q ${YADISK_MOUNT}; then - echo " ❌ ОШИБКА: Яндекс.Диск не примонтирован!" - exit 1 +# Проверяем, примонтирован ли Яндекс.Диск (не падаем, если нет — копия локально) +if timeout 10 mountpoint -q ${YADISK_MOUNT}; then + YADISK_UP=1 + mkdir -p ${YADISK_TARGET} 2>/dev/null +else + echo " ⚠️ Яндекс.Диск не примонтирован — копия только локально" + YADISK_UP=0 fi -# Создаём папку для бэкапов на Яндекс.Диске (если её нет) -mkdir -p ${YADISK_TARGET} 2>/dev/null - # 1. Бэкап проекта одним архивом ARCHIVE="${BACKUP_DIR}/vesti_${DATE}.tar.gz" echo " → Архив проекта..." @@ -31,9 +31,14 @@ tar czf ${ARCHIVE} -C ${PROJECT_DIR} ${INCLUDE} 2>/dev/null if [ -s "${ARCHIVE}" ]; then echo " ✅ Архив: $(du -h ${ARCHIVE} | cut -f1)" - echo " ☁️ Копирование на Яндекс.Диск..." - cp ${ARCHIVE} ${YADISK_TARGET}/ - echo " ✅ скопирован на ЯД (${YADISK_TARGET}/)" + # Копирование на ЯД только если архив <= 2GiB (больше — вешает davfs/systemd1) + archive_size=$(stat -c %s "${ARCHIVE}" 2>/dev/null || echo 0) + if [ "${YADISK_UP}" = "1" ] && [ "${archive_size}" -le $((2 * 1024 * 1024 * 1024)) ]; then + echo " ☁️ Копирование на Яндекс.Диск..." + timeout 3600 cp ${ARCHIVE} ${YADISK_TARGET}/ && echo " ✅ скопирован на ЯД (${YADISK_TARGET}/)" + else + echo " 📦 архив ${archive_size}B — НЕ копирую на ЯД (>2GiB или ЯД офлайн), остаётся локально" + fi else echo " ❌ Ошибка создания архива!" fi @@ -42,8 +47,10 @@ fi # Локально: оставляем 7 дней find ${BACKUP_DIR} -type f -name "vesti_*" -mtime +7 -delete 2>/dev/null -# На Яндекс.Диске: оставляем 30 дней -find ${YADISK_TARGET} -type f -name "vesti_*" -mtime +30 -delete 2>/dev/null +# На Яндекс.Диске: оставляем 30 дней (только если смонтирован) +if [ "${YADISK_UP}" = "1" ]; then + timeout 60 find ${YADISK_TARGET} -type f -name "vesti_*" -mtime +30 -delete 2>/dev/null +fi echo "✅ [$(date)] Резервное копирование завершено!" diff --git a/classifier/classify.py b/classifier/classify.py index 670f7aa..f1849a9 100644 --- a/classifier/classify.py +++ b/classifier/classify.py @@ -24,16 +24,33 @@ def classify_text(post: dict) -> dict: пробуем LLM-уточнение (relevance/interest/summary). При недоступности LLM — фолбэк: direction из словаря, relevance=low, interest=1, classified=False. + Если у поста задан source_slug с правилом SOURCE_RULES (например, lwn → tech) — + направление принудительное, LLM не вызывается (детерминированно, быстро). + СВОЙ контент (is_own=1): «сильный кандидат» — при словарном попадании relevance=critical, classified=True даже без LLM (фолбэк 'dict-own'). Если словарь не дал направление — классификации нет, пост не теряется. - post: {"text": str, "views": int, "reactions_total": int, "is_own": int, ...} + post: {"text": str, "views": int, "reactions_total": int, "is_own": int, "source_slug": str|None, ...} """ text = (post.get("text") or "").strip() views = int(post.get("views") or 0) reactions = int(post.get("reactions_total") or 0) is_own = int(post.get("is_own") or 0) == 1 + source_slug = post.get("source_slug") + + # правило источника (SOURCE_RULES): принудительное направление, без LLM + from classifier.keywords import source_directions + forced = source_directions(source_slug) + if forced: + return { + "direction": forced[0], + "relevance": "high", + "interest": 3, + "summary": "", + "classified": True, + "method": "source-rule", + } direction = find_direction(text) if not direction: @@ -170,7 +187,9 @@ def classify_posts_in_db(db_path: str, limit: int = 200, direction: str | None = where += f" AND direction IS NULL" # не классифицированы в этом направлении rows = cur.execute( - f"SELECT id, text, views, reactions_total, is_own FROM posts {where} ORDER BY is_own DESC, id LIMIT ?", + f"SELECT p.id, p.text, p.views, p.reactions_total, p.is_own, s.slug AS source_slug " + f"FROM posts p LEFT JOIN sources s ON s.id = p.source_id {where} " + f"ORDER BY p.is_own DESC, p.id LIMIT ?", (limit,), ).fetchall() @@ -188,7 +207,7 @@ def classify_posts_in_db(db_path: str, limit: int = 200, direction: str | None = ) # мультинаправления: все словарные попадания → classifications (для fan-out) from classifier.keywords import find_directions_all - dirs_all = find_directions_all(row["text"] or "") + dirs_all = find_directions_all(row["text"] or "", row["source_slug"]) if res["direction"] and res["direction"] not in dirs_all: dirs_all.append(res["direction"]) for d in dirs_all: diff --git a/classifier/keywords.py b/classifier/keywords.py index e13065e..0c827dc 100644 --- a/classifier/keywords.py +++ b/classifier/keywords.py @@ -93,9 +93,28 @@ import re # Направление по умолчанию, если ничего не подошло DEFAULT_DIRECTION = "linux" # прототип: все посты про linux окружение +# Правила по источнику (slug в sources.yaml): принудительное направление +# для всего контента источника, независимо от словаря/LLM. +# LWN — технологическое СМИ: весь контент → tech. +SOURCE_RULES: dict[str, list[str]] = { + "lwn": ["tech"], +} -def find_directions_all(text: str) -> list[str]: - """Возвращает ВСЕ направления, которым соответствует текст (для мультинаправлений/fan-out).""" + +def source_directions(source_slug: str | None) -> list[str]: + """Направления, принудительно назначенные источнику (SOURCE_RULES).""" + return list(SOURCE_RULES.get(source_slug or "", [])) + + +def find_directions_all(text: str, source_slug: str | None = None) -> list[str]: + """Возвращает ВСЕ направления, которым соответствует текст (для мультинаправлений/fan-out). + + Если для источника задано правило SOURCE_RULES — возвращает его направления + (принудительно), иначе — словарные попадания. + """ + forced = source_directions(source_slug) + if forced: + return forced text_low = (text or "").lower() hits = set() for direction, cfg in DIRECTIONS.items(): @@ -112,8 +131,14 @@ def find_directions_all(text: str) -> list[str]: return list(hits) -def find_direction(text: str) -> str | None: - """Возвращает направление, если текст совпал со словарём (иначе None).""" +def find_direction(text: str, source_slug: str | None = None) -> str | None: + """Возвращает направление: правило источника (SOURCE_RULES) или словарное. + + Принудительное правило источника имеет приоритет над словарём. + """ + forced = source_directions(source_slug) + if forced: + return forced[0] dirs = find_directions_all(text) if dirs: return dirs[0] diff --git a/docs/agents/issue-tracker.md b/docs/agents/issue-tracker.md new file mode 100644 index 0000000..b33fb1d --- /dev/null +++ b/docs/agents/issue-tracker.md @@ -0,0 +1,63 @@ +# Issue Tracker + +**Tracker:** Gitea (self-hosted, https://gitea.nixg.ru) +**Repository:** `estorozhenko/vesti` (mirror from gitverse.ru/kpa39l/vesti) +**Web UI:** https://gitea.nixg.ru/estorozhenko/vesti + +This is a custom ("Other") tracker configuration for the Matt Pocock engineering skills (`to-tickets`, `to-spec`, `triage`). Gitea is not natively supported by `/setup-matt-pocock-skills`, so the workflow is recorded here as freeform prose. + +## How to publish issues + +Use the **`tea` CLI** (v0.16.0, at `~/.local/bin/tea`) with the login **`gitea.nixg-full`**: + +```bash +export PATH="$HOME/.local/bin:$PATH" + +# Create an issue +tea issues create --login gitea.nixg-full --repo estorozhenko/vesti \ + --title "Ticket title" \ + --description "Ticket body (use the to-tickets per-ticket template)" \ + --labels "ready-for-agent" + +# List issues +tea issues list --login gitea.nixg-full --repo estorozhenko/vesti -o simple + +# Show one issue (with comments) +tea issues --login gitea.nixg-full --repo estorozhenko/vesti 42 --comments + +# Add a comment +tea comment --login gitea.nixg-full --repo estorozhenko/vesti 42 "comment" + +# Close / reopen +tea issues close --login gitea.nixg-full --repo estorozhenko/vesti 42 +``` + +**Alternative:** the `gitea` MCP server (gitea-mcp, 72 tools) is configured in Hermes `mcp_servers.gitea` — same instance, same token. Prefer MCP tools in agent sessions, `tea` for quick shell checks. + +## Repository and credentials + +- Instance: `https://gitea.nixg.ru` (API v1.26.2, external VPS 5.129.217.146) +- Owner/repo: `estorozhenko/vesti` (mirror of `https://gitverse.ru/kpa39l/vesti.git`, issues enabled: yes) +- Login (tea): `gitea.nixg-full` — token `hermes-mcp-full` +- Token source: `/opt/hermes/.hermes/secrets/git-tokens.env` (`GITEA_NIXG_TOKEN`); also in Hermes `config.yaml` → `mcp_servers.gitea` env `GITEA_TOKEN`. +- Git push: to the **origin** (gitverse.ru) via SSH key `/home/estorozhenko/.ssh/gitverse` — the Gitea repo is a read-only mirror and is NOT the push target. + +## Triage labels (created) + +| Label | Color | ID | +|---|---|---| +| `needs-triage` | `#e11d21` | 6 | +| `needs-info` | `#fbca04` | 7 | +| `ready-for-agent` | `#0e8a16` | 8 | +| `ready-for-human` | `#006b75` | 9 | +| `wontfix` | `#ffffff` | 10 | + +Use `ready-for-agent` for tickets published by `to-tickets`. + +## Blocking edges + +Gitea has no native issue-blocking links. Follow the `to-tickets` rule for trackers without native blocking: publish tickets in dependency order (blockers first), and set each ticket's **"Blocked by"** section to the blocking issues' real identifiers (e.g. `#12`, `#13`). + +## Scope + +Tickets for the VESTI project live in `estorozhenko/vesti`. Local fallback: `.scratch//issues/` in this project when no Gitea write access. \ No newline at end of file diff --git a/publisher/card.py b/publisher/card.py index 767e303..0442388 100644 --- a/publisher/card.py +++ b/publisher/card.py @@ -25,7 +25,9 @@ def make_card(post: dict, comment: str | None = None) -> dict: это был «агрегаторный» вид карточки. """ comment = (comment or "").strip() - body = (post.get("text") or "").strip() + # тело: пересказ (rewritten_text) приоритетнее исходного текста + body = (post.get("rewritten_text") or post.get("text") or "").strip() + rewritten = bool((post.get("rewritten_text") or "").strip()) # ссылка на оригинал / атрибуция своего контента is_own = int(post.get("is_own") or 0) == 1 @@ -34,7 +36,10 @@ def make_card(post: dict, comment: str | None = None) -> dict: orig = f"https://t.me/dedinit/{post['tg_post_id']}" elif post.get("url"): orig = post.get("url") - if is_own: + if rewritten: + # пересказ — публикация от имени автора канала (не пересылка) + tail = "✍️ Дед в АйТи (@dedinit)" + (f" · Источник: {orig}" if orig else "") + elif is_own: tail = "✍️ Дед в АйТи (@dedinit)" + (f" · {orig}" if orig else "") elif orig: tail = f"🔗 Оригинал: {orig}" diff --git a/sources/sources.yaml b/sources/sources.yaml index 6579106..ef8bead 100644 --- a/sources/sources.yaml +++ b/sources/sources.yaml @@ -1,8 +1,27 @@ -# VESTI sources — реестр источников (git). Формат продолжает /opt/news. -# Прототип: направление linux, язык ru. -# Каналы из списка пользователя (2026-09-08). +- slug: ai_machinelearning_big_data + name: Machinelearning + url: https://t.me/ai_machinelearning_big_data + channel: ai_machinelearning_big_data + crawler: telegram + feed_url: null + direction: llm + lang: ru + priority: P0 + enabled: true + own: false +- slug: dedinit + name: Дед в АйТи + url: https://t.me/dedinit + channel: dedinit + crawler: telegram + direction: null + lang: ru + priority: P0 + enabled: true + own: true + notes: собственный канал пользователя — источник своего контента (own-content-hub) - slug: linuxklub - name: "Linux Klub" + name: Linux Klub url: https://t.me/linuxklub channel: linuxklub crawler: telegram @@ -10,10 +29,9 @@ lang: ru priority: P0 enabled: true - notes: "линукс-канал из списка пользователя" - + notes: линукс-канал из списка пользователя - slug: linuxos_tg - name: "Linux OS" + name: Linux OS url: https://t.me/linuxos_tg channel: linuxos_tg crawler: telegram @@ -21,9 +39,31 @@ lang: ru priority: P0 enabled: true - +- slug: lwn + name: LWN.net + url: https://lwn.net/ + feed_url: https://lwn.net/headlines/rss + channel: null + crawler: rss + direction: linux + lang: en + priority: P0 + enabled: true + notes: первый реальный RSS-источник для проверки rss_crawler + own: false +- slug: bleepingcomputer + name: omarchy + url: https://www.bleepingcomputer.com + channel: null + crawler: rss + feed_url: https://www.bleepingcomputer.com/feed/ + direction: null + lang: ru + priority: P1 + enabled: true + own: false - slug: dotfiles_linux - name: "Dotfiles Linux" + name: Dotfiles Linux url: https://t.me/dotfiles_linux channel: dotfiles_linux crawler: telegram @@ -31,39 +71,8 @@ lang: ru priority: P1 enabled: true - -- slug: linux_education - name: "Linux Education" - url: https://t.me/linux_education - channel: linux_education - crawler: telegram - direction: linux - lang: ru - priority: P1 - enabled: true - -- slug: linuxmastery - name: "Linux Mastery" - url: https://t.me/LinuxMastery - channel: LinuxMastery - crawler: telegram - direction: linux - lang: ru - priority: P1 - enabled: true - -- slug: linuxcamp_tg - name: "Linux Camp" - url: https://t.me/linuxcamp_tg - channel: linuxcamp_tg - crawler: telegram - direction: linux - lang: ru - priority: P1 - enabled: true - - slug: gitgate - name: "GitGate" + name: GitGate url: https://t.me/gitgate channel: gitgate crawler: telegram @@ -71,10 +80,9 @@ lang: ru priority: P1 enabled: true - notes: "git/devops — смежный с linux" - + notes: git/devops — смежный с linux - slug: krxnotes - name: "KRX Notes" + name: KRX Notes url: https://t.me/krxnotes channel: krxnotes crawler: telegram @@ -82,18 +90,64 @@ lang: ru priority: P1 enabled: true - -# --- Собственный контент (own: true) — канал пользователя --------------- -# Посты этого канала — СВОЙ контент пользователя (is_own=1), медиа скачивается, -# кандидат получает приоритет (relevance=critical) и fan-out по всем направлениям при approve. -- slug: dedinit - name: "Дед в АйТи" - url: https://t.me/dedinit - channel: dedinit +- slug: linux_education + name: Linux Education + url: https://t.me/linux_education + channel: linux_education crawler: telegram - direction: null # свой канал — направление не фиксировано (классификатор решает) + direction: linux lang: ru - priority: P0 + priority: P1 enabled: true - own: true - notes: "собственный канал пользователя — источник своего контента (own-content-hub)" +- slug: linuxcamp_tg + name: Linux Camp + url: https://t.me/linuxcamp_tg + channel: linuxcamp_tg + crawler: telegram + direction: linux + lang: ru + priority: P1 + enabled: true +- slug: linuxmastery + name: Linux Mastery + url: https://t.me/LinuxMastery + channel: LinuxMastery + crawler: telegram + direction: linux + lang: ru + priority: P1 + enabled: true +- slug: mknewsru + name: Мой Компьютер + url: https://t.me/mknewsru + channel: mknewsru + crawler: telegram + feed_url: null + direction: tech + lang: ru + priority: P1 + enabled: true + own: false +- slug: omarchy + name: omarchy + url: https://omarchy.org/ + channel: null + crawler: rss + feed_url: https://omarchy.org/news/rss.xml + direction: null + lang: ru + priority: P1 + enabled: false + own: false + +# Пример (включить, раскомментировав и поправив direction/приоритет): +# - slug: opennet +# name: "OpenNET" +# url: https://www.opennet.ru/ +# feed_url: https://www.opennet.ru/opennews/opennews_all.rss +# channel: null +# crawler: rss +# direction: linux +# lang: ru +# priority: P1 +# enabled: true