Add telegram-archiver microservice (MVP)

This commit is contained in:
2026-02-19 21:41:52 +03:00
parent 7166f535d6
commit 6332b91ba4
13 changed files with 1635 additions and 0 deletions
+374
View File
@@ -0,0 +1,374 @@
"""
Core archiver logic.
Handles message processing, markdown generation, and file organization.
"""
import asyncio
import logging
from datetime import datetime
from pathlib import Path
from typing import Optional
import yaml
from telethon.tl.types import Message
from app.models import PostData, MediaFile, MediaType
from app.telethon_client import TelethonArchiver
from config import Settings, get_settings
logger = logging.getLogger(__name__)
class ChannelArchiver:
"""
Handles archiving of a single Telegram channel.
Manages bundle creation, markdown generation, and media downloads.
"""
def __init__(
self,
telethon_client: TelethonArchiver,
settings: Optional[Settings] = None,
):
self.client = telethon_client
self.settings = settings or get_settings()
self.channel_path: Optional[Path] = None
self.log_file: Optional[Path] = None
# Statistics
self.stats = {
"posts_archived": 0,
"posts_skipped": 0,
"media_downloaded": 0,
"media_skipped_large": 0,
"large_files": [],
}
def _setup_logging(self, channel_username: str) -> None:
"""Setup file logging for this channel archive session."""
timestamp = datetime.now().strftime("%Y%m%d")
log_filename = f"{timestamp}-{channel_username}.log"
self.log_file = Path(__file__).parent.parent / log_filename
# Create file handler
file_handler = logging.FileHandler(self.log_file, encoding="utf-8")
file_handler.setLevel(self.settings.log_level)
# Create formatter
formatter = logging.Formatter(
"%(asctime)s - %(name)s - %(levelname)s - %(message)s"
)
file_handler.setFormatter(formatter)
# Add handler to logger
logger.addHandler(file_handler)
logger.info(f"=== Archive session started for {channel_username} ===")
def _get_channel_path(self, channel_username: str) -> Path:
"""Get the output directory path for this channel."""
# Remove @ prefix if present
clean_username = channel_username.lstrip("@")
base_path = self.settings.output_dir / clean_username
base_path.mkdir(parents=True, exist_ok=True)
return base_path
async def archive_channel(
self,
channel: str,
limit: Optional[int] = None,
from_message_id: Optional[int] = None,
force: bool = False,
) -> dict:
"""
Archive all messages from a channel.
Args:
channel: Channel username or ID
limit: Maximum number of messages to archive
from_message_id: Start from this message ID
force: Force re-download of already archived posts
Returns:
Statistics dictionary
"""
start_time = datetime.now()
# Setup
self._setup_logging(channel)
self.channel_path = self._get_channel_path(channel)
logger.info(f"Starting archive of {channel}")
logger.info(f"Output directory: {self.channel_path}")
if limit:
logger.info(f"Limit: {limit} messages")
if from_message_id:
logger.info(f"Starting from message ID: {from_message_id}")
# Get channel info
entity = await self.client.get_channel_info(channel)
channel_title = getattr(entity, "title", channel)
# Reset stats
self.stats = {
"posts_archived": 0,
"posts_skipped": 0,
"media_downloaded": 0,
"media_skipped_large": 0,
"large_files": [],
}
# Process messages
async for message in self.client.fetch_messages(
channel, limit=limit, from_message_id=from_message_id
):
await self._process_message(message, force=force)
# Write large files report
large_files_report = await self._write_large_files_report()
# Calculate duration
duration = (datetime.now() - start_time).total_seconds()
logger.info(f"=== Archive completed in {duration:.2f}s ===")
logger.info(f"Posts archived: {self.stats['posts_archived']}")
logger.info(f"Posts skipped: {self.stats['posts_skipped']}")
logger.info(f"Media downloaded: {self.stats['media_downloaded']}")
logger.info(f"Media skipped (large): {self.stats['media_skipped_large']}")
return {
"channel": channel,
"channel_title": channel_title,
"posts_archived": self.stats["posts_archived"],
"posts_skipped": self.stats["posts_skipped"],
"media_downloaded": self.stats["media_downloaded"],
"media_skipped_large": self.stats["media_skipped_large"],
"output_path": str(self.channel_path),
"log_file": str(self.log_file),
"large_files_report": large_files_report,
"duration_seconds": duration,
}
async def _process_message(
self, message: Message, force: bool = False
) -> None:
"""
Process a single message: create bundle, download media, generate markdown.
Args:
message: Telegram message to process
force: Force re-processing if bundle exists
"""
message_id = message.id
bundle_dir = self.channel_path / str(message_id)
# Check if already archived (skip if exists and not force)
if bundle_dir.exists() and not force:
logger.debug(f"Skipping message {message_id}: already archived")
self.stats["posts_skipped"] += 1
return
logger.debug(f"Processing message {message_id}...")
# Create bundle directory
bundle_dir.mkdir(parents=True, exist_ok=True)
# Extract post data
post_data = await self._extract_post_data(message)
# Download media
if message.media:
media_info = await self.client.download_media(
message, bundle_dir, self.settings.max_file_size
)
# Update post_data with media info
for info in media_info:
media_file = MediaFile(
filename=info["filename"],
type=info["type"],
caption=info.get("caption"),
size=info.get("size"),
is_too_large=info.get("is_too_large", False),
)
post_data.media_files.append(media_file)
if info.get("is_too_large"):
self.stats["media_skipped_large"] += 1
self.stats["large_files"].append(
{
"message_id": message_id,
"filename": info["filename"],
"size": info.get("size"),
}
)
else:
self.stats["media_downloaded"] += 1
# Generate markdown
markdown_content = self._generate_markdown(post_data)
# Write index.md
index_path = bundle_dir / "index.md"
index_path.write_text(markdown_content, encoding="utf-8")
self.stats["posts_archived"] += 1
logger.debug(f"Message {message_id} archived successfully")
async def _extract_post_data(self, message: Message) -> PostData:
"""
Extract structured data from a Telegram message.
Args:
message: Telegram message
Returns:
PostData object with all metadata
"""
# Get channel info
channel = await message.get_chat()
channel_username = getattr(channel, "username", "unknown")
channel_title = getattr(channel, "title", channel_username)
# Convert message text to markdown
text = message.message or ""
if message.message:
# Use Telethon's built-in markdown converter
text = message.get_message_text(markdown=True)
# Handle reply-to
reply_to = None
if message.reply_to and message.reply_to.reply_to_msg_id:
reply_to = message.reply_to.reply_to_msg_id
# Handle repost/forward
repost_from = None
repost_channel = None
if message.fwd_from:
repost_from = message.fwd_from.channel_post
if message.fwd_from.from_id:
# Try to get original channel info
try:
fwd_channel = await self.client.client.get_entity(
message.fwd_from.from_id
)
repost_channel = getattr(fwd_channel, "title", None)
except Exception:
pass
return PostData(
message_id=message.id,
date=message.date,
text=text,
author=channel_title,
channel_username=channel_username,
reply_to=reply_to,
repost_from=repost_from,
repost_channel=repost_channel,
views=getattr(message, "views", None),
)
def _generate_markdown(self, post: PostData) -> str:
"""
Generate markdown content with front-matter for Hugo.
Args:
post: PostData object
Returns:
Markdown string with YAML front-matter
"""
# Build front-matter
frontmatter = {
"message_id": post.message_id,
"date": post.date.isoformat(),
"author": post.author,
}
# Add optional fields
if post.reply_to:
# Relative link to the replied post's index.md
frontmatter["reply_to"] = f"../{post.reply_to}/index.md"
if post.repost_from:
frontmatter["repost_from"] = post.repost_from
if post.repost_channel:
frontmatter["repost_channel"] = post.repost_channel
if post.media_files:
frontmatter["media_files"] = [
{
"filename": mf.filename,
"type": mf.type,
"caption": mf.caption,
"size": mf.size,
"is_too_large": mf.is_too_large,
}
for mf in post.media_files
]
# Build markdown content
content = []
# Add media embeds for downloaded files
for mf in post.media_files:
if not mf.is_too_large:
if mf.type == "photo":
content.append(f"![{mf.caption or ''}]({mf.filename})")
elif mf.type == "video":
content.append(f"@[video]({mf.filename})")
elif mf.type == "audio":
content.append(f"@[audio]({mf.filename})")
elif mf.type == "document":
content.append(f"📎 [{mf.filename}]({mf.filename})")
# Add text content
if post.text:
content.append(post.text)
# Mark reposts visually
if post.repost_from:
content.insert(0, f"> *Repost from {post.repost_channel or 'Unknown'}*")
content.insert(1, "")
# Combine front-matter and content
yaml_frontmatter = yaml.dump(
frontmatter,
allow_unicode=True,
default_flow_style=False,
sort_keys=False,
)
return f"---\n{yaml_frontmatter}---\n\n{''.join(content)}"
async def _write_large_files_report(self) -> Optional[str]:
"""
Write report of files that were too large to download.
Returns:
Path to report file or None if no large files
"""
if not self.stats["large_files"]:
return None
report_path = self.channel_path / "2big2get.md"
content = ["# Files Too Large to Download\n\n"]
content.append(
f"Generated: {datetime.now().isoformat()}\n\n"
)
content.append(
f"Total files skipped: {len(self.stats['large_files'])}\n\n"
)
content.append("| Message ID | Filename | Size (bytes) |\n")
content.append("|------------|----------|-------------|\n")
for file_info in self.stats["large_files"]:
content.append(
f"| {file_info['message_id']} | {file_info['filename']} | "
f"{file_info['size']} |\n"
)
report_path.write_text("".join(content), encoding="utf-8")
logger.info(f"Large files report written to {report_path}")
return str(report_path)