Fix duplicate posting, HTML escaping, restore sendRichMessage correctly

- Stop re-sending to a platform that already succeeded when the other
  platform failed (was causing duplicate posts on partial failure)
- Escape VK post text before sending with parse_mode=HTML instead of
  only escaping matched bracket-link substrings (bare & < > broke sends)
- Restore Telegram sendRichMessage (Bot API 10.1) ported faithfully from
  new_vk_parser's proven implementation, with fallback to standard
  send_photo/send_video/send_media_group
- Restore "media unavailable" note when a video/photo fails to download
  instead of silently dropping it
- Handle YouTube link attachments from VK posts (route into yt-dlp)
- MAX: chunk media beyond 10 items into follow-up messages instead of
  dropping them
- yt-dlp format selector no longer falls back to unconstrained height
- Wire up VK_RATE_LIMIT_RPS, drop unused Pillow dep and dead config fields
This commit is contained in:
2026-08-14 21:21:57 +05:00
parent d63d3fb937
commit 6a7a995d5f
8 changed files with 289 additions and 122 deletions
-1
View File
@@ -3,5 +3,4 @@ aiohttp==3.12.13
aiosqlite==0.21.0 aiosqlite==0.21.0
loguru==0.7.3 loguru==0.7.3
pydantic-settings==2.10.1 pydantic-settings==2.10.1
Pillow==11.3.0
yt-dlp>=2026.1.1 yt-dlp>=2026.1.1
+3 -2
View File
@@ -26,8 +26,9 @@ class Settings(BaseSettings):
tg_media_channel_id: str = "" # Optional storage channel tg_media_channel_id: str = "" # Optional storage channel
tg_admin_ids: str = "" # Comma-separated admin IDs for reports, e.g. "123456,789012" tg_admin_ids: str = "" # Comma-separated admin IDs for reports, e.g. "123456,789012"
local_bot_api_url: str = "" # e.g., "http://127.0.0.1:8081" local_bot_api_url: str = "" # e.g., "http://127.0.0.1:8081"
telegram_api_id: str = "" # NOTE: TELEGRAM_API_ID / TELEGRAM_API_HASH are intentionally not modeled here -
telegram_api_hash: str = "" # they're only consumed by docker-entrypoint.sh (raw env) to start the local
# telegram-bot-api binary, never read from Python.
# MAX Messenger Settings # MAX Messenger Settings
max_bot_token: str = "" max_bot_token: str = ""
+43 -34
View File
@@ -59,7 +59,7 @@ class ServiceApp:
if not settings.vk_source: if not settings.vk_source:
raise ValueError("VK_SOURCE is not set in configuration") raise ValueError("VK_SOURCE is not set in configuration")
async with VKClient() as vk: async with VKClient(rps=settings.vk_rate_limit_rps) as vk:
screen_name, owner_id, name = await vk.resolve_group(settings.vk_source) screen_name, owner_id, name = await vk.resolve_group(settings.vk_source)
self.vk_group_owner_id = owner_id self.vk_group_owner_id = owner_id
self.vk_group_name = name self.vk_group_name = name
@@ -81,6 +81,9 @@ class ServiceApp:
text=post.text, text=post.text,
raw_data=post.raw, raw_data=post.raw,
) )
existing = await self.db.get_post(post.owner_id, post.post_id) or {}
tg_done = existing.get("tg_status") in ("published", "skipped")
max_done = existing.get("max_status") in ("published", "skipped")
media_processor = MediaProcessor(is_local_tg_api=self.tg_poster.is_local_api) media_processor = MediaProcessor(is_local_tg_api=self.tg_poster.is_local_api)
processed_media = [] processed_media = []
@@ -88,44 +91,53 @@ class ServiceApp:
result_summary: dict[str, Any] = { result_summary: dict[str, Any] = {
"vk_post_id": post.post_id, "vk_post_id": post.post_id,
"vk_post_url": vk_url, "vk_post_url": vk_url,
"tg_status": "pending", "tg_status": existing.get("tg_status", "pending") if tg_done else "pending",
"tg_url": None, "tg_url": existing.get("tg_url") if tg_done else None,
"tg_error": None, "tg_error": None,
"max_status": "pending", "max_status": existing.get("max_status", "pending") if max_done else "pending",
"max_url": None, "max_url": existing.get("max_url") if max_done else None,
"max_error": None, "max_error": None,
} }
try: try:
# 1. Download/extract media # 1. Download/extract media (skip entirely if both platforms are already done)
if post.media: if post.media and not (tg_done and max_done):
logger.info("Downloading {} media items for post #{}...", len(post.media), post.post_id) logger.info("Downloading {} media items for post #{}...", len(post.media), post.post_id)
processed_media = await media_processor.process_media_items(post.media) processed_media = await media_processor.process_media_items(post.media)
# 2. Publish to Telegram # 2. Publish to Telegram (only if not already published/skipped for this post)
try: if tg_done:
tg_mids, tg_url = await self.tg_poster.post_to_telegram( logger.debug("Post #{} already resolved for Telegram ({}), skipping resend.", post.post_id, existing.get("tg_status"))
raw_text=post.text, else:
media_items=processed_media, try:
vk_url=vk_url, tg_mids, tg_url = await self.tg_poster.post_to_telegram(
) raw_text=post.text,
await self.db.update_tg_result( media_items=processed_media,
post_db_id=post_db_id, vk_url=vk_url,
status="published", )
message_ids=tg_mids, await self.db.update_tg_result(
url=tg_url, post_db_id=post_db_id,
) status="published",
result_summary["tg_status"] = "published" message_ids=tg_mids,
result_summary["tg_url"] = tg_url url=tg_url,
except Exception as exc: )
err = str(exc) result_summary["tg_status"] = "published"
logger.exception("Telegram post error for #{}: {}", post.post_id, exc) result_summary["tg_url"] = tg_url
await self.db.update_tg_result(post_db_id=post_db_id, status="failed", error=err) except Exception as exc:
result_summary["tg_status"] = "failed" err = str(exc)
result_summary["tg_error"] = err logger.exception("Telegram post error for #{}: {}", post.post_id, exc)
await self.db.update_tg_result(post_db_id=post_db_id, status="failed", error=err)
result_summary["tg_status"] = "failed"
result_summary["tg_error"] = err
# 3. Publish to MAX Messenger # 3. Publish to MAX Messenger (only if not already published/skipped for this post)
if settings.max_bot_token and settings.max_chat_id: if not (settings.max_bot_token and settings.max_chat_id):
if not max_done:
await self.db.update_max_result(post_db_id=post_db_id, status="skipped")
result_summary["max_status"] = "skipped"
elif max_done:
logger.debug("Post #{} already resolved for MAX ({}), skipping resend.", post.post_id, existing.get("max_status"))
else:
try: try:
max_mids, max_url = await self.max_poster.post_to_max( max_mids, max_url = await self.max_poster.post_to_max(
raw_text=post.text, raw_text=post.text,
@@ -146,9 +158,6 @@ class ServiceApp:
await self.db.update_max_result(post_db_id=post_db_id, status="failed", error=err) await self.db.update_max_result(post_db_id=post_db_id, status="failed", error=err)
result_summary["max_status"] = "failed" result_summary["max_status"] = "failed"
result_summary["max_error"] = err result_summary["max_error"] = err
else:
await self.db.update_max_result(post_db_id=post_db_id, status="skipped")
result_summary["max_status"] = "skipped"
finally: finally:
# Immediate cleanup of temporary media files # Immediate cleanup of temporary media files
@@ -167,7 +176,7 @@ class ServiceApp:
published_reports: list[dict[str, Any]] = [] published_reports: list[dict[str, Any]] = []
try: try:
async with VKClient() as vk: async with VKClient(rps=settings.vk_rate_limit_rps) as vk:
latest_posts = await vk.get_latest_posts( latest_posts = await vk.get_latest_posts(
owner_id=self.vk_group_owner_id, owner_id=self.vk_group_owner_id,
count=settings.vk_check_count, count=settings.vk_check_count,
+42 -15
View File
@@ -11,11 +11,11 @@ import aiohttp
try: try:
from .config import settings from .config import settings
from .media_processor import ProcessedMedia from .media_processor import ProcessedMedia
from .text_formatter import format_post_text, split_message_chunks from .text_formatter import build_media_unavailable_note, format_post_text, split_message_chunks
except (ImportError, ValueError): except (ImportError, ValueError):
from config import settings from config import settings
from media_processor import ProcessedMedia from media_processor import ProcessedMedia
from text_formatter import format_post_text, split_message_chunks from text_formatter import build_media_unavailable_note, format_post_text, split_message_chunks
MAX_MESSAGE_LIMIT = 4000 MAX_MESSAGE_LIMIT = 4000
MAX_MEDIA_ITEMS = 10 MAX_MEDIA_ITEMS = 10
@@ -233,6 +233,21 @@ class MAXPoster:
return {"type": media_type, "payload": payload} return {"type": media_type, "payload": payload}
async def upload_media_group(
self, client: MAXAPIClient, items: list[ProcessedMedia]
) -> list[dict[str, Any]]:
attachments: list[dict[str, Any]] = []
for item in items:
try:
att = await self.upload_media_item(client, item)
if att:
attachments.append(att)
except Exception as exc:
logger.warning("MAX media upload failed for {}: {}", item.attachment_id, exc)
if attachments:
await self.wait_for_videos(client, attachments)
return attachments
async def post_to_max( async def post_to_max(
self, self,
raw_text: str, raw_text: str,
@@ -250,27 +265,29 @@ class MAXPoster:
vk_url=vk_url, vk_url=vk_url,
) )
link_only = [m for m in media_items if m.is_link_only]
note = build_media_unavailable_note(link_only, parse_mode="html")
if note:
formatted_text = f"{formatted_text}\n\n{note}" if formatted_text else note
chunks = split_message_chunks(formatted_text, self.message_limit) chunks = split_message_chunks(formatted_text, self.message_limit)
valid_media = [m for m in media_items if not m.is_link_only and m.local_path][:MAX_MEDIA_ITEMS] valid_media = [m for m in media_items if not m.is_link_only and m.local_path]
# Chunk into groups of MAX_MEDIA_ITEMS instead of silently dropping the excess:
# the first group rides with the text message, extra groups go out as follow-ups.
media_groups = (
[valid_media[i : i + MAX_MEDIA_ITEMS] for i in range(0, len(valid_media), MAX_MEDIA_ITEMS)]
if valid_media
else [[]]
)
message_ids: list[str] = [] message_ids: list[str] = []
first_url: Optional[str] = None first_url: Optional[str] = None
async with MAXAPIClient(self.token, self.api_base_url) as client: async with MAXAPIClient(self.token, self.api_base_url) as client:
attachments: list[dict[str, Any]] = [] first_attachments = await self.upload_media_group(client, media_groups[0])
for item in valid_media:
try:
att = await self.upload_media_item(client, item)
if att:
attachments.append(att)
except Exception as exc:
logger.warning("MAX media upload failed for {}: {}", item.attachment_id, exc)
if attachments:
await self.wait_for_videos(client, attachments)
first_text = chunks[0] if chunks else "" first_text = chunks[0] if chunks else ""
res = await self.send_message_waiting_for_media(client, first_text, attachments) res = await self.send_message_waiting_for_media(client, first_text, first_attachments)
first_mid = self.message_id_from_response(res) first_mid = self.message_id_from_response(res)
if first_mid: if first_mid:
@@ -289,6 +306,16 @@ class MAXPoster:
except Exception as exc: except Exception as exc:
logger.warning("MAX auto reaction failed: {}", exc) logger.warning("MAX auto reaction failed: {}", exc)
for group in media_groups[1:]:
attachments = await self.upload_media_group(client, group)
if not attachments:
continue
await asyncio.sleep(1.0)
sub_res = await self.send_message_waiting_for_media(client, "", attachments)
sub_mid = self.message_id_from_response(sub_res)
if sub_mid:
message_ids.append(sub_mid)
for chunk in chunks[1:]: for chunk in chunks[1:]:
await asyncio.sleep(1.0) await asyncio.sleep(1.0)
sub_res = await client.send_message(self.chat_id, chunk) sub_res = await client.send_message(self.chat_id, chunk)
+5 -4
View File
@@ -123,13 +123,14 @@ class MediaProcessor:
"--no-playlist", "--no-playlist",
"--match-filter", f"duration <= {settings.video_max_duration_sec}", "--match-filter", f"duration <= {settings.video_max_duration_sec}",
"--merge-output-format", "mp4", "--merge-output-format", "mp4",
# Every fallback keeps the height cap: without it yt-dlp can pull an
# arbitrarily large/high-res stream only to have it discarded afterwards
# by the size check below, burning bandwidth and the whole timeout budget.
"-f", ( "-f", (
f"best[height<={settings.video_max_height}][filesize<{max_size}]" f"best[height<={settings.video_max_height}][filesize<{max_size}]"
f"/best[height<={settings.video_max_height}]" f"/best[height<={settings.video_max_height}]"
f"/bestvideo[height<={settings.video_max_height}][filesize<{max_size}]+bestaudio/best" f"/bestvideo[height<={settings.video_max_height}][filesize<{max_size}]+bestaudio"
f"/bestvideo[height<={settings.video_max_height}]+bestaudio/best" f"/bestvideo[height<={settings.video_max_height}]+bestaudio"
f"/best[filesize<{max_size}]"
f"/best"
), ),
"--quiet", "--no-warnings", "--quiet", "--no-warnings",
]) ])
+131 -28
View File
@@ -2,6 +2,9 @@ from __future__ import annotations
import html import html
import re import re
from html.parser import HTMLParser
from typing import Optional
from urllib.parse import urlparse
try: try:
from .config import settings from .config import settings
except (ImportError, ValueError): except (ImportError, ValueError):
@@ -43,33 +46,134 @@ def strip_trailing_hashtags(text: str) -> str:
return "\n".join(lines).rstrip() return "\n".join(lines).rstrip()
# Matches VK bracket markup: [club123|Name], [id123|Name], [public123|Name],
# [event123|Name] or a raw-URL bracket link [https://example.com|Title]
_VK_LINK_RE = re.compile(
r"\[(?:(?P<prefix>club|id|public|event)(?P<obj_id>\d+)|(?P<url>https?://[^\s\|\]]+))\|(?P<title>[^\]]+)\]"
)
def clean_vk_wiki_links(text: str, parse_mode: str = "html") -> str: def clean_vk_wiki_links(text: str, parse_mode: str = "html") -> str:
""" """
Converts VK wiki links to clickable links: Converts VK bracket markup into real links, escaping everything else so the
result is always safe to send with parse_mode="HTML":
- [club12345|Name] -> <a href="https://vk.com/club12345">Name</a> - [club12345|Name] -> <a href="https://vk.com/club12345">Name</a>
- [id12345|Name] -> <a href="https://vk.com/id12345">Name</a> - [id12345|Name] -> <a href="https://vk.com/id12345">Name</a>
- [public12345|Name] -> <a href="https://vk.com/public12345">Name</a>
- [event12345|Name] -> <a href="https://vk.com/event12345">Name</a>
- [https://example.com|Title] -> <a href="https://example.com">Title</a> - [https://example.com|Title] -> <a href="https://example.com">Title</a>
Any text outside of recognized bracket markup is HTML-escaped, so stray
"&", "<", ">" characters in real VK post text never break the parser.
""" """
def _replace_vk(match: re.Match) -> str: if parse_mode != "html":
prefix = match.group(1) def _replace_plain(match: re.Match) -> str:
obj_id = match.group(2) title = match.group("title")
title = match.group(3) if match.group("prefix"):
if parse_mode == "html": target = f"https://vk.com/{match.group('prefix')}{match.group('obj_id')}"
return f'<a href="https://vk.com/{prefix}{obj_id}">{html.escape(title)}</a>' else:
return f"[{title}](https://vk.com/{prefix}{obj_id})" target = match.group("url")
return f"[{title}]({target})"
return _VK_LINK_RE.sub(_replace_plain, text)
def _replace_url(match: re.Match) -> str: out: list[str] = []
url = match.group(1) last_end = 0
title = match.group(2) for match in _VK_LINK_RE.finditer(text):
if parse_mode == "html": out.append(html.escape(text[last_end:match.start()]))
return f'<a href="{html.escape(url, quote=True)}">{html.escape(title)}</a>' title = html.escape(match.group("title"))
return f"[{title}]({url})" if match.group("prefix"):
href = f"https://vk.com/{match.group('prefix')}{match.group('obj_id')}"
else:
href = html.escape(match.group("url"), quote=True)
out.append(f'<a href="{href}">{title}</a>')
last_end = match.end()
out.append(html.escape(text[last_end:]))
return "".join(out)
text = re.sub(r"\[(club|id|public|event)(\d+)\|([^\]]+)\]", _replace_vk, text)
text = re.sub(r"\[(https?://[^\s\|]+)\|([^\]]+)\]", _replace_url, text) _ALLOWED_WRAPPER_TAGS = {"b", "strong", "i", "em", "u", "s", "code", "a"}
return text
class _SafeHTMLNormalizer(HTMLParser):
"""Normalizes admin-supplied header/footer/tags text: keeps a small safe
subset of inline HTML tags (with href-scheme validation for <a>) and
HTML-escapes everything else, instead of the previous all-or-nothing
'"<" not in text' heuristic."""
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.out: list[str] = []
self.open_tags: list[str] = []
def handle_starttag(self, tag: str, attrs: list[tuple[str, Optional[str]]]) -> None:
if tag not in _ALLOWED_WRAPPER_TAGS:
return
if tag == "a":
href = next((v for k, v in attrs if k == "href" and v), "")
parsed = urlparse(href)
if parsed.scheme not in ("http", "https") or not parsed.netloc:
return
self.out.append(f'<a href="{html.escape(href, quote=True)}">')
else:
self.out.append(f"<{tag}>")
self.open_tags.append(tag)
def handle_startendtag(self, tag: str, attrs: list[tuple[str, Optional[str]]]) -> None:
if tag == "br":
self.out.append("\n")
def handle_endtag(self, tag: str) -> None:
if tag == "br":
self.out.append("\n")
return
if tag in self.open_tags:
while self.open_tags:
t = self.open_tags.pop()
self.out.append(f"</{t}>")
if t == tag:
break
def handle_data(self, data: str) -> None:
self.out.append(html.escape(data))
def normalized(self) -> str:
while self.open_tags:
self.out.append(f"</{self.open_tags.pop()}>")
return "".join(self.out)
def normalize_wrapper_text(text: str, parse_mode: str = "html") -> str:
"""Safely prepares admin-configured header/footer/tags text for sending."""
text = str(text or "").strip()
if not text:
return ""
if parse_mode != "html":
return text
parser = _SafeHTMLNormalizer()
parser.feed(text)
parser.close()
return parser.normalized()
def build_media_unavailable_note(link_only_items: list, parse_mode: str = "html") -> str:
"""Builds a short note listing media that couldn't be attached (too big,
too long, private, download failed, etc.) with a link to the original,
so readers aren't left with no idea that media was omitted."""
if not link_only_items:
return ""
lines = []
for item in link_only_items:
url = str(getattr(item, "original_url", "") or "").strip()
media_type = str(getattr(item, "media_type", "медиа") or "медиа")
if not url:
continue
if parse_mode == "html":
lines.append(f'- {html.escape(media_type)}: <a href="{html.escape(url, quote=True)}">ссылка</a>')
else:
lines.append(f"- {media_type}: {url}")
if not lines:
return ""
header = "Не удалось прикрепить медиа, оригинал:"
if parse_mode == "html":
header = f"<i>{html.escape(header)}</i>"
return header + "\n" + "\n".join(lines)
def format_post_text( def format_post_text(
@@ -82,9 +186,9 @@ def format_post_text(
bold_first_line: bool = True, bold_first_line: bool = True,
vk_url: Optional[str] = None, vk_url: Optional[str] = None,
) -> str: ) -> str:
header = (settings.header_text if header is None else header).strip() header = normalize_wrapper_text(settings.header_text if header is None else header, parse_mode)
footer = (settings.footer_text if footer is None else footer).strip() footer = normalize_wrapper_text(settings.footer_text if footer is None else footer, parse_mode)
tags = (settings.common_tags if tags is None else tags).strip() tags = normalize_wrapper_text(settings.common_tags if tags is None else tags, parse_mode)
body = strip_trailing_hashtags(raw_text) body = strip_trailing_hashtags(raw_text)
lines = clean_dividers(body.splitlines()) lines = clean_dividers(body.splitlines())
@@ -97,10 +201,9 @@ def format_post_text(
formatted_lines: list[str] = [] formatted_lines: list[str] = []
for idx, line in enumerate(lines): for idx, line in enumerate(lines):
cleaned_line = clean_vk_wiki_links(line, parse_mode=parse_mode) if parse_mode == "html" else line cleaned_line = clean_vk_wiki_links(line, parse_mode=parse_mode)
if idx == title_idx and bold_first_line and cleaned_line.strip(): if idx == title_idx and bold_first_line and cleaned_line.strip():
if parse_mode == "html": if parse_mode == "html":
# If clean_vk_wiki_links was run, keep existing <a> tags safe
formatted_lines.append(f"<b>{cleaned_line.strip()}</b>") formatted_lines.append(f"<b>{cleaned_line.strip()}</b>")
else: else:
formatted_lines.append(f"**{cleaned_line.strip()}**") formatted_lines.append(f"**{cleaned_line.strip()}**")
@@ -115,12 +218,12 @@ def format_post_text(
parts: list[str] = [] parts: list[str] = []
if header: if header:
parts.append(html.escape(header) if parse_mode == "html" and "<" not in header else header) parts.append(header)
if normalized_body: if normalized_body:
parts.append(normalized_body) parts.append(normalized_body)
if footer: if footer:
parts.append(html.escape(footer) if parse_mode == "html" and "<" not in footer else footer) parts.append(footer)
if tags: if tags:
parts.append(html.escape(tags) if parse_mode == "html" and "<" not in tags else tags) parts.append(tags)
return "\n\n".join(part for part in parts if part).strip() return "\n\n".join(part for part in parts if part).strip()
+46 -35
View File
@@ -2,8 +2,6 @@ from __future__ import annotations
from loguru import logger from loguru import logger
import asyncio import asyncio
import os
import re
from pathlib import Path from pathlib import Path
from typing import Any, Optional from typing import Any, Optional
import aiohttp import aiohttp
@@ -15,11 +13,11 @@ from aiogram.types import FSInputFile, InputMediaPhoto, InputMediaVideo
try: try:
from .config import settings from .config import settings
from .media_processor import ProcessedMedia from .media_processor import ProcessedMedia
from .text_formatter import format_post_text, split_message_chunks from .text_formatter import build_media_unavailable_note, format_post_text, split_message_chunks
except (ImportError, ValueError): except (ImportError, ValueError):
from config import settings from config import settings
from media_processor import ProcessedMedia from media_processor import ProcessedMedia
from text_formatter import format_post_text, split_message_chunks from text_formatter import build_media_unavailable_note, format_post_text, split_message_chunks
MAX_MEDIA_GROUP = 10 MAX_MEDIA_GROUP = 10
MAX_RICH_MEDIA = 50 MAX_RICH_MEDIA = 50
@@ -175,8 +173,8 @@ class TelegramPoster:
def build_rich_text_html(self, text: str) -> str: def build_rich_text_html(self, text: str) -> str:
paragraphs = [] paragraphs = []
for p in text.strip().split("\n\n"): for paragraph in str(text or "").strip().split("\n\n"):
body = "<br/>".join(line for line in p.splitlines() if line.strip()) body = "<br/>".join(line for line in paragraph.splitlines() if line.strip())
if body: if body:
paragraphs.append(f"<p>{body}</p>") paragraphs.append(f"<p>{body}</p>")
return "\n".join(paragraphs) return "\n".join(paragraphs)
@@ -184,18 +182,18 @@ class TelegramPoster:
def build_rich_message( def build_rich_message(
self, text: str, media_items: list[ProcessedMedia], file_ids: dict[str, str] self, text: str, media_items: list[ProcessedMedia], file_ids: dict[str, str]
) -> Optional[dict[str, Any]]: ) -> Optional[dict[str, Any]]:
rich_media = [] """Ported from new_vk_parser's tg_poster.py: bails to None (legacy fallback)
media_tags = [] if any media item lacks a file_id, since a rich-message collage can't
partially reference missing media."""
valid_items = [ if len(media_items) > MAX_RICH_MEDIA:
m for m in media_items if m.attachment_id in file_ids
][:MAX_RICH_MEDIA]
if not valid_items:
return None return None
for idx, item in enumerate(valid_items): rich_media = []
f_id = file_ids[item.attachment_id] media_tags = []
for idx, item in enumerate(media_items[:MAX_RICH_MEDIA]):
f_id = file_ids.get(item.attachment_id)
if not f_id:
return None
media_id = f"m{idx}" media_id = f"m{idx}"
m_type = "photo" if item.media_type == "photo" else "video" m_type = "photo" if item.media_type == "photo" else "video"
rich_media.append({"id": media_id, "media": {"type": m_type, "media": f_id}}) rich_media.append({"id": media_id, "media": {"type": m_type, "media": f_id}})
@@ -204,6 +202,9 @@ class TelegramPoster:
else: else:
media_tags.append(f'<video src="tg://video?id={media_id}"></video>') media_tags.append(f'<video src="tg://video?id={media_id}"></video>')
if not rich_media:
return None
rich_text = self.build_rich_text_html(text) rich_text = self.build_rich_text_html(text)
if len(rich_text) > MAX_RICH_TEXT: if len(rich_text) > MAX_RICH_TEXT:
return None return None
@@ -224,9 +225,10 @@ class TelegramPoster:
if self.thread_id: if self.thread_id:
data["message_thread_id"] = int(self.thread_id) data["message_thread_id"] = int(self.thread_id)
base_url = (settings.local_bot_api_url or "").strip().rstrip("/") or "https://api.telegram.org" # Always the cloud endpoint: this call only references already-uploaded
url = f"{base_url}/bot{settings.tg_bot_token}/sendRichMessage" # file_ids, no raw bytes cross the wire, so routing via the local Bot API
# server buys nothing here.
url = f"https://api.telegram.org/bot{settings.tg_bot_token}/sendRichMessage"
timeout = aiohttp.ClientTimeout(total=90) timeout = aiohttp.ClientTimeout(total=90)
async with aiohttp.ClientSession(timeout=timeout) as session: async with aiohttp.ClientSession(timeout=timeout) as session:
async with session.post(url, json=data) as resp: async with session.post(url, json=data) as resp:
@@ -237,12 +239,17 @@ class TelegramPoster:
mid = res.get("message_id") mid = res.get("message_id")
if mid: if mid:
return [int(mid)] return [int(mid)]
raise RichMessageUnavailable("sendRichMessage ok but no message_id") raise RichMessageUnavailable("sendRichMessage returned no message_id")
desc = str(payload.get("description") or f"HTTP {resp.status}") description = str(payload.get("description") or f"HTTP {resp.status}")
raise RichMessageUnavailable(desc) if "Too Many Requests" in description and isinstance(payload.get("parameters"), dict):
retry_after = float(payload["parameters"].get("retry_after") or 0)
if retry_after > 0:
logger.warning("Telegram rich message flood control, sleep {}s", retry_after)
await asyncio.sleep(retry_after + 0.5)
raise RichMessageUnavailable(description)
async def send_legacy_media_post( async def send_media_post(
self, text: str, media_items: list[ProcessedMedia], file_ids: dict[str, str] self, text: str, media_items: list[ProcessedMedia], file_ids: dict[str, str]
) -> list[int]: ) -> list[int]:
if not self.bot: if not self.bot:
@@ -366,10 +373,11 @@ class TelegramPoster:
) -> tuple[list[int], Optional[str]]: ) -> tuple[list[int], Optional[str]]:
""" """
Main Telegram posting routine: Main Telegram posting routine:
1. Formats text for HTML parse mode. 1. Formats text for HTML parse mode and notes any media that couldn't be attached.
2. Uploads media (or storage channel if configured) to get file_ids. 2. Uploads media to the storage channel (if configured) to obtain reusable file_ids.
3. Tries sendRichMessage first. 3. Tries sendRichMessage (Bot API 10.1+) for a proper collage + rich text.
4. If unavailable, falls back to legacy media groups / single media / text. 4. Falls back to standard aiogram calls (send_photo/send_video/send_media_group)
if rich message is unavailable (no storage channel, old Bot API server, etc).
""" """
formatted_text = format_post_text( formatted_text = format_post_text(
raw_text, raw_text,
@@ -379,26 +387,29 @@ class TelegramPoster:
) )
valid_media = [m for m in media_items if not m.is_link_only and m.local_path] valid_media = [m for m in media_items if not m.is_link_only and m.local_path]
link_only = [m for m in media_items if m.is_link_only]
note = build_media_unavailable_note(link_only, parse_mode="html")
if note:
formatted_text = f"{formatted_text}\n\n{note}" if formatted_text else note
# 1. Obtain file_ids if we have storage channel or if we want rich message # Obtain file_ids from the storage channel, if configured, to avoid re-uploading
# (also a prerequisite for rich messages, which reference media by file_id).
file_ids: dict[str, str] = {} file_ids: dict[str, str] = {}
if valid_media and self.storage_chat_id: if valid_media and self.storage_chat_id:
file_ids = await self.upload_media_for_file_ids(valid_media) file_ids = await self.upload_media_for_file_ids(valid_media)
# 2. Try Rich Message if file_ids are available if valid_media and file_ids:
if file_ids:
rich_msg = self.build_rich_message(formatted_text, valid_media, file_ids) rich_msg = self.build_rich_message(formatted_text, valid_media, file_ids)
if rich_msg: if rich_msg:
try: try:
mids = await self.send_rich_message(rich_msg) mids = await self.send_rich_message(rich_msg)
url = tg_message_url(self.chat_id, mids[0]) if mids else None url = tg_message_url(self.chat_id, mids[0]) if mids else None
logger.info("Sent Telegram Rich Message: {}", mids) logger.info("Sent Telegram rich message: {}", mids)
return mids, url return mids, url
except RichMessageUnavailable as exc: except RichMessageUnavailable as exc:
logger.warning("Telegram sendRichMessage failed: {}. Falling back to legacy.", exc) logger.warning("Telegram sendRichMessage failed: {}. Falling back to standard send.", exc)
# 3. Fallback to legacy media group / text mids = await self.send_media_post(formatted_text, valid_media, file_ids)
mids = await self.send_legacy_media_post(formatted_text, valid_media, file_ids)
url = tg_message_url(self.chat_id, mids[0]) if mids else None url = tg_message_url(self.chat_id, mids[0]) if mids else None
logger.info("Sent Telegram Legacy Message: {}", mids) logger.info("Sent Telegram message: {}", mids)
return mids, url return mids, url
+19 -3
View File
@@ -6,6 +6,7 @@ import re
import time import time
from dataclasses import dataclass from dataclasses import dataclass
from typing import Any, Optional from typing import Any, Optional
from urllib.parse import urlparse
import aiohttp import aiohttp
try: try:
from .config import settings from .config import settings
@@ -54,7 +55,6 @@ class VKPost:
text: str text: str
media: list[VKMediaItem] media: list[VKMediaItem]
raw: dict[str, Any] raw: dict[str, Any]
is_pinned: bool = False
is_repost: bool = False is_repost: bool = False
@@ -200,6 +200,24 @@ class VKClient:
title=video.get("title"), title=video.get("title"),
) )
) )
elif att_type == "link":
link = att.get("link") or {}
link_url = str(link.get("url") or "")
host = urlparse(link_url).hostname or ""
host = host.lower().removeprefix("www.").removeprefix("m.")
if host in ("youtube.com", "youtube-nocookie.com", "youtu.be"):
items.append(
VKMediaItem(
media_type="video",
url=link_url,
attachment_id=f"link_{abs(hash(link_url))}",
title=str(link.get("title") or ""),
)
)
else:
logger.debug("Skipping unsupported VK link attachment: {}", link_url)
elif att_type not in ("photo", "video"):
logger.debug("Skipping unsupported VK attachment type: {}", att_type)
return items return items
async def get_latest_posts(self, owner_id: int, count: int = 10) -> list[VKPost]: async def get_latest_posts(self, owner_id: int, count: int = 10) -> list[VKPost]:
@@ -209,7 +227,6 @@ class VKClient:
for raw in items: for raw in items:
if raw.get("is_deleted") or not raw.get("id") or not raw.get("date"): if raw.get("is_deleted") or not raw.get("id") or not raw.get("date"):
continue continue
is_pinned = bool(raw.get("is_pinned"))
is_repost = bool(raw.get("copy_history")) is_repost = bool(raw.get("copy_history"))
media = self.extract_media(raw) media = self.extract_media(raw)
posts.append( posts.append(
@@ -220,7 +237,6 @@ class VKClient:
text=str(raw.get("text") or ""), text=str(raw.get("text") or ""),
media=media, media=media,
raw=raw, raw=raw,
is_pinned=is_pinned,
is_repost=is_repost, is_repost=is_repost,
) )
) )