diff --git a/src/bridge.py b/src/bridge.py index dda8b01..b3ae02a 100644 --- a/src/bridge.py +++ b/src/bridge.py @@ -110,7 +110,8 @@ class MaxToTelegramBridge: ) return - if parsed.text.strip() and total_media == 0: + should_send_plain_text = total_media == 0 and not parsed.file_urls and bool(text.strip()) + if should_send_plain_text: target_chat_id, sent = await self._send_with_migration_retry( target_chat_id=target_chat_id, max_chat_title_norm=normalized, @@ -200,12 +201,13 @@ class MaxToTelegramBridge: if not sent_any: # Последняя страховка: гарантируем уведомление в Telegram даже для пустых/неизвестных payload. + fallback_text = text.strip() or self._build_fallback_unknown_notice(parsed) target_chat_id, sent = await self._send_with_migration_retry( target_chat_id=target_chat_id, max_chat_title_norm=normalized, max_chat_title=parsed.chat_name, send_action=lambda chat_id: self._telegram.send_text( - chat_id, self._build_fallback_unknown_notice(parsed), reply_to_message_id=reply_telegram_mid + chat_id, fallback_text, reply_to_message_id=reply_telegram_mid ), ) mid = sent.get("result", {}).get("message_id") if isinstance(sent.get("result"), dict) else None @@ -359,7 +361,8 @@ class MaxToTelegramBridge: if urls: parsed.file_urls.extend(urls) else: - parsed.unknown_attachments.append(type(attach).__name__) + if not self._is_forward_attach_like(attach): + parsed.unknown_attachments.append(type(attach).__name__) # Убираем дубли URL, если парсер и enrich нашли одинаковые вложения. parsed.image_urls = list(dict.fromkeys(parsed.image_urls)) @@ -445,6 +448,17 @@ class MaxToTelegramBridge: walk(node) return list(dict.fromkeys(urls)) + @staticmethod + def _is_forward_attach_like(attach: Any) -> bool: + name = type(attach).__name__.lower() + if "forward" in name or "share" in name or "quote" in name: + return True + if hasattr(attach, "__dict__"): + keys = {str(k).lower() for k in vars(attach).keys()} + if {"forward", "forwarded", "forwards", "link", "message", "messages", "origin", "payload"} & keys: + return True + return False + @staticmethod def _append_unknown_attachment_notice(*, parsed: ParsedMessage, text: str) -> str: if not parsed.unknown_attachments: diff --git a/src/max_parser.py b/src/max_parser.py index 0b9f5c5..0f7357d 100644 --- a/src/max_parser.py +++ b/src/max_parser.py @@ -36,6 +36,107 @@ def _is_video(media_type: str) -> bool: return "video" in value or value in {"mp4", "mov", "mkv", "avi"} +def _is_forward_like(data: dict[str, Any]) -> bool: + media_type = _stringify(data.get("type") or data.get("media_type") or data.get("kind")).lower() + if "forward" in media_type or "share" in media_type or "quote" in media_type: + return True + forward_keys = { + "forward", + "forwarded", + "forwards", + "link", + "message", + "messages", + "payload", + "quote", + "origin", + } + return any(key in data for key in forward_keys) + + +def _classify_url(url: str, media_type: str) -> str: + lowered = url.lower() + if _is_image(media_type) or lowered.endswith((".jpg", ".jpeg", ".png", ".webp", ".gif")): + return "image" + if _is_video(media_type) or lowered.endswith((".mp4", ".mov", ".mkv", ".avi", ".webm")): + return "video" + return "file" + + +def _collect_urls(node: Any) -> list[str]: + urls: list[str] = [] + seen_ids: set[int] = set() + + def walk(value: Any) -> None: + if value is None: + return + obj_id = id(value) + if obj_id in seen_ids: + return + seen_ids.add(obj_id) + + if isinstance(value, str): + if value.startswith("http://") or value.startswith("https://"): + urls.append(value.strip()) + return + if isinstance(value, (list, tuple, set)): + for item in value: + walk(item) + return + if isinstance(value, dict): + for nested in value.values(): + walk(nested) + return + if hasattr(value, "__dict__"): + walk(vars(value)) + + walk(node) + return list(dict.fromkeys(urls)) + + +def _extract_forwarded_text(message: Any) -> str: + candidates: list[str] = [] + seen_ids: set[int] = set() + forward_keys = {"forward", "forwarded", "forwards", "link", "message", "messages", "payload", "quote", "origin"} + text_keys = {"text", "message", "body", "caption"} + + def walk(value: Any, inside_forward: bool) -> None: + if value is None: + return + obj_id = id(value) + if obj_id in seen_ids: + return + seen_ids.add(obj_id) + + if isinstance(value, str): + if inside_forward: + text = value.strip() + if text and not (text.startswith("http://") or text.startswith("https://")): + candidates.append(text) + return + if isinstance(value, (list, tuple, set)): + for item in value: + walk(item, inside_forward) + return + if isinstance(value, dict): + current_is_forward = inside_forward or any(key in value for key in forward_keys) + for key, nested in value.items(): + if key in text_keys and current_is_forward and isinstance(nested, str): + text = nested.strip() + if text and not (text.startswith("http://") or text.startswith("https://")): + candidates.append(text) + walk(nested, current_is_forward) + return + if hasattr(value, "__dict__"): + walk(vars(value), inside_forward) + + walk(message, False) + uniq = list(dict.fromkeys(candidates)) + if not uniq: + return "" + return "\n\n".join(uniq[:5]) + + def _extract_media_urls(message: Any) -> tuple[list[str], list[str], list[str], list[str]]: image_urls: list[str] = [] video_urls: list[str] = [] @@ -47,6 +148,7 @@ def _extract_media_urls(message: Any) -> tuple[list[str], list[str], list[str], for item in raw_attachments: data = _as_dict(item) media_type = _stringify(data.get("type") or data.get("media_type") or data.get("kind")) + is_forward_like = _is_forward_like(data) url = _stringify( data.get("base_url") or data.get("url") @@ -69,6 +171,23 @@ def _extract_media_urls(message: Any) -> tuple[list[str], list[str], list[str], media_type = _stringify(nested_data.get("type") or nested_data.get("media_type")) if not url: + nested_urls = _collect_urls(item) + for nested_url in nested_urls: + kind = _classify_url(nested_url, media_type) + if kind == "image": + image_urls.append(nested_url) + elif kind == "video": + video_urls.append(nested_url) + else: + file_urls.append(nested_url) + if nested_urls: + continue + + if not url: + if is_forward_like: + # Forward-пакет может не содержать прямого URL в верхнем уровне; + # текст/медиа достанем рекурсивно в других этапах. + continue kind = media_type or _stringify(type(item).__name__) or "unknown" unknown_attachments.append(kind) continue @@ -120,6 +239,11 @@ def parse_message(message: Any) -> ParsedMessage: message_id = _stringify(_get_attr(message, ["id", "message_id", "mid"])) or "unknown-id" chat_id = _stringify(_get_attr(message, ["chat_id", "dialog_id", "peer_id"])) or "unknown-chat" text = _stringify(_get_attr(message, ["text", "message", "body"])) + forwarded_text = _extract_forwarded_text(message) + if text and forwarded_text and forwarded_text != text: + text = f"{text}\n\n[forwarded]\n{forwarded_text}" + elif not text and forwarded_text: + text = forwarded_text image_urls, video_urls, file_urls, unknown_attachments = _extract_media_urls(message) reply_mid, reply_preview = _extract_max_reply(message)