Add lightweight web control panel
This commit is contained in:
+227
@@ -0,0 +1,227 @@
|
|||||||
|
data/
|
||||||
|
.codex
|
||||||
|
.python-version
|
||||||
|
uv.lock
|
||||||
|
.uv-cache/
|
||||||
|
|
||||||
|
|
||||||
|
# Byte-compiled / optimized / DLL files
|
||||||
|
__pycache__/
|
||||||
|
*.py[codz]
|
||||||
|
*$py.class
|
||||||
|
|
||||||
|
# C extensions
|
||||||
|
*.so
|
||||||
|
|
||||||
|
# Distribution / packaging
|
||||||
|
.Python
|
||||||
|
build/
|
||||||
|
develop-eggs/
|
||||||
|
dist/
|
||||||
|
downloads/
|
||||||
|
eggs/
|
||||||
|
.eggs/
|
||||||
|
lib/
|
||||||
|
lib64/
|
||||||
|
parts/
|
||||||
|
sdist/
|
||||||
|
var/
|
||||||
|
wheels/
|
||||||
|
share/python-wheels/
|
||||||
|
*.egg-info/
|
||||||
|
.installed.cfg
|
||||||
|
*.egg
|
||||||
|
MANIFEST
|
||||||
|
|
||||||
|
# PyInstaller
|
||||||
|
# Usually these files are written by a python script from a template
|
||||||
|
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||||
|
*.manifest
|
||||||
|
*.spec
|
||||||
|
|
||||||
|
# Installer logs
|
||||||
|
pip-log.txt
|
||||||
|
pip-delete-this-directory.txt
|
||||||
|
|
||||||
|
# Unit test / coverage reports
|
||||||
|
htmlcov/
|
||||||
|
.tox/
|
||||||
|
.nox/
|
||||||
|
.coverage
|
||||||
|
.coverage.*
|
||||||
|
.cache
|
||||||
|
nosetests.xml
|
||||||
|
coverage.xml
|
||||||
|
*.cover
|
||||||
|
*.py.cover
|
||||||
|
*.lcov
|
||||||
|
.hypothesis/
|
||||||
|
.pytest_cache/
|
||||||
|
cover/
|
||||||
|
|
||||||
|
# Translations
|
||||||
|
*.mo
|
||||||
|
*.pot
|
||||||
|
|
||||||
|
# Django stuff:
|
||||||
|
*.log
|
||||||
|
local_settings.py
|
||||||
|
db.sqlite3
|
||||||
|
db.sqlite3-journal
|
||||||
|
|
||||||
|
# Flask stuff:
|
||||||
|
instance/
|
||||||
|
.webassets-cache
|
||||||
|
|
||||||
|
# Scrapy stuff:
|
||||||
|
.scrapy
|
||||||
|
|
||||||
|
# Sphinx documentation
|
||||||
|
docs/_build/
|
||||||
|
|
||||||
|
# PyBuilder
|
||||||
|
.pybuilder/
|
||||||
|
target/
|
||||||
|
|
||||||
|
# Jupyter Notebook
|
||||||
|
.ipynb_checkpoints
|
||||||
|
|
||||||
|
# IPython
|
||||||
|
profile_default/
|
||||||
|
ipython_config.py
|
||||||
|
|
||||||
|
# pyenv
|
||||||
|
# For a library or package, you might want to ignore these files since the code is
|
||||||
|
# intended to run in multiple environments; otherwise, check them in:
|
||||||
|
# .python-version
|
||||||
|
|
||||||
|
# pipenv
|
||||||
|
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
||||||
|
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
||||||
|
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
||||||
|
# install all needed dependencies.
|
||||||
|
# Pipfile.lock
|
||||||
|
|
||||||
|
# UV
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
|
||||||
|
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||||
|
# commonly ignored for libraries.
|
||||||
|
# uv.lock
|
||||||
|
|
||||||
|
# poetry
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
||||||
|
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||||
|
# commonly ignored for libraries.
|
||||||
|
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
||||||
|
# poetry.lock
|
||||||
|
# poetry.toml
|
||||||
|
|
||||||
|
# pdm
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
||||||
|
# pdm recommends including project-wide configuration in pdm.toml, but excluding .pdm-python.
|
||||||
|
# https://pdm-project.org/en/latest/usage/project/#working-with-version-control
|
||||||
|
# pdm.lock
|
||||||
|
# pdm.toml
|
||||||
|
.pdm-python
|
||||||
|
.pdm-build/
|
||||||
|
|
||||||
|
# pixi
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include pixi.lock in version control.
|
||||||
|
# pixi.lock
|
||||||
|
# Pixi creates a virtual environment in the .pixi directory, just like venv module creates one
|
||||||
|
# in the .venv directory. It is recommended not to include this directory in version control.
|
||||||
|
.pixi/*
|
||||||
|
!.pixi/config.toml
|
||||||
|
|
||||||
|
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
||||||
|
__pypackages__/
|
||||||
|
|
||||||
|
# Celery stuff
|
||||||
|
celerybeat-schedule*
|
||||||
|
celerybeat.pid
|
||||||
|
|
||||||
|
# Redis
|
||||||
|
*.rdb
|
||||||
|
*.aof
|
||||||
|
*.pid
|
||||||
|
|
||||||
|
# RabbitMQ
|
||||||
|
mnesia/
|
||||||
|
rabbitmq/
|
||||||
|
rabbitmq-data/
|
||||||
|
|
||||||
|
# ActiveMQ
|
||||||
|
activemq-data/
|
||||||
|
|
||||||
|
# SageMath parsed files
|
||||||
|
*.sage.py
|
||||||
|
|
||||||
|
# Environments
|
||||||
|
.env
|
||||||
|
.envrc
|
||||||
|
.venv
|
||||||
|
env/
|
||||||
|
venv/
|
||||||
|
ENV/
|
||||||
|
env.bak/
|
||||||
|
venv.bak/
|
||||||
|
|
||||||
|
# Spyder project settings
|
||||||
|
.spyderproject
|
||||||
|
.spyproject
|
||||||
|
|
||||||
|
# Rope project settings
|
||||||
|
.ropeproject
|
||||||
|
|
||||||
|
# mkdocs documentation
|
||||||
|
/site
|
||||||
|
|
||||||
|
# mypy
|
||||||
|
.mypy_cache/
|
||||||
|
.dmypy.json
|
||||||
|
dmypy.json
|
||||||
|
|
||||||
|
# Pyre type checker
|
||||||
|
.pyre/
|
||||||
|
|
||||||
|
# pytype static type analyzer
|
||||||
|
.pytype/
|
||||||
|
|
||||||
|
# Cython debug symbols
|
||||||
|
cython_debug/
|
||||||
|
|
||||||
|
# PyCharm
|
||||||
|
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
||||||
|
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
||||||
|
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
||||||
|
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
||||||
|
# .idea/
|
||||||
|
|
||||||
|
# Abstra
|
||||||
|
# Abstra is an AI-powered process automation framework.
|
||||||
|
# Ignore directories containing user credentials, local state, and settings.
|
||||||
|
# Learn more at https://abstra.io/docs
|
||||||
|
.abstra/
|
||||||
|
|
||||||
|
# Visual Studio Code
|
||||||
|
# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
|
||||||
|
# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
|
||||||
|
# and can be added to the global gitignore or merged into this file. However, if you prefer,
|
||||||
|
# you could uncomment the following to ignore the entire vscode folder
|
||||||
|
# .vscode/
|
||||||
|
# Temporary file for partial code execution
|
||||||
|
tempCodeRunnerFile.py
|
||||||
|
|
||||||
|
# Ruff stuff:
|
||||||
|
.ruff_cache/
|
||||||
|
|
||||||
|
# PyPI configuration file
|
||||||
|
.pypirc
|
||||||
|
|
||||||
|
# Marimo
|
||||||
|
marimo/_static/
|
||||||
|
marimo/_lsp/
|
||||||
|
__marimo__/
|
||||||
|
|
||||||
|
# Streamlit
|
||||||
|
.streamlit/secrets.toml
|
||||||
+18
@@ -0,0 +1,18 @@
|
|||||||
|
FROM python:3.11-slim
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
|
||||||
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
gcc \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
COPY requirements.txt .
|
||||||
|
RUN pip install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
|
COPY . .
|
||||||
|
|
||||||
|
EXPOSE 8080
|
||||||
|
VOLUME ["/app/data"]
|
||||||
|
VOLUME ["/app/session"]
|
||||||
|
|
||||||
|
CMD ["python", "main.py"]
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
services:
|
||||||
|
telegram-scraper:
|
||||||
|
build:
|
||||||
|
context: .
|
||||||
|
dockerfile: Dockerfile
|
||||||
|
container_name: telegram-scraper
|
||||||
|
user: 1000:1000
|
||||||
|
restart: unless-stopped
|
||||||
|
tty: true
|
||||||
|
stdin_open: true
|
||||||
|
ports:
|
||||||
|
- "8080:8080"
|
||||||
|
volumes:
|
||||||
|
- ./data:/app/data
|
||||||
|
- session:/app/session
|
||||||
|
volumes:
|
||||||
|
session:
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
from webui_server import run_server
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
run_server()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
[project]
|
||||||
|
name = "telegram-scraper"
|
||||||
|
version = "0.1.0"
|
||||||
|
description = "Telegram-scraper with webui"
|
||||||
|
readme = "README.md"
|
||||||
|
requires-python = ">=3.11"
|
||||||
|
dependencies = [
|
||||||
|
"aiohappyeyeballs==2.6.1",
|
||||||
|
"aiohttp==3.12.14",
|
||||||
|
"aiosignal==1.4.0",
|
||||||
|
"asyncio==3.4.3",
|
||||||
|
"attrs==25.3.0",
|
||||||
|
"frozenlist==1.7.0",
|
||||||
|
"idna==3.10",
|
||||||
|
"multidict==6.6.3",
|
||||||
|
"propcache==0.3.2",
|
||||||
|
"pyaes==1.6.1",
|
||||||
|
"pyasn1==0.6.1",
|
||||||
|
"qrcode==8.0",
|
||||||
|
"rsa==4.9.1",
|
||||||
|
"telethon==1.40.0",
|
||||||
|
"yarl==1.20.1",
|
||||||
|
]
|
||||||
+177
@@ -0,0 +1,177 @@
|
|||||||
|
const state = {
|
||||||
|
dashboard: null,
|
||||||
|
};
|
||||||
|
|
||||||
|
async function api(path, options = {}) {
|
||||||
|
const response = await fetch(path, {
|
||||||
|
headers: { "Content-Type": "application/json" },
|
||||||
|
...options,
|
||||||
|
});
|
||||||
|
const data = await response.json();
|
||||||
|
if (!response.ok) {
|
||||||
|
throw new Error(data.error || "Request failed");
|
||||||
|
}
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
function formatDate(value) {
|
||||||
|
if (!value) return "—";
|
||||||
|
return value.replace("T", " ").replace("+00:00", " UTC");
|
||||||
|
}
|
||||||
|
|
||||||
|
function setBusy(button, busy) {
|
||||||
|
if (!button) return;
|
||||||
|
button.disabled = busy;
|
||||||
|
}
|
||||||
|
|
||||||
|
function renderAuth(auth) {
|
||||||
|
document.getElementById("auth-status").textContent = auth.status;
|
||||||
|
document.getElementById("auth-details").textContent = auth.details || "";
|
||||||
|
}
|
||||||
|
|
||||||
|
function renderSummary(data) {
|
||||||
|
document.getElementById("channel-count").textContent = String(data.state.channel_count);
|
||||||
|
document.getElementById("forwarding-count").textContent = String(data.state.forwarding_rules.length);
|
||||||
|
const toggle = document.getElementById("scrape-media-toggle");
|
||||||
|
toggle.checked = Boolean(data.state.scrape_media);
|
||||||
|
document.getElementById("scrape-media-label").textContent = toggle.checked ? "ON" : "OFF";
|
||||||
|
}
|
||||||
|
|
||||||
|
function renderChannels(channels) {
|
||||||
|
const tbody = document.getElementById("channels-table");
|
||||||
|
tbody.innerHTML = "";
|
||||||
|
const template = document.getElementById("channel-row-template");
|
||||||
|
|
||||||
|
channels.forEach((channel) => {
|
||||||
|
const node = template.content.firstElementChild.cloneNode(true);
|
||||||
|
node.querySelector(".channel-name").textContent = channel.name;
|
||||||
|
node.querySelector(".channel-id").textContent = channel.channel_id;
|
||||||
|
node.querySelector(".message-count").textContent = String(channel.message_count);
|
||||||
|
node.querySelector(".media-count").textContent = String(channel.media_count);
|
||||||
|
node.querySelector(".last-date").textContent = channel.last_date || "—";
|
||||||
|
|
||||||
|
node.querySelector(".scrape-btn").addEventListener("click", async () => {
|
||||||
|
await api("/api/jobs/scrape", {
|
||||||
|
method: "POST",
|
||||||
|
body: JSON.stringify({ channel_id: channel.channel_id }),
|
||||||
|
});
|
||||||
|
await refreshDashboard();
|
||||||
|
});
|
||||||
|
|
||||||
|
node.querySelector(".export-view-btn").addEventListener("click", () => {
|
||||||
|
window.location.href = `/viewer?channel=${encodeURIComponent(channel.channel_id)}`;
|
||||||
|
});
|
||||||
|
|
||||||
|
node.querySelector(".media-btn").addEventListener("click", async () => {
|
||||||
|
await api("/api/jobs/rescrape-media", {
|
||||||
|
method: "POST",
|
||||||
|
body: JSON.stringify({ channel_id: channel.channel_id }),
|
||||||
|
});
|
||||||
|
await refreshDashboard();
|
||||||
|
});
|
||||||
|
|
||||||
|
node.querySelector(".remove-btn").addEventListener("click", async () => {
|
||||||
|
await api("/api/channels/remove", {
|
||||||
|
method: "POST",
|
||||||
|
body: JSON.stringify({ channel_id: channel.channel_id }),
|
||||||
|
});
|
||||||
|
await refreshDashboard();
|
||||||
|
});
|
||||||
|
|
||||||
|
tbody.appendChild(node);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
function renderJobs(jobs) {
|
||||||
|
const root = document.getElementById("jobs-list");
|
||||||
|
root.innerHTML = "";
|
||||||
|
const template = document.getElementById("job-template");
|
||||||
|
|
||||||
|
if (!jobs.length) {
|
||||||
|
root.innerHTML = '<div class="muted">Пока задач нет.</div>';
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
jobs.forEach((job) => {
|
||||||
|
const node = template.content.firstElementChild.cloneNode(true);
|
||||||
|
node.querySelector(".job-title").textContent = job.title;
|
||||||
|
node.querySelector(".job-status").textContent = job.status;
|
||||||
|
node.querySelector(".job-time").textContent = [job.started_at, job.finished_at].filter(Boolean).join(" -> ");
|
||||||
|
node.querySelector(".job-logs").textContent = job.logs || job.error || "Без логов";
|
||||||
|
root.appendChild(node);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
async function refreshDashboard() {
|
||||||
|
const data = await api("/api/dashboard");
|
||||||
|
state.dashboard = data;
|
||||||
|
renderAuth(data.auth);
|
||||||
|
renderSummary(data);
|
||||||
|
renderChannels(data.channels);
|
||||||
|
renderJobs(data.jobs);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function main() {
|
||||||
|
document.getElementById("scrape-all-btn").addEventListener("click", async (event) => {
|
||||||
|
setBusy(event.currentTarget, true);
|
||||||
|
try {
|
||||||
|
await api("/api/jobs/scrape", { method: "POST", body: JSON.stringify({}) });
|
||||||
|
await refreshDashboard();
|
||||||
|
} finally {
|
||||||
|
setBusy(event.currentTarget, false);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
document.getElementById("export-all-btn").addEventListener("click", async (event) => {
|
||||||
|
setBusy(event.currentTarget, true);
|
||||||
|
try {
|
||||||
|
await api("/api/jobs/export", { method: "POST", body: JSON.stringify({}) });
|
||||||
|
await refreshDashboard();
|
||||||
|
} finally {
|
||||||
|
setBusy(event.currentTarget, false);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
document.getElementById("refresh-dialogs-btn").addEventListener("click", async (event) => {
|
||||||
|
setBusy(event.currentTarget, true);
|
||||||
|
try {
|
||||||
|
await api("/api/jobs/refresh-dialogs", { method: "POST", body: JSON.stringify({}) });
|
||||||
|
await refreshDashboard();
|
||||||
|
} finally {
|
||||||
|
setBusy(event.currentTarget, false);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
document.getElementById("refresh-jobs-btn").addEventListener("click", refreshDashboard);
|
||||||
|
|
||||||
|
document.getElementById("scrape-media-toggle").addEventListener("change", async (event) => {
|
||||||
|
const checked = event.currentTarget.checked;
|
||||||
|
document.getElementById("scrape-media-label").textContent = checked ? "ON" : "OFF";
|
||||||
|
await api("/api/settings/media", {
|
||||||
|
method: "POST",
|
||||||
|
body: JSON.stringify({ value: checked }),
|
||||||
|
});
|
||||||
|
await refreshDashboard();
|
||||||
|
});
|
||||||
|
|
||||||
|
document.getElementById("add-channel-form").addEventListener("submit", async (event) => {
|
||||||
|
event.preventDefault();
|
||||||
|
const channelId = document.getElementById("add-channel-id").value.trim();
|
||||||
|
const name = document.getElementById("add-channel-name").value.trim();
|
||||||
|
if (!channelId) return;
|
||||||
|
await api("/api/channels/add", {
|
||||||
|
method: "POST",
|
||||||
|
body: JSON.stringify({ channel_id: channelId, name }),
|
||||||
|
});
|
||||||
|
event.currentTarget.reset();
|
||||||
|
await refreshDashboard();
|
||||||
|
});
|
||||||
|
|
||||||
|
await refreshDashboard();
|
||||||
|
window.setInterval(refreshDashboard, 5000);
|
||||||
|
}
|
||||||
|
|
||||||
|
main().catch((error) => {
|
||||||
|
console.error(error);
|
||||||
|
alert(error.message);
|
||||||
|
});
|
||||||
@@ -0,0 +1,131 @@
|
|||||||
|
<!DOCTYPE html>
|
||||||
|
<html lang="ru">
|
||||||
|
<head>
|
||||||
|
<meta charset="UTF-8" />
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||||
|
<title>Telegram Scraper Control Panel</title>
|
||||||
|
<link rel="stylesheet" href="/static/style.css" />
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class="app-shell">
|
||||||
|
<aside class="sidebar">
|
||||||
|
<div class="brand-block">
|
||||||
|
<div class="eyebrow">Telegram Scraper</div>
|
||||||
|
<h1>Control Panel</h1>
|
||||||
|
<p class="muted">Обычная админка для сервера: состояния, каналы, фоновые задачи.</p>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<nav class="nav-links">
|
||||||
|
<a class="nav-link active" href="/">Панель</a>
|
||||||
|
<a class="nav-link" href="/viewer">Просмотр сообщений</a>
|
||||||
|
</nav>
|
||||||
|
|
||||||
|
<section class="status-card">
|
||||||
|
<div class="section-title">Состояние Telegram</div>
|
||||||
|
<div id="auth-status" class="status-badge">Проверяем...</div>
|
||||||
|
<p id="auth-details" class="muted small"></p>
|
||||||
|
</section>
|
||||||
|
|
||||||
|
<section class="status-card">
|
||||||
|
<div class="section-title">Быстрые действия</div>
|
||||||
|
<button class="button primary" id="scrape-all-btn">Скрапить все каналы</button>
|
||||||
|
<button class="button" id="export-all-btn">Экспортировать всё</button>
|
||||||
|
<button class="button" id="refresh-dialogs-btn">Обновить список диалогов</button>
|
||||||
|
</section>
|
||||||
|
</aside>
|
||||||
|
|
||||||
|
<main class="content">
|
||||||
|
<section class="panel-grid">
|
||||||
|
<div class="panel stat-panel">
|
||||||
|
<div class="section-title">Каналы</div>
|
||||||
|
<div id="channel-count" class="stat-value">0</div>
|
||||||
|
</div>
|
||||||
|
<div class="panel stat-panel">
|
||||||
|
<div class="section-title">Media scraping</div>
|
||||||
|
<label class="toggle-row">
|
||||||
|
<span id="scrape-media-label">OFF</span>
|
||||||
|
<input id="scrape-media-toggle" type="checkbox" />
|
||||||
|
</label>
|
||||||
|
</div>
|
||||||
|
<div class="panel stat-panel">
|
||||||
|
<div class="section-title">Forwarding rules</div>
|
||||||
|
<div id="forwarding-count" class="stat-value">0</div>
|
||||||
|
</div>
|
||||||
|
</section>
|
||||||
|
|
||||||
|
<section class="panel">
|
||||||
|
<div class="panel-header">
|
||||||
|
<div>
|
||||||
|
<h2>Отслеживаемые каналы</h2>
|
||||||
|
<p class="muted">Берутся из <code>data/state.json</code>, статистика подтягивается из SQLite.</p>
|
||||||
|
</div>
|
||||||
|
<form id="add-channel-form" class="inline-form">
|
||||||
|
<input id="add-channel-id" name="channel_id" placeholder="ID или @username" required />
|
||||||
|
<input id="add-channel-name" name="name" placeholder="Псевдоним (необязательно)" />
|
||||||
|
<button class="button primary" type="submit">Добавить</button>
|
||||||
|
</form>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div class="table-wrap">
|
||||||
|
<table>
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th>Канал</th>
|
||||||
|
<th>Сообщений</th>
|
||||||
|
<th>Медиа</th>
|
||||||
|
<th>Последнее сообщение</th>
|
||||||
|
<th>Действия</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody id="channels-table"></tbody>
|
||||||
|
</table>
|
||||||
|
</div>
|
||||||
|
</section>
|
||||||
|
|
||||||
|
<section class="panel jobs-panel">
|
||||||
|
<div class="panel-header">
|
||||||
|
<div>
|
||||||
|
<h2>Фоновые задачи</h2>
|
||||||
|
<p class="muted">Очередь простая: задачи выполняются по одной, чтобы не драться за Telegram session.</p>
|
||||||
|
</div>
|
||||||
|
<button class="button" id="refresh-jobs-btn">Обновить</button>
|
||||||
|
</div>
|
||||||
|
<div id="jobs-list" class="jobs-list"></div>
|
||||||
|
</section>
|
||||||
|
</main>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<template id="channel-row-template">
|
||||||
|
<tr>
|
||||||
|
<td>
|
||||||
|
<div class="channel-name"></div>
|
||||||
|
<div class="muted small channel-id"></div>
|
||||||
|
</td>
|
||||||
|
<td class="message-count"></td>
|
||||||
|
<td class="media-count"></td>
|
||||||
|
<td class="last-date"></td>
|
||||||
|
<td>
|
||||||
|
<div class="action-row">
|
||||||
|
<button class="button button-small scrape-btn">Scrape</button>
|
||||||
|
<button class="button button-small export-view-btn">Viewer</button>
|
||||||
|
<button class="button button-small media-btn">Rescrape media</button>
|
||||||
|
<button class="button button-small danger remove-btn">Удалить</button>
|
||||||
|
</div>
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
|
</template>
|
||||||
|
|
||||||
|
<template id="job-template">
|
||||||
|
<article class="job-card">
|
||||||
|
<div class="job-head">
|
||||||
|
<div class="job-title"></div>
|
||||||
|
<div class="job-status"></div>
|
||||||
|
</div>
|
||||||
|
<div class="muted small job-time"></div>
|
||||||
|
<pre class="job-logs"></pre>
|
||||||
|
</article>
|
||||||
|
</template>
|
||||||
|
|
||||||
|
<script src="/static/app.js"></script>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
+365
@@ -0,0 +1,365 @@
|
|||||||
|
:root {
|
||||||
|
color-scheme: dark;
|
||||||
|
--bg: #0f1723;
|
||||||
|
--panel: #17212b;
|
||||||
|
--panel-alt: #101922;
|
||||||
|
--line: #253241;
|
||||||
|
--soft: #8ea2b5;
|
||||||
|
--text: #edf3f9;
|
||||||
|
--accent: #53a7ff;
|
||||||
|
--accent-strong: #2f8cff;
|
||||||
|
--danger: #ff6b6b;
|
||||||
|
--bubble: #182533;
|
||||||
|
--bubble-self: #1f3a4d;
|
||||||
|
}
|
||||||
|
|
||||||
|
* {
|
||||||
|
box-sizing: border-box;
|
||||||
|
}
|
||||||
|
|
||||||
|
body {
|
||||||
|
margin: 0;
|
||||||
|
font-family: Inter, Arial, sans-serif;
|
||||||
|
background: var(--bg);
|
||||||
|
color: var(--text);
|
||||||
|
}
|
||||||
|
|
||||||
|
a {
|
||||||
|
color: inherit;
|
||||||
|
text-decoration: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
code,
|
||||||
|
pre,
|
||||||
|
input,
|
||||||
|
button {
|
||||||
|
font: inherit;
|
||||||
|
}
|
||||||
|
|
||||||
|
.app-shell,
|
||||||
|
.viewer-shell {
|
||||||
|
min-height: 100vh;
|
||||||
|
display: grid;
|
||||||
|
}
|
||||||
|
|
||||||
|
.app-shell {
|
||||||
|
grid-template-columns: 320px minmax(0, 1fr);
|
||||||
|
}
|
||||||
|
|
||||||
|
.sidebar,
|
||||||
|
.viewer-sidebar {
|
||||||
|
background: var(--panel-alt);
|
||||||
|
border-right: 1px solid var(--line);
|
||||||
|
padding: 24px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.content,
|
||||||
|
.viewer-main {
|
||||||
|
padding: 24px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.brand-block h1,
|
||||||
|
.viewer-sidebar-head h1,
|
||||||
|
.viewer-header h2,
|
||||||
|
.panel-header h2 {
|
||||||
|
margin: 0;
|
||||||
|
font-size: 24px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.eyebrow {
|
||||||
|
color: var(--accent);
|
||||||
|
font-size: 12px;
|
||||||
|
text-transform: uppercase;
|
||||||
|
margin-bottom: 8px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.muted {
|
||||||
|
color: var(--soft);
|
||||||
|
}
|
||||||
|
|
||||||
|
.small {
|
||||||
|
font-size: 12px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nav-links {
|
||||||
|
display: grid;
|
||||||
|
gap: 8px;
|
||||||
|
margin: 24px 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nav-link,
|
||||||
|
.button {
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
background: var(--panel);
|
||||||
|
color: var(--text);
|
||||||
|
padding: 10px 14px;
|
||||||
|
border-radius: 8px;
|
||||||
|
cursor: pointer;
|
||||||
|
}
|
||||||
|
|
||||||
|
.nav-link.active,
|
||||||
|
.button.primary {
|
||||||
|
background: var(--accent-strong);
|
||||||
|
border-color: var(--accent-strong);
|
||||||
|
}
|
||||||
|
|
||||||
|
.button.danger {
|
||||||
|
color: #ffd0d0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.button.button-small {
|
||||||
|
padding: 8px 10px;
|
||||||
|
font-size: 13px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.status-card,
|
||||||
|
.panel,
|
||||||
|
.stat-panel {
|
||||||
|
background: var(--panel);
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
border-radius: 8px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.status-card {
|
||||||
|
padding: 16px;
|
||||||
|
margin-top: 16px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.section-title {
|
||||||
|
font-size: 12px;
|
||||||
|
text-transform: uppercase;
|
||||||
|
color: var(--soft);
|
||||||
|
margin-bottom: 12px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.status-badge {
|
||||||
|
display: inline-flex;
|
||||||
|
align-items: center;
|
||||||
|
border-radius: 999px;
|
||||||
|
padding: 8px 12px;
|
||||||
|
background: #183248;
|
||||||
|
color: #a7d8ff;
|
||||||
|
margin-bottom: 10px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.panel-grid {
|
||||||
|
display: grid;
|
||||||
|
gap: 16px;
|
||||||
|
grid-template-columns: repeat(3, minmax(0, 1fr));
|
||||||
|
margin-bottom: 16px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.stat-panel {
|
||||||
|
padding: 18px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.stat-value {
|
||||||
|
font-size: 32px;
|
||||||
|
font-weight: 700;
|
||||||
|
}
|
||||||
|
|
||||||
|
.toggle-row {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
justify-content: space-between;
|
||||||
|
gap: 12px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.panel {
|
||||||
|
padding: 18px;
|
||||||
|
margin-bottom: 16px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.panel-header,
|
||||||
|
.viewer-sidebar-head,
|
||||||
|
.viewer-header {
|
||||||
|
display: flex;
|
||||||
|
align-items: flex-start;
|
||||||
|
justify-content: space-between;
|
||||||
|
gap: 16px;
|
||||||
|
margin-bottom: 16px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.inline-form {
|
||||||
|
display: flex;
|
||||||
|
gap: 8px;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
}
|
||||||
|
|
||||||
|
input {
|
||||||
|
min-width: 160px;
|
||||||
|
padding: 10px 12px;
|
||||||
|
border-radius: 8px;
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
background: #0d141c;
|
||||||
|
color: var(--text);
|
||||||
|
}
|
||||||
|
|
||||||
|
.table-wrap {
|
||||||
|
overflow-x: auto;
|
||||||
|
}
|
||||||
|
|
||||||
|
table {
|
||||||
|
width: 100%;
|
||||||
|
border-collapse: collapse;
|
||||||
|
}
|
||||||
|
|
||||||
|
th,
|
||||||
|
td {
|
||||||
|
text-align: left;
|
||||||
|
border-bottom: 1px solid var(--line);
|
||||||
|
padding: 14px 10px;
|
||||||
|
vertical-align: top;
|
||||||
|
}
|
||||||
|
|
||||||
|
tbody tr:hover {
|
||||||
|
background: rgba(255, 255, 255, 0.02);
|
||||||
|
}
|
||||||
|
|
||||||
|
.action-row {
|
||||||
|
display: flex;
|
||||||
|
gap: 8px;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
}
|
||||||
|
|
||||||
|
.jobs-list {
|
||||||
|
display: grid;
|
||||||
|
gap: 12px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.job-card {
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
border-radius: 8px;
|
||||||
|
padding: 14px;
|
||||||
|
background: #121c25;
|
||||||
|
}
|
||||||
|
|
||||||
|
.job-head {
|
||||||
|
display: flex;
|
||||||
|
justify-content: space-between;
|
||||||
|
gap: 12px;
|
||||||
|
margin-bottom: 6px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.job-status {
|
||||||
|
text-transform: uppercase;
|
||||||
|
font-size: 12px;
|
||||||
|
color: var(--soft);
|
||||||
|
}
|
||||||
|
|
||||||
|
.job-logs {
|
||||||
|
margin: 10px 0 0;
|
||||||
|
padding: 12px;
|
||||||
|
border-radius: 8px;
|
||||||
|
background: #0d141c;
|
||||||
|
max-height: 240px;
|
||||||
|
overflow: auto;
|
||||||
|
white-space: pre-wrap;
|
||||||
|
}
|
||||||
|
|
||||||
|
.viewer-shell {
|
||||||
|
grid-template-columns: 320px minmax(0, 1fr);
|
||||||
|
}
|
||||||
|
|
||||||
|
.viewer-channel-list {
|
||||||
|
display: grid;
|
||||||
|
gap: 8px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.viewer-channel-item {
|
||||||
|
display: grid;
|
||||||
|
gap: 4px;
|
||||||
|
width: 100%;
|
||||||
|
text-align: left;
|
||||||
|
padding: 12px;
|
||||||
|
border-radius: 8px;
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
background: var(--panel);
|
||||||
|
color: var(--text);
|
||||||
|
cursor: pointer;
|
||||||
|
}
|
||||||
|
|
||||||
|
.viewer-channel-item.active {
|
||||||
|
border-color: var(--accent);
|
||||||
|
background: #193246;
|
||||||
|
}
|
||||||
|
|
||||||
|
.viewer-channel-name {
|
||||||
|
font-weight: 600;
|
||||||
|
}
|
||||||
|
|
||||||
|
.viewer-channel-meta {
|
||||||
|
font-size: 12px;
|
||||||
|
color: var(--soft);
|
||||||
|
}
|
||||||
|
|
||||||
|
.messages-list {
|
||||||
|
display: grid;
|
||||||
|
gap: 12px;
|
||||||
|
align-content: start;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-card {
|
||||||
|
max-width: 760px;
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
border-radius: 12px;
|
||||||
|
padding: 14px;
|
||||||
|
background: var(--bubble);
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-meta {
|
||||||
|
display: flex;
|
||||||
|
justify-content: space-between;
|
||||||
|
gap: 12px;
|
||||||
|
margin-bottom: 10px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-author {
|
||||||
|
color: #8fd3ff;
|
||||||
|
font-weight: 600;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-reply:empty,
|
||||||
|
.message-text:empty,
|
||||||
|
.message-footer:empty,
|
||||||
|
.message-media:empty {
|
||||||
|
display: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-text {
|
||||||
|
white-space: pre-wrap;
|
||||||
|
line-height: 1.45;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-media {
|
||||||
|
margin-top: 12px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-media img,
|
||||||
|
.message-media video {
|
||||||
|
max-width: 100%;
|
||||||
|
border-radius: 10px;
|
||||||
|
border: 1px solid var(--line);
|
||||||
|
display: block;
|
||||||
|
}
|
||||||
|
|
||||||
|
.message-footer {
|
||||||
|
margin-top: 10px;
|
||||||
|
}
|
||||||
|
|
||||||
|
@media (max-width: 1100px) {
|
||||||
|
.app-shell,
|
||||||
|
.viewer-shell {
|
||||||
|
grid-template-columns: 1fr;
|
||||||
|
}
|
||||||
|
|
||||||
|
.sidebar,
|
||||||
|
.viewer-sidebar {
|
||||||
|
border-right: 0;
|
||||||
|
border-bottom: 1px solid var(--line);
|
||||||
|
}
|
||||||
|
|
||||||
|
.panel-grid {
|
||||||
|
grid-template-columns: 1fr;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
<!DOCTYPE html>
|
||||||
|
<html lang="ru">
|
||||||
|
<head>
|
||||||
|
<meta charset="UTF-8" />
|
||||||
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||||
|
<title>Telegram Scraper Viewer</title>
|
||||||
|
<link rel="stylesheet" href="/static/style.css" />
|
||||||
|
</head>
|
||||||
|
<body class="viewer-body">
|
||||||
|
<div class="viewer-shell">
|
||||||
|
<aside class="viewer-sidebar">
|
||||||
|
<div class="viewer-sidebar-head">
|
||||||
|
<div>
|
||||||
|
<div class="eyebrow">Export Viewer</div>
|
||||||
|
<h1>Сообщения</h1>
|
||||||
|
</div>
|
||||||
|
<a class="button button-small" href="/">Панель</a>
|
||||||
|
</div>
|
||||||
|
<div id="viewer-channel-list" class="viewer-channel-list"></div>
|
||||||
|
</aside>
|
||||||
|
|
||||||
|
<main class="viewer-main">
|
||||||
|
<header class="viewer-header">
|
||||||
|
<div>
|
||||||
|
<h2 id="viewer-title">Выберите канал</h2>
|
||||||
|
<p id="viewer-subtitle" class="muted">Показываем данные из локальной SQLite базы.</p>
|
||||||
|
</div>
|
||||||
|
<div class="viewer-actions">
|
||||||
|
<button class="button" id="load-older-btn" disabled>Загрузить старее</button>
|
||||||
|
</div>
|
||||||
|
</header>
|
||||||
|
|
||||||
|
<section id="messages-list" class="messages-list"></section>
|
||||||
|
</main>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<template id="viewer-channel-template">
|
||||||
|
<button class="viewer-channel-item">
|
||||||
|
<span class="viewer-channel-name"></span>
|
||||||
|
<span class="viewer-channel-meta"></span>
|
||||||
|
</button>
|
||||||
|
</template>
|
||||||
|
|
||||||
|
<template id="message-template">
|
||||||
|
<article class="message-card">
|
||||||
|
<div class="message-meta">
|
||||||
|
<span class="message-author"></span>
|
||||||
|
<span class="message-date"></span>
|
||||||
|
</div>
|
||||||
|
<div class="message-reply muted small"></div>
|
||||||
|
<div class="message-text"></div>
|
||||||
|
<div class="message-media"></div>
|
||||||
|
<div class="message-footer muted small"></div>
|
||||||
|
</article>
|
||||||
|
</template>
|
||||||
|
|
||||||
|
<script src="/static/viewer.js"></script>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
+137
@@ -0,0 +1,137 @@
|
|||||||
|
const viewerState = {
|
||||||
|
channels: [],
|
||||||
|
channelId: null,
|
||||||
|
oldestMessageId: null,
|
||||||
|
};
|
||||||
|
|
||||||
|
async function api(path) {
|
||||||
|
const response = await fetch(path);
|
||||||
|
const data = await response.json();
|
||||||
|
if (!response.ok) {
|
||||||
|
throw new Error(data.error || "Request failed");
|
||||||
|
}
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
function textNodeWithBreaks(text) {
|
||||||
|
const fragment = document.createDocumentFragment();
|
||||||
|
const parts = (text || "").split("\n");
|
||||||
|
parts.forEach((part, index) => {
|
||||||
|
if (index > 0) fragment.appendChild(document.createElement("br"));
|
||||||
|
fragment.appendChild(document.createTextNode(part));
|
||||||
|
});
|
||||||
|
return fragment;
|
||||||
|
}
|
||||||
|
|
||||||
|
function renderChannelList() {
|
||||||
|
const root = document.getElementById("viewer-channel-list");
|
||||||
|
root.innerHTML = "";
|
||||||
|
const template = document.getElementById("viewer-channel-template");
|
||||||
|
viewerState.channels.forEach((channel) => {
|
||||||
|
const node = template.content.firstElementChild.cloneNode(true);
|
||||||
|
node.querySelector(".viewer-channel-name").textContent = channel.name;
|
||||||
|
node.querySelector(".viewer-channel-meta").textContent = `${channel.message_count} сообщений`;
|
||||||
|
if (channel.channel_id === viewerState.channelId) {
|
||||||
|
node.classList.add("active");
|
||||||
|
}
|
||||||
|
node.addEventListener("click", () => {
|
||||||
|
viewerState.oldestMessageId = null;
|
||||||
|
loadChannel(channel.channel_id);
|
||||||
|
});
|
||||||
|
root.appendChild(node);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
function renderMessages(payload, append = false) {
|
||||||
|
const root = document.getElementById("messages-list");
|
||||||
|
const template = document.getElementById("message-template");
|
||||||
|
if (!append) root.innerHTML = "";
|
||||||
|
|
||||||
|
const channel = payload.channel;
|
||||||
|
document.getElementById("viewer-title").textContent = channel?.name || payload.channel_id;
|
||||||
|
document.getElementById("viewer-subtitle").textContent = channel
|
||||||
|
? `${channel.message_count} сообщений, последнее: ${channel.last_date || "—"}`
|
||||||
|
: "База данных для канала пока не найдена.";
|
||||||
|
|
||||||
|
payload.messages.forEach((message) => {
|
||||||
|
const node = template.content.firstElementChild.cloneNode(true);
|
||||||
|
node.dataset.messageId = String(message.message_id);
|
||||||
|
node.querySelector(".message-author").textContent = message.sender_name || "Unknown";
|
||||||
|
node.querySelector(".message-date").textContent = message.date || "";
|
||||||
|
|
||||||
|
const reply = node.querySelector(".message-reply");
|
||||||
|
if (message.reply_to) {
|
||||||
|
reply.textContent = `Ответ на сообщение #${message.reply_to}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
node.querySelector(".message-text").appendChild(textNodeWithBreaks(message.text || ""));
|
||||||
|
|
||||||
|
const media = node.querySelector(".message-media");
|
||||||
|
if (message.media_url && message.media_kind === "image") {
|
||||||
|
const img = document.createElement("img");
|
||||||
|
img.src = message.media_url;
|
||||||
|
img.loading = "lazy";
|
||||||
|
media.appendChild(img);
|
||||||
|
} else if (message.media_url && message.media_kind === "video") {
|
||||||
|
const video = document.createElement("video");
|
||||||
|
video.src = message.media_url;
|
||||||
|
video.controls = true;
|
||||||
|
video.preload = "metadata";
|
||||||
|
media.appendChild(video);
|
||||||
|
} else if (message.media_url) {
|
||||||
|
const link = document.createElement("a");
|
||||||
|
link.href = message.media_url;
|
||||||
|
link.target = "_blank";
|
||||||
|
link.rel = "noreferrer";
|
||||||
|
link.textContent = "Открыть файл";
|
||||||
|
media.appendChild(link);
|
||||||
|
}
|
||||||
|
|
||||||
|
const footerParts = [];
|
||||||
|
if (message.views) footerParts.push(`views: ${message.views}`);
|
||||||
|
if (message.forwards) footerParts.push(`forwards: ${message.forwards}`);
|
||||||
|
if (message.reactions) footerParts.push(`reactions: ${message.reactions}`);
|
||||||
|
node.querySelector(".message-footer").textContent = footerParts.join(" | ");
|
||||||
|
|
||||||
|
if (append) {
|
||||||
|
root.prepend(node);
|
||||||
|
} else {
|
||||||
|
root.appendChild(node);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
if (payload.messages.length) {
|
||||||
|
viewerState.oldestMessageId = payload.messages[0].message_id;
|
||||||
|
}
|
||||||
|
|
||||||
|
document.getElementById("load-older-btn").disabled = !payload.messages.length;
|
||||||
|
}
|
||||||
|
|
||||||
|
async function loadChannel(channelId, append = false) {
|
||||||
|
viewerState.channelId = channelId;
|
||||||
|
renderChannelList();
|
||||||
|
const before = append && viewerState.oldestMessageId ? `&before=${viewerState.oldestMessageId}` : "";
|
||||||
|
const payload = await api(`/api/channels/${encodeURIComponent(channelId)}/messages?limit=80${before}`);
|
||||||
|
renderMessages(payload, append);
|
||||||
|
}
|
||||||
|
|
||||||
|
async function main() {
|
||||||
|
viewerState.channels = await api("/api/channels");
|
||||||
|
const params = new URLSearchParams(window.location.search);
|
||||||
|
viewerState.channelId = params.get("channel") || viewerState.channels[0]?.channel_id || null;
|
||||||
|
renderChannelList();
|
||||||
|
|
||||||
|
if (viewerState.channelId) {
|
||||||
|
await loadChannel(viewerState.channelId);
|
||||||
|
}
|
||||||
|
|
||||||
|
document.getElementById("load-older-btn").addEventListener("click", async () => {
|
||||||
|
if (!viewerState.channelId || !viewerState.oldestMessageId) return;
|
||||||
|
await loadChannel(viewerState.channelId, true);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
main().catch((error) => {
|
||||||
|
console.error(error);
|
||||||
|
alert(error.message);
|
||||||
|
});
|
||||||
+623
@@ -0,0 +1,623 @@
|
|||||||
|
import asyncio
|
||||||
|
import contextlib
|
||||||
|
import io
|
||||||
|
import json
|
||||||
|
import mimetypes
|
||||||
|
import os
|
||||||
|
import posixpath
|
||||||
|
import queue
|
||||||
|
import sqlite3
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
import traceback
|
||||||
|
import urllib.parse
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from http import HTTPStatus
|
||||||
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Callable, Dict, List, Optional
|
||||||
|
|
||||||
|
|
||||||
|
BASE_DIR = Path(__file__).resolve().parent
|
||||||
|
DATA_DIR = BASE_DIR / "data"
|
||||||
|
WEBUI_DIR = BASE_DIR / "webui"
|
||||||
|
STATE_FILE = DATA_DIR / "state.json"
|
||||||
|
DEFAULT_HOST = os.environ.get("TELEGRAM_SCRAPER_HOST", "0.0.0.0")
|
||||||
|
DEFAULT_PORT = int(os.environ.get("TELEGRAM_SCRAPER_PORT", "8080"))
|
||||||
|
|
||||||
|
|
||||||
|
def utc_now_iso() -> str:
|
||||||
|
return datetime.now(timezone.utc).isoformat()
|
||||||
|
|
||||||
|
|
||||||
|
def load_state() -> Dict[str, Any]:
|
||||||
|
if STATE_FILE.exists():
|
||||||
|
with STATE_FILE.open("r", encoding="utf-8") as handle:
|
||||||
|
return json.load(handle)
|
||||||
|
return {
|
||||||
|
"api_id": None,
|
||||||
|
"api_hash": None,
|
||||||
|
"channels": {},
|
||||||
|
"channel_names": {},
|
||||||
|
"scrape_media": True,
|
||||||
|
"forwarding_rules": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def save_state(state: Dict[str, Any]) -> None:
|
||||||
|
DATA_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
with STATE_FILE.open("w", encoding="utf-8") as handle:
|
||||||
|
json.dump(state, handle, ensure_ascii=False, indent=2)
|
||||||
|
|
||||||
|
|
||||||
|
def channel_db_path(channel_id: str) -> Path:
|
||||||
|
return DATA_DIR / channel_id / f"{channel_id}.db"
|
||||||
|
|
||||||
|
|
||||||
|
def guess_media_kind(media_path: Optional[str], media_type: Optional[str]) -> Optional[str]:
|
||||||
|
if not media_path:
|
||||||
|
return None
|
||||||
|
suffix = Path(media_path).suffix.lower()
|
||||||
|
if suffix in {".jpg", ".jpeg", ".png", ".webp", ".gif"}:
|
||||||
|
return "image"
|
||||||
|
if suffix in {".mp4", ".webm", ".mov", ".m4v"}:
|
||||||
|
return "video"
|
||||||
|
if suffix in {".mp3", ".ogg", ".wav", ".m4a"}:
|
||||||
|
return "audio"
|
||||||
|
if media_type == "MessageMediaPhoto":
|
||||||
|
return "image"
|
||||||
|
return "file"
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_media_url(media_path: Optional[str]) -> Optional[str]:
|
||||||
|
if not media_path:
|
||||||
|
return None
|
||||||
|
path = Path(media_path)
|
||||||
|
if path.is_absolute():
|
||||||
|
try:
|
||||||
|
relative = path.resolve().relative_to(DATA_DIR.resolve())
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
else:
|
||||||
|
normalized = Path(media_path)
|
||||||
|
if normalized.parts and normalized.parts[0] == "data":
|
||||||
|
normalized = Path(*normalized.parts[1:])
|
||||||
|
relative = normalized
|
||||||
|
return "/media/" + urllib.parse.quote(str(relative).replace("\\", "/"))
|
||||||
|
|
||||||
|
|
||||||
|
def database_summary(channel_id: str) -> Dict[str, Any]:
|
||||||
|
db_path = channel_db_path(channel_id)
|
||||||
|
summary = {
|
||||||
|
"message_count": 0,
|
||||||
|
"last_date": None,
|
||||||
|
"first_date": None,
|
||||||
|
"media_count": 0,
|
||||||
|
"has_database": db_path.exists(),
|
||||||
|
}
|
||||||
|
if not db_path.exists():
|
||||||
|
return summary
|
||||||
|
conn = sqlite3.connect(str(db_path))
|
||||||
|
try:
|
||||||
|
cursor = conn.cursor()
|
||||||
|
cursor.execute(
|
||||||
|
"SELECT COUNT(*), MIN(date), MAX(date), "
|
||||||
|
"SUM(CASE WHEN media_type IS NOT NULL AND media_type != 'MessageMediaWebPage' THEN 1 ELSE 0 END) "
|
||||||
|
"FROM messages"
|
||||||
|
)
|
||||||
|
row = cursor.fetchone() or (0, None, None, 0)
|
||||||
|
summary.update(
|
||||||
|
{
|
||||||
|
"message_count": row[0] or 0,
|
||||||
|
"first_date": row[1],
|
||||||
|
"last_date": row[2],
|
||||||
|
"media_count": row[3] or 0,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return summary
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
|
||||||
|
def channel_display_name(state: Dict[str, Any], channel_id: str) -> str:
|
||||||
|
return (
|
||||||
|
state.get("channel_names", {}).get(channel_id)
|
||||||
|
or channel_id
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def list_channels_snapshot() -> List[Dict[str, Any]]:
|
||||||
|
state = load_state()
|
||||||
|
channels: List[Dict[str, Any]] = []
|
||||||
|
for channel_id, last_message_id in state.get("channels", {}).items():
|
||||||
|
summary = database_summary(channel_id)
|
||||||
|
channels.append(
|
||||||
|
{
|
||||||
|
"channel_id": channel_id,
|
||||||
|
"name": channel_display_name(state, channel_id),
|
||||||
|
"last_message_id": last_message_id,
|
||||||
|
**summary,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
channels.sort(
|
||||||
|
key=lambda item: (
|
||||||
|
item["last_date"] or "",
|
||||||
|
item["message_count"],
|
||||||
|
),
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
return channels
|
||||||
|
|
||||||
|
|
||||||
|
def load_messages(channel_id: str, limit: int = 120, before_message_id: Optional[int] = None) -> List[Dict[str, Any]]:
|
||||||
|
db_path = channel_db_path(channel_id)
|
||||||
|
if not db_path.exists():
|
||||||
|
return []
|
||||||
|
conn = sqlite3.connect(str(db_path))
|
||||||
|
conn.row_factory = sqlite3.Row
|
||||||
|
try:
|
||||||
|
query = (
|
||||||
|
"SELECT message_id, date, sender_id, first_name, last_name, username, message, "
|
||||||
|
"media_type, media_path, reply_to, post_author, views, forwards, reactions "
|
||||||
|
"FROM messages "
|
||||||
|
)
|
||||||
|
params: List[Any] = []
|
||||||
|
if before_message_id is not None:
|
||||||
|
query += "WHERE message_id < ? "
|
||||||
|
params.append(before_message_id)
|
||||||
|
query += "ORDER BY message_id DESC LIMIT ?"
|
||||||
|
params.append(limit)
|
||||||
|
rows = conn.execute(query, params).fetchall()
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
messages: List[Dict[str, Any]] = []
|
||||||
|
for row in reversed(rows):
|
||||||
|
sender_name = " ".join(
|
||||||
|
part for part in [row["first_name"], row["last_name"]] if part
|
||||||
|
).strip()
|
||||||
|
if not sender_name:
|
||||||
|
sender_name = row["username"] or row["post_author"] or channel_id
|
||||||
|
messages.append(
|
||||||
|
{
|
||||||
|
"message_id": row["message_id"],
|
||||||
|
"date": row["date"],
|
||||||
|
"sender_id": row["sender_id"],
|
||||||
|
"sender_name": sender_name,
|
||||||
|
"username": row["username"],
|
||||||
|
"text": row["message"] or "",
|
||||||
|
"media_type": row["media_type"],
|
||||||
|
"media_path": row["media_path"],
|
||||||
|
"media_url": normalize_media_url(row["media_path"]),
|
||||||
|
"media_kind": guess_media_kind(row["media_path"], row["media_type"]),
|
||||||
|
"reply_to": row["reply_to"],
|
||||||
|
"post_author": row["post_author"],
|
||||||
|
"views": row["views"],
|
||||||
|
"forwards": row["forwards"],
|
||||||
|
"reactions": row["reactions"],
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return messages
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Job:
|
||||||
|
job_id: str
|
||||||
|
job_type: str
|
||||||
|
title: str
|
||||||
|
payload: Dict[str, Any]
|
||||||
|
status: str = "queued"
|
||||||
|
created_at: str = field(default_factory=utc_now_iso)
|
||||||
|
started_at: Optional[str] = None
|
||||||
|
finished_at: Optional[str] = None
|
||||||
|
logs: str = ""
|
||||||
|
error: Optional[str] = None
|
||||||
|
|
||||||
|
def to_dict(self) -> Dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"job_id": self.job_id,
|
||||||
|
"job_type": self.job_type,
|
||||||
|
"title": self.title,
|
||||||
|
"payload": self.payload,
|
||||||
|
"status": self.status,
|
||||||
|
"created_at": self.created_at,
|
||||||
|
"started_at": self.started_at,
|
||||||
|
"finished_at": self.finished_at,
|
||||||
|
"logs": self.logs,
|
||||||
|
"error": self.error,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class JobRunner:
|
||||||
|
def __init__(self) -> None:
|
||||||
|
self.jobs: Dict[str, Job] = {}
|
||||||
|
self.job_order: List[str] = []
|
||||||
|
self.queue: "queue.Queue[Job]" = queue.Queue()
|
||||||
|
self.lock = threading.Lock()
|
||||||
|
self.worker = threading.Thread(target=self._run, daemon=True)
|
||||||
|
self.worker.start()
|
||||||
|
|
||||||
|
def create_job(self, job_type: str, title: str, payload: Dict[str, Any]) -> Job:
|
||||||
|
job_id = f"job-{int(time.time() * 1000)}"
|
||||||
|
job = Job(job_id=job_id, job_type=job_type, title=title, payload=payload)
|
||||||
|
with self.lock:
|
||||||
|
self.jobs[job_id] = job
|
||||||
|
self.job_order.insert(0, job_id)
|
||||||
|
self.job_order = self.job_order[:20]
|
||||||
|
self.queue.put(job)
|
||||||
|
return job
|
||||||
|
|
||||||
|
def recent_jobs(self) -> List[Dict[str, Any]]:
|
||||||
|
with self.lock:
|
||||||
|
return [self.jobs[job_id].to_dict() for job_id in self.job_order]
|
||||||
|
|
||||||
|
def get_job(self, job_id: str) -> Optional[Dict[str, Any]]:
|
||||||
|
with self.lock:
|
||||||
|
job = self.jobs.get(job_id)
|
||||||
|
return job.to_dict() if job else None
|
||||||
|
|
||||||
|
def _run(self) -> None:
|
||||||
|
while True:
|
||||||
|
job = self.queue.get()
|
||||||
|
self._execute(job)
|
||||||
|
self.queue.task_done()
|
||||||
|
|
||||||
|
def _execute(self, job: Job) -> None:
|
||||||
|
with self.lock:
|
||||||
|
job.status = "running"
|
||||||
|
job.started_at = utc_now_iso()
|
||||||
|
buffer = io.StringIO()
|
||||||
|
try:
|
||||||
|
with contextlib.redirect_stdout(buffer), contextlib.redirect_stderr(buffer):
|
||||||
|
run_job(job.job_type, job.payload)
|
||||||
|
with self.lock:
|
||||||
|
job.status = "done"
|
||||||
|
except Exception as exc:
|
||||||
|
with self.lock:
|
||||||
|
job.status = "failed"
|
||||||
|
job.error = str(exc)
|
||||||
|
traceback.print_exc(file=buffer)
|
||||||
|
finally:
|
||||||
|
with self.lock:
|
||||||
|
job.logs = buffer.getvalue()
|
||||||
|
job.finished_at = utc_now_iso()
|
||||||
|
|
||||||
|
|
||||||
|
def import_scraper_class():
|
||||||
|
from telegram_scraper_with_forwarding import OptimizedTelegramScraper
|
||||||
|
|
||||||
|
return OptimizedTelegramScraper
|
||||||
|
|
||||||
|
|
||||||
|
def run_job(job_type: str, payload: Dict[str, Any]) -> None:
|
||||||
|
if job_type == "set_scrape_media":
|
||||||
|
state = load_state()
|
||||||
|
state["scrape_media"] = bool(payload["value"])
|
||||||
|
save_state(state)
|
||||||
|
print(f"Media scraping set to {state['scrape_media']}")
|
||||||
|
return
|
||||||
|
|
||||||
|
async def _async_job() -> None:
|
||||||
|
ScraperClass = import_scraper_class()
|
||||||
|
scraper = ScraperClass()
|
||||||
|
initialized = await scraper.initialize_client(interactive=False)
|
||||||
|
if not initialized:
|
||||||
|
raise RuntimeError("Telegram client is not ready. Check credentials and session.")
|
||||||
|
try:
|
||||||
|
if job_type == "scrape_channel":
|
||||||
|
channel_id = payload["channel_id"]
|
||||||
|
state = scraper.load_state()
|
||||||
|
offset = int(state.get("channels", {}).get(channel_id, 0) or 0)
|
||||||
|
await scraper.scrape_channel(channel_id, offset)
|
||||||
|
elif job_type == "scrape_all":
|
||||||
|
channels = list(scraper.state.get("channels", {}).keys())
|
||||||
|
for channel_id in channels:
|
||||||
|
offset = int(scraper.state.get("channels", {}).get(channel_id, 0) or 0)
|
||||||
|
await scraper.scrape_channel(channel_id, offset)
|
||||||
|
elif job_type == "export_all":
|
||||||
|
await scraper.export_data()
|
||||||
|
elif job_type == "rescrape_media":
|
||||||
|
await scraper.rescrape_media(payload["channel_id"])
|
||||||
|
elif job_type == "fix_missing_media":
|
||||||
|
await scraper.fix_missing_media(payload["channel_id"])
|
||||||
|
elif job_type == "refresh_dialogs":
|
||||||
|
await scraper.list_channels()
|
||||||
|
else:
|
||||||
|
raise RuntimeError(f"Unsupported job type: {job_type}")
|
||||||
|
finally:
|
||||||
|
scraper.close_db_connections()
|
||||||
|
if scraper.client:
|
||||||
|
await scraper.client.disconnect()
|
||||||
|
|
||||||
|
asyncio.run(_async_job())
|
||||||
|
|
||||||
|
|
||||||
|
def auth_status() -> Dict[str, Any]:
|
||||||
|
state = load_state()
|
||||||
|
status = {
|
||||||
|
"has_api_credentials": bool(state.get("api_id") and state.get("api_hash")),
|
||||||
|
"telethon_available": False,
|
||||||
|
"session_ready": False,
|
||||||
|
"status": "read_only",
|
||||||
|
"details": "",
|
||||||
|
}
|
||||||
|
try:
|
||||||
|
import_scraper_class()
|
||||||
|
status["telethon_available"] = True
|
||||||
|
except Exception as exc:
|
||||||
|
status["details"] = f"Python deps are missing: {exc}"
|
||||||
|
return status
|
||||||
|
|
||||||
|
session_file = BASE_DIR / "session" / "session.session"
|
||||||
|
status["session_ready"] = session_file.exists()
|
||||||
|
if status["has_api_credentials"] and status["session_ready"]:
|
||||||
|
status["status"] = "ready"
|
||||||
|
status["details"] = "Scraping actions should be available after dependencies are installed."
|
||||||
|
elif status["has_api_credentials"]:
|
||||||
|
status["status"] = "needs_auth"
|
||||||
|
status["details"] = "API keys found, but Telegram session file is missing."
|
||||||
|
else:
|
||||||
|
status["status"] = "needs_credentials"
|
||||||
|
status["details"] = "api_id/api_hash are missing in data/state.json."
|
||||||
|
return status
|
||||||
|
|
||||||
|
|
||||||
|
def dashboard_payload(job_runner: JobRunner) -> Dict[str, Any]:
|
||||||
|
state = load_state()
|
||||||
|
channels = list_channels_snapshot()
|
||||||
|
return {
|
||||||
|
"state": {
|
||||||
|
"scrape_media": bool(state.get("scrape_media", True)),
|
||||||
|
"channel_count": len(state.get("channels", {})),
|
||||||
|
"forwarding_rules": state.get("forwarding_rules", []),
|
||||||
|
},
|
||||||
|
"auth": auth_status(),
|
||||||
|
"channels": channels,
|
||||||
|
"jobs": job_runner.recent_jobs(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class TelegramScraperRequestHandler(BaseHTTPRequestHandler):
|
||||||
|
server_version = "TelegramScraperWebUI/0.1"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def app(self) -> "TelegramScraperWebServer":
|
||||||
|
return self.server # type: ignore[return-value]
|
||||||
|
|
||||||
|
def do_GET(self) -> None:
|
||||||
|
parsed = urllib.parse.urlparse(self.path)
|
||||||
|
path = parsed.path
|
||||||
|
query = urllib.parse.parse_qs(parsed.query)
|
||||||
|
|
||||||
|
if path == "/":
|
||||||
|
return self.serve_file(WEBUI_DIR / "index.html", "text/html; charset=utf-8")
|
||||||
|
if path == "/viewer":
|
||||||
|
return self.serve_file(WEBUI_DIR / "viewer.html", "text/html; charset=utf-8")
|
||||||
|
if path.startswith("/static/"):
|
||||||
|
relative = path.removeprefix("/static/")
|
||||||
|
return self.serve_static(relative)
|
||||||
|
if path.startswith("/media/"):
|
||||||
|
relative = path.removeprefix("/media/")
|
||||||
|
return self.serve_media(relative)
|
||||||
|
if path == "/api/dashboard":
|
||||||
|
return self.send_json(dashboard_payload(self.app.job_runner))
|
||||||
|
if path == "/api/jobs":
|
||||||
|
return self.send_json(self.app.job_runner.recent_jobs())
|
||||||
|
if path.startswith("/api/jobs/"):
|
||||||
|
job_id = path.rsplit("/", 1)[-1]
|
||||||
|
job = self.app.job_runner.get_job(job_id)
|
||||||
|
if not job:
|
||||||
|
return self.send_error_json(HTTPStatus.NOT_FOUND, "Job not found")
|
||||||
|
return self.send_json(job)
|
||||||
|
if path == "/api/channels":
|
||||||
|
return self.send_json(list_channels_snapshot())
|
||||||
|
if path.startswith("/api/channels/") and path.endswith("/messages"):
|
||||||
|
parts = path.split("/")
|
||||||
|
channel_id = parts[3]
|
||||||
|
limit = max(1, min(int(query.get("limit", ["120"])[0]), 300))
|
||||||
|
before = query.get("before")
|
||||||
|
before_message_id = int(before[0]) if before else None
|
||||||
|
payload = {
|
||||||
|
"channel_id": channel_id,
|
||||||
|
"messages": load_messages(channel_id, limit=limit, before_message_id=before_message_id),
|
||||||
|
"channel": next(
|
||||||
|
(item for item in list_channels_snapshot() if item["channel_id"] == channel_id),
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
return self.send_json(payload)
|
||||||
|
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
|
||||||
|
|
||||||
|
def do_HEAD(self) -> None:
|
||||||
|
parsed = urllib.parse.urlparse(self.path)
|
||||||
|
path = parsed.path
|
||||||
|
|
||||||
|
if path == "/":
|
||||||
|
return self.serve_file(WEBUI_DIR / "index.html", "text/html; charset=utf-8", head_only=True)
|
||||||
|
if path == "/viewer":
|
||||||
|
return self.serve_file(WEBUI_DIR / "viewer.html", "text/html; charset=utf-8", head_only=True)
|
||||||
|
if path.startswith("/static/"):
|
||||||
|
relative = path.removeprefix("/static/")
|
||||||
|
return self.serve_static(relative, head_only=True)
|
||||||
|
if path.startswith("/media/"):
|
||||||
|
relative = path.removeprefix("/media/")
|
||||||
|
return self.serve_media(relative, head_only=True)
|
||||||
|
if path in {"/api/dashboard", "/api/jobs", "/api/channels"} or path.startswith("/api/jobs/") or (
|
||||||
|
path.startswith("/api/channels/") and path.endswith("/messages")
|
||||||
|
):
|
||||||
|
self.send_response(HTTPStatus.OK)
|
||||||
|
self.send_header("Content-Type", "application/json; charset=utf-8")
|
||||||
|
self.end_headers()
|
||||||
|
return
|
||||||
|
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
|
||||||
|
|
||||||
|
def do_POST(self) -> None:
|
||||||
|
parsed = urllib.parse.urlparse(self.path)
|
||||||
|
path = parsed.path
|
||||||
|
body = self.read_json_body()
|
||||||
|
if body is None:
|
||||||
|
return self.send_error_json(HTTPStatus.BAD_REQUEST, "Expected JSON body")
|
||||||
|
|
||||||
|
if path == "/api/settings/media":
|
||||||
|
value = bool(body.get("value"))
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"set_scrape_media",
|
||||||
|
"Update media scraping setting",
|
||||||
|
{"value": value},
|
||||||
|
)
|
||||||
|
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||||
|
|
||||||
|
if path == "/api/channels/add":
|
||||||
|
channel_id = str(body.get("channel_id", "")).strip()
|
||||||
|
if not channel_id:
|
||||||
|
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
|
||||||
|
state = load_state()
|
||||||
|
if channel_id not in state["channels"]:
|
||||||
|
state["channels"][channel_id] = 0
|
||||||
|
if body.get("name"):
|
||||||
|
state.setdefault("channel_names", {})[channel_id] = str(body["name"]).strip()
|
||||||
|
save_state(state)
|
||||||
|
return self.send_json({"ok": True, "channel_id": channel_id})
|
||||||
|
|
||||||
|
if path == "/api/channels/remove":
|
||||||
|
channel_id = str(body.get("channel_id", "")).strip()
|
||||||
|
state = load_state()
|
||||||
|
existed = channel_id in state.get("channels", {})
|
||||||
|
state.get("channels", {}).pop(channel_id, None)
|
||||||
|
save_state(state)
|
||||||
|
return self.send_json({"ok": existed, "channel_id": channel_id})
|
||||||
|
|
||||||
|
if path == "/api/jobs/scrape":
|
||||||
|
channel_id = body.get("channel_id")
|
||||||
|
if channel_id:
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"scrape_channel",
|
||||||
|
f"Scrape channel {channel_id}",
|
||||||
|
{"channel_id": str(channel_id)},
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"scrape_all",
|
||||||
|
"Scrape all tracked channels",
|
||||||
|
{},
|
||||||
|
)
|
||||||
|
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||||
|
|
||||||
|
if path == "/api/jobs/export":
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"export_all",
|
||||||
|
"Export all tracked channels",
|
||||||
|
{},
|
||||||
|
)
|
||||||
|
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||||
|
|
||||||
|
if path == "/api/jobs/rescrape-media":
|
||||||
|
channel_id = str(body.get("channel_id", "")).strip()
|
||||||
|
if not channel_id:
|
||||||
|
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"rescrape_media",
|
||||||
|
f"Rescrape media for {channel_id}",
|
||||||
|
{"channel_id": channel_id},
|
||||||
|
)
|
||||||
|
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||||
|
|
||||||
|
if path == "/api/jobs/fix-media":
|
||||||
|
channel_id = str(body.get("channel_id", "")).strip()
|
||||||
|
if not channel_id:
|
||||||
|
return self.send_error_json(HTTPStatus.BAD_REQUEST, "channel_id is required")
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"fix_missing_media",
|
||||||
|
f"Fix missing media for {channel_id}",
|
||||||
|
{"channel_id": channel_id},
|
||||||
|
)
|
||||||
|
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||||
|
|
||||||
|
if path == "/api/jobs/refresh-dialogs":
|
||||||
|
job = self.app.job_runner.create_job(
|
||||||
|
"refresh_dialogs",
|
||||||
|
"Refresh Telegram dialogs list",
|
||||||
|
{},
|
||||||
|
)
|
||||||
|
return self.send_json(job.to_dict(), status=HTTPStatus.ACCEPTED)
|
||||||
|
|
||||||
|
return self.send_error_json(HTTPStatus.NOT_FOUND, "Not found")
|
||||||
|
|
||||||
|
def read_json_body(self) -> Optional[Dict[str, Any]]:
|
||||||
|
length = int(self.headers.get("Content-Length", "0"))
|
||||||
|
if length <= 0:
|
||||||
|
return {}
|
||||||
|
raw = self.rfile.read(length)
|
||||||
|
try:
|
||||||
|
return json.loads(raw.decode("utf-8"))
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
def serve_static(self, relative_path: str, head_only: bool = False) -> None:
|
||||||
|
file_path = (WEBUI_DIR / relative_path).resolve()
|
||||||
|
try:
|
||||||
|
file_path.relative_to(WEBUI_DIR.resolve())
|
||||||
|
except ValueError:
|
||||||
|
return self.send_error_json(HTTPStatus.FORBIDDEN, "Forbidden")
|
||||||
|
if not file_path.exists() or not file_path.is_file():
|
||||||
|
return self.send_error_json(HTTPStatus.NOT_FOUND, "File not found")
|
||||||
|
mime_type = mimetypes.guess_type(str(file_path))[0] or "application/octet-stream"
|
||||||
|
return self.serve_file(file_path, mime_type, head_only=head_only)
|
||||||
|
|
||||||
|
def serve_media(self, relative_path: str, head_only: bool = False) -> None:
|
||||||
|
clean = Path(posixpath.normpath(urllib.parse.unquote(relative_path)))
|
||||||
|
file_path = (DATA_DIR / clean).resolve()
|
||||||
|
try:
|
||||||
|
file_path.relative_to(DATA_DIR.resolve())
|
||||||
|
except ValueError:
|
||||||
|
return self.send_error_json(HTTPStatus.FORBIDDEN, "Forbidden")
|
||||||
|
if not file_path.exists() or not file_path.is_file():
|
||||||
|
return self.send_error_json(HTTPStatus.NOT_FOUND, "Media file not found")
|
||||||
|
mime_type = mimetypes.guess_type(str(file_path))[0] or "application/octet-stream"
|
||||||
|
return self.serve_file(file_path, mime_type, head_only=head_only)
|
||||||
|
|
||||||
|
def serve_file(self, file_path: Path, content_type: str, head_only: bool = False) -> None:
|
||||||
|
data = file_path.read_bytes()
|
||||||
|
self.send_response(HTTPStatus.OK)
|
||||||
|
self.send_header("Content-Type", content_type)
|
||||||
|
self.send_header("Content-Length", str(len(data)))
|
||||||
|
self.end_headers()
|
||||||
|
if not head_only:
|
||||||
|
self.wfile.write(data)
|
||||||
|
|
||||||
|
def send_json(self, payload: Any, status: HTTPStatus = HTTPStatus.OK) -> None:
|
||||||
|
raw = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||||
|
self.send_response(status)
|
||||||
|
self.send_header("Content-Type", "application/json; charset=utf-8")
|
||||||
|
self.send_header("Content-Length", str(len(raw)))
|
||||||
|
self.end_headers()
|
||||||
|
self.wfile.write(raw)
|
||||||
|
|
||||||
|
def send_error_json(self, status: HTTPStatus, message: str) -> None:
|
||||||
|
self.send_json({"error": message, "status": status}, status=status)
|
||||||
|
|
||||||
|
def log_message(self, format: str, *args: Any) -> None:
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
class TelegramScraperWebServer(ThreadingHTTPServer):
|
||||||
|
def __init__(self, server_address: tuple[str, int]):
|
||||||
|
super().__init__(server_address, TelegramScraperRequestHandler)
|
||||||
|
self.job_runner = JobRunner()
|
||||||
|
|
||||||
|
|
||||||
|
def run_server(host: str = DEFAULT_HOST, port: int = DEFAULT_PORT) -> None:
|
||||||
|
server = TelegramScraperWebServer((host, port))
|
||||||
|
print(f"Telegram Scraper Web UI listening on http://{host}:{port}")
|
||||||
|
print("Press Ctrl+C to stop.")
|
||||||
|
try:
|
||||||
|
server.serve_forever()
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
pass
|
||||||
|
finally:
|
||||||
|
server.server_close()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
run_server()
|
||||||
Reference in New Issue
Block a user