fix: a Discord source takes only messages with an image attached, from the posters it names, and knows its channel's name (#4481)
CI and images / lint (push) Successful in 4s
CI and images / extension-version (push) Successful in 3s
CI and images / frontend-build (push) Successful in 21s
CI and images / extension-test (push) Successful in 20s
CI and images / backend-lint-and-test (push) Successful in 34s
CI and images / integration (push) Successful in 2m24s
CI and images / sign-extension (push) Successful in 3s
CI and images / build-agent (push) Successful in 7s
CI and images / build-web (push) Successful in 1m42s
CI and images / smoke-web (push) Successful in 59s
CI and images / promote (push) Successful in 2s
CI and images / lint (push) Successful in 4s
CI and images / extension-version (push) Successful in 3s
CI and images / frontend-build (push) Successful in 21s
CI and images / extension-test (push) Successful in 20s
CI and images / backend-lint-and-test (push) Successful in 34s
CI and images / integration (push) Successful in 2m24s
CI and images / sign-extension (push) Successful in 3s
CI and images / build-agent (push) Successful in 7s
CI and images / build-web (push) Successful in 1m42s
CI and images / smoke-web (push) Successful in 59s
CI and images / promote (push) Successful in 2s
- A message with no image or video attachment is chat: extract_media returns nothing for it, so nothing downloads and no post record is written. A message that passes keeps every file, numbered as gallery-dl numbers them. - `discord_authors` in a source's config limits the walk to those posters (id, username or display name); the edit dialog has a field for it and no longer drops config keys it has no field for. - source.display_name (migration 0115), refreshed by every Discord walk, shows on Subscriptions in place of the two-id URL; a Discord post card names its channel from the record's `channel`. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,27 @@
|
||||
"""source.display_name — what a source is called on its platform.
|
||||
|
||||
#4481. A Discord source's URL is a server id and a channel id, so the
|
||||
Subscriptions page could only show two numbers. The Discord ingester already
|
||||
loads the server and channel names to walk them; it now keeps them here,
|
||||
refreshed on every walk. NULL until a walk has read it.
|
||||
|
||||
Revision ID: 0115
|
||||
Revises: 0114
|
||||
Create Date: 2026-09-28
|
||||
|
||||
"""
|
||||
import sqlalchemy as sa
|
||||
from alembic import op
|
||||
|
||||
revision = "0115"
|
||||
down_revision = "0114"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("source", sa.Column("display_name", sa.Text(), nullable=True))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("source", "display_name")
|
||||
@@ -46,6 +46,11 @@ class Source(Base):
|
||||
enabled: Mapped[bool] = mapped_column(Boolean, nullable=False, default=True, server_default="true")
|
||||
|
||||
config_overrides: Mapped[dict | None] = mapped_column(JSON, nullable=True)
|
||||
# alembic 0115: what the source is called on its platform, where the URL
|
||||
# alone doesn't say — a Discord link is two numbers. The ingester that
|
||||
# walks it refreshes it every walk, so a rename reads as the new name.
|
||||
# NULL until a walk has read it, and on platforms that don't write it.
|
||||
display_name: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
|
||||
last_checked_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
|
||||
last_error: Mapped[str | None] = mapped_column(Text, nullable=True)
|
||||
|
||||
@@ -32,6 +32,13 @@ Two deliberate departures, both about the walk order, neither about content:
|
||||
the walk. gallery-dl skips only nested channels; one private thread the
|
||||
token cannot read would otherwise stop every channel after it.
|
||||
|
||||
And one about content: a message is taken only if it has an image or video
|
||||
ATTACHMENT (`has_visual_attachment`). Discord is where creators chat as well
|
||||
as post, and the operator wants the art, not the conversation (2026-09-28): a
|
||||
text line, a lone archive or PSD, a pasted link's preview or a Tenor GIF is
|
||||
chat. A message that passes keeps every file gallery-dl would take, numbered
|
||||
as gallery-dl numbers them, so the on-disk names still match.
|
||||
|
||||
FC runs on a plain-HTTP homelab; nothing here uses a secure-context Web API.
|
||||
"""
|
||||
|
||||
@@ -79,6 +86,13 @@ _FORUM = frozenset({15, 16}) # forum, media: threads only
|
||||
_CATEGORY = 4
|
||||
_SERVER_WALK = _TEXT | _FORUM
|
||||
_EMBED_TYPES = frozenset({"image", "gifv", "video"})
|
||||
# What makes a message art rather than chat: an attached file FC imports as an
|
||||
# image or video. Mirrors importer.ALL_EXTS (not imported: that module is the
|
||||
# import pipeline, and this one is a network client).
|
||||
_VISUAL_EXTS = frozenset({
|
||||
"png", "jpg", "jpeg", "gif", "webp", "bmp", "tif", "tiff",
|
||||
"mp4", "mov", "avi", "mkv", "webm", "m4v", "wmv", "flv",
|
||||
})
|
||||
|
||||
_URL_RE = re.compile(
|
||||
r"^(?:https?://)?(?:www\.|ptb\.|canary\.)?discord(?:app)?\.com/channels/"
|
||||
@@ -163,6 +177,33 @@ def message_text(message: dict) -> str:
|
||||
return "\n".join(p for p in parts if p)
|
||||
|
||||
|
||||
def _message_and_snapshots(message: dict) -> list[dict]:
|
||||
"""The message itself, then each forwarded message it carries."""
|
||||
return [message] + [
|
||||
(s or {}).get("message") or {}
|
||||
for s in message.get("message_snapshots") or []
|
||||
if ((s or {}).get("message") or {}).get("type", 0) in MESSAGE_TYPES
|
||||
]
|
||||
|
||||
|
||||
def _is_visual(attachment: dict) -> bool:
|
||||
ctype = (attachment.get("content_type") or "").lower()
|
||||
if ctype.startswith(("image/", "video/")):
|
||||
return True
|
||||
name = attachment.get("filename") or attachment.get("url") or ""
|
||||
return nameext_from_url(name)[1] in _VISUAL_EXTS
|
||||
|
||||
|
||||
def has_visual_attachment(message: dict) -> bool:
|
||||
"""Does this message, or a message it forwards, have an image or video
|
||||
attached? The one test for "art, not chat" — see the module docstring."""
|
||||
return any(
|
||||
att.get("url") and _is_visual(att)
|
||||
for snap in _message_and_snapshots(message)
|
||||
for att in snap.get("attachments") or []
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class MediaItem:
|
||||
"""One file of a Discord message. `filename`/`extension` are gallery-dl's
|
||||
@@ -212,6 +253,8 @@ class DiscordClient:
|
||||
self._server: dict = {}
|
||||
self._channels: dict[str, dict] = {}
|
||||
self._skip_feed = False
|
||||
# Whose messages the walk takes (`only_from`); empty takes everyone's.
|
||||
self._authors: frozenset[str] = frozenset()
|
||||
|
||||
# -- request -----------------------------------------------------------
|
||||
|
||||
@@ -373,6 +416,43 @@ class DiscordClient:
|
||||
if meta["channel_type"] in _SERVER_WALK:
|
||||
yield from self._feeds(meta["channel_id"], safe=True)
|
||||
|
||||
def only_from(self, authors) -> None:
|
||||
"""Take only messages posted by these people: user ids, usernames or
|
||||
display names, any case. A creator's server is full of other members
|
||||
posting their own pictures; the source is subscribed to the creator.
|
||||
Empty or None takes everyone's, as before."""
|
||||
self._authors = frozenset(
|
||||
str(a).strip().lower() for a in authors or () if str(a).strip()
|
||||
)
|
||||
|
||||
def _author_wanted(self, message: dict) -> bool:
|
||||
if not self._authors:
|
||||
return True
|
||||
author = message.get("author") or {}
|
||||
names = (author.get("id"), author.get("username"), author.get("global_name"))
|
||||
return any(str(n).lower() in self._authors for n in names if n)
|
||||
|
||||
def source_label(self, url: str) -> str | None:
|
||||
"""What the source is called in Discord — `Server · #channel`, a
|
||||
thread as `#parent › thread`, a whole server by its name — from the
|
||||
metadata the walk loaded. None when the walk never got that far."""
|
||||
try:
|
||||
_, channel_id = parse_source_url(url)
|
||||
except DiscordAPIError:
|
||||
return None
|
||||
server = self._server.get("server") or None
|
||||
if channel_id is None:
|
||||
return server
|
||||
meta = self._channels.get(channel_id) or {}
|
||||
name = meta.get("channel")
|
||||
if not name:
|
||||
return None
|
||||
if name != "DMs":
|
||||
name = f"#{name}"
|
||||
if meta.get("is_thread") and meta.get("parent"):
|
||||
name = f"#{meta['parent']} › {meta['channel']}"
|
||||
return f"{server} · {name}" if server else name
|
||||
|
||||
def skip_feed(self) -> None:
|
||||
"""Optional core seam (#4413): end the current channel and go on to the
|
||||
next one. A tick's early-out means THIS channel has nothing new, not
|
||||
@@ -440,6 +520,8 @@ class DiscordClient:
|
||||
for message in messages:
|
||||
if message.get("type") not in MESSAGE_TYPES:
|
||||
continue
|
||||
if not self._author_wanted(message):
|
||||
continue
|
||||
message["_meta"] = meta
|
||||
yield message, meta, page_cursor
|
||||
if self._skip_feed:
|
||||
@@ -454,13 +536,14 @@ class DiscordClient:
|
||||
def extract_media(post: dict, included: dict | None = None) -> list[MediaItem]:
|
||||
"""gallery-dl's file list for one message: attachments, then the first
|
||||
of video/image/thumbnail `proxy_url` of each file-bearing embed, then
|
||||
the same for every forwarded snapshot; numbered from 1 across them."""
|
||||
the same for every forwarded snapshot; numbered from 1 across them.
|
||||
|
||||
Empty for a message with no image or video attached: that is chat,
|
||||
whatever else it carries (`has_visual_attachment`)."""
|
||||
if not has_visual_attachment(post):
|
||||
return []
|
||||
mid = str(post.get("id") or "")
|
||||
snapshots = [post] + [
|
||||
(s or {}).get("message") or {}
|
||||
for s in post.get("message_snapshots") or []
|
||||
if ((s or {}).get("message") or {}).get("type", 0) in MESSAGE_TYPES
|
||||
]
|
||||
snapshots = _message_and_snapshots(post)
|
||||
found: list[tuple[str, str, str | None]] = []
|
||||
for snap in snapshots:
|
||||
for att in snap.get("attachments") or []:
|
||||
@@ -498,10 +581,10 @@ class DiscordClient:
|
||||
"""`(message:<id>, <id>)` — gates the message record through the seen
|
||||
ledger, like `post:<id>` on the other platforms.
|
||||
|
||||
None for a message with no files. gallery-dl wrote a sidecar only
|
||||
beside a file, so a text-only chat line never became a post, and the
|
||||
drop grouping (discord_grouping) is built on that: a channel's chatter
|
||||
recorded as posts would bury the drops it exists to surface."""
|
||||
None for a message `extract_media` takes nothing from, i.e. one with
|
||||
no image or video attached. The drop grouping (discord_grouping) is
|
||||
built on that: a channel's chatter recorded as posts would bury the
|
||||
drops it exists to surface."""
|
||||
mid = post.get("id")
|
||||
mid = str(mid) if mid is not None else ""
|
||||
if not mid or not cls.extract_media(post):
|
||||
|
||||
@@ -13,6 +13,11 @@ Two things differ from the cookie platforms:
|
||||
something; a Discord drop is routinely files and nothing else, so on
|
||||
Discord that is an ordinary backfill, not a broken parser.
|
||||
|
||||
And two things only a Discord source has: `discord_authors` in its
|
||||
config_overrides limits it to the creator's own messages (a server's other
|
||||
members post pictures too), and it learns its own name — the server and
|
||||
channel it walks — into `source.display_name`.
|
||||
|
||||
`campaign_id` is the source URL (a server, channel, thread or category link).
|
||||
FC runs on a plain-HTTP homelab; nothing here uses a secure-context Web API.
|
||||
"""
|
||||
@@ -20,15 +25,23 @@ FC runs on a plain-HTTP homelab; nothing here uses a secure-context Web API.
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from collections.abc import Callable
|
||||
from pathlib import Path
|
||||
|
||||
from ..models import DiscordFailedMedia, DiscordSeenMedia
|
||||
from sqlalchemy import select, update
|
||||
|
||||
from ..models import DiscordFailedMedia, DiscordSeenMedia, Source
|
||||
from .discord_client import DiscordAPIError, DiscordClient, MediaItem
|
||||
from .discord_downloader import DiscordDownloader
|
||||
from .ingest_core import Ingester
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_LEDGER_KEY_MAX = 128
|
||||
# `config_overrides` key: the people whose messages this source takes (user
|
||||
# ids, usernames or display names). Absent or empty takes everyone's.
|
||||
AUTHORS_KEY = "discord_authors"
|
||||
|
||||
|
||||
def _ledger_key(media: MediaItem) -> str:
|
||||
@@ -74,6 +87,46 @@ class DiscordIngester(Ingester):
|
||||
body_canary=False,
|
||||
)
|
||||
|
||||
def run(self, **kwargs):
|
||||
"""The core walk, bracketed by the two things only a Discord source
|
||||
has: whose messages it takes, read before the walk, and the name of
|
||||
what it walks, written after it (the walk is what loads that name)."""
|
||||
source_id = kwargs["source_id"]
|
||||
self.client.only_from(self._source_authors(source_id))
|
||||
try:
|
||||
return super().run(**kwargs)
|
||||
finally:
|
||||
self._record_display_name(source_id, kwargs["campaign_id"])
|
||||
|
||||
def _source_authors(self, source_id: int) -> list:
|
||||
if self.session_factory is None:
|
||||
return []
|
||||
with self.session_factory() as session:
|
||||
overrides = session.execute(
|
||||
select(Source.config_overrides).where(Source.id == source_id)
|
||||
).scalar_one_or_none() or {}
|
||||
authors = overrides.get(AUTHORS_KEY) or []
|
||||
return authors if isinstance(authors, list) else []
|
||||
|
||||
def _record_display_name(self, source_id: int, url: str) -> None:
|
||||
"""Refreshed on every walk, never written once: a renamed channel
|
||||
should read as its new name. A name the walk couldn't read leaves the
|
||||
stored one alone, and a failure here never fails the walk."""
|
||||
label = self.client.source_label(url)
|
||||
if not label or self.session_factory is None:
|
||||
return
|
||||
try:
|
||||
with self.session_factory() as session:
|
||||
session.execute(
|
||||
update(Source)
|
||||
.where(Source.id == source_id)
|
||||
.where(Source.display_name.is_distinct_from(label))
|
||||
.values(display_name=label)
|
||||
)
|
||||
session.commit()
|
||||
except Exception as exc: # a name is decoration — never fail the walk
|
||||
log.warning("Discord: couldn't record source %s's name: %s", source_id, exc)
|
||||
|
||||
|
||||
async def verify_discord_credential(url: str, auth_token: str | None) -> tuple[bool | None, str]:
|
||||
"""The uniform `(ok, message)` probe: is the token valid, and can its
|
||||
|
||||
@@ -479,6 +479,10 @@ class PostFeedService:
|
||||
# the UI uses this to explain why a post it linked to is not in the
|
||||
# stream.
|
||||
"absorbed_by_post_id": post.absorbed_by_post_id,
|
||||
# #4481. The Discord channel a message was posted in: both the native
|
||||
# post record and gallery-dl's sidecars carry it as `channel`. None
|
||||
# on every other platform, and on a Discord post that never said.
|
||||
"channel": _discord_channel(post, source),
|
||||
"artist": {"id": artist.id, "name": artist.name, "slug": artist.slug},
|
||||
"source": (
|
||||
{"id": source.id, "platform": source.platform}
|
||||
@@ -488,3 +492,13 @@ class PostFeedService:
|
||||
"thumbnails_more": thumbs_entry["more"],
|
||||
"attachments": atts_map.get(post.id, []),
|
||||
}
|
||||
|
||||
|
||||
def _discord_channel(post: Post, source: Source | None) -> str | None:
|
||||
if source is None or source.platform != "discord":
|
||||
return None
|
||||
raw = post.raw_metadata if isinstance(post.raw_metadata, dict) else {}
|
||||
channel = raw.get("channel")
|
||||
if not isinstance(channel, str):
|
||||
return None
|
||||
return channel.strip() or None
|
||||
|
||||
@@ -68,6 +68,9 @@ class SourceRecord:
|
||||
artist_slug: str
|
||||
platform: str
|
||||
url: str
|
||||
# alembic 0115: the name the platform gives it, where the URL is opaque
|
||||
# (a Discord link is two ids). None until a walk has read it.
|
||||
display_name: str | None
|
||||
enabled: bool
|
||||
config_overrides: dict | None
|
||||
last_checked_at: str | None
|
||||
@@ -110,6 +113,7 @@ class SourceRecord:
|
||||
"artist_slug": self.artist_slug,
|
||||
"platform": self.platform,
|
||||
"url": self.url,
|
||||
"display_name": self.display_name,
|
||||
"enabled": self.enabled,
|
||||
"config_overrides": self.config_overrides,
|
||||
"last_checked_at": self.last_checked_at,
|
||||
@@ -301,6 +305,7 @@ class SourceService:
|
||||
artist_slug=artist.slug,
|
||||
platform=source.platform,
|
||||
url=source.url,
|
||||
display_name=source.display_name,
|
||||
enabled=source.enabled,
|
||||
config_overrides=source.config_overrides,
|
||||
last_checked_at=source.last_checked_at.isoformat() if source.last_checked_at else None,
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
:to="{ name: 'artist', params: { slug: post.artist.slug } }"
|
||||
class="fc-post-card__artist"
|
||||
>{{ post.artist.name }}</RouterLink>
|
||||
<span v-if="post.channel" class="fc-post-card__meta">#{{ post.channel }}</span>
|
||||
<span class="fc-post-card__date" :title="absoluteDate">{{ relativeDate }}</span>
|
||||
<span v-if="totalImages" class="fc-post-card__meta">
|
||||
· {{ totalImages }} image{{ totalImages === 1 ? '' : 's' }}
|
||||
|
||||
@@ -16,7 +16,7 @@
|
||||
<a
|
||||
:href="source.url" target="_blank" rel="noopener"
|
||||
class="fc-source-card__url" @click.stop
|
||||
>{{ source.url }}</a>
|
||||
>{{ source.display_name || source.url }}</a>
|
||||
<v-btn
|
||||
icon="mdi-pencil" size="x-small" variant="text"
|
||||
@click.stop="$emit('edit', source)"
|
||||
|
||||
@@ -44,6 +44,14 @@
|
||||
v-model="structuredSince" label="Skip posts older than (YYYY-MM-DD)"
|
||||
placeholder="2024-01-01" hide-details class="mt-2"
|
||||
/>
|
||||
<!-- A creator's server is full of other members posting their own
|
||||
pictures; the source is subscribed to the creator (#4481). -->
|
||||
<v-text-field
|
||||
v-if="platform === 'discord'"
|
||||
v-model="structuredAuthors" label="Only posts by (Discord names or ids)"
|
||||
placeholder="Todding" hint="Comma-separated. Empty takes everyone's."
|
||||
persistent-hint class="mt-2"
|
||||
/>
|
||||
<p class="text-caption mt-2" style="opacity: 0.75">
|
||||
More per-platform fields land here over time. Use Advanced JSON for everything else.
|
||||
</p>
|
||||
@@ -98,6 +106,24 @@ const urlError = ref('')
|
||||
const configTab = ref('structured')
|
||||
const structuredVideos = ref(true)
|
||||
const structuredSince = ref('')
|
||||
const structuredAuthors = ref('')
|
||||
// Keys the structured view has no field for, carried through its saves so
|
||||
// switching tabs never drops what the JSON view set.
|
||||
const otherConfig = ref({})
|
||||
const STRUCTURED_KEYS = ['videos', 'since', 'discord_authors']
|
||||
|
||||
function splitAuthors(txt) {
|
||||
return (txt || '').split(',').map(s => s.trim()).filter(Boolean)
|
||||
}
|
||||
|
||||
function takeConfig(co) {
|
||||
structuredVideos.value = co.videos !== false
|
||||
structuredSince.value = co.since ?? ''
|
||||
structuredAuthors.value = Array.isArray(co.discord_authors) ? co.discord_authors.join(', ') : ''
|
||||
otherConfig.value = Object.fromEntries(
|
||||
Object.entries(co).filter(([k]) => !STRUCTURED_KEYS.includes(k)),
|
||||
)
|
||||
}
|
||||
const jsonText = ref('{}')
|
||||
const jsonError = ref('')
|
||||
|
||||
@@ -105,13 +131,15 @@ const busy = ref(false)
|
||||
|
||||
// Sync config_overrides between the two views.
|
||||
const config = computed(() => {
|
||||
const out = {}
|
||||
const out = { ...otherConfig.value }
|
||||
if (!structuredVideos.value) out.videos = false
|
||||
if (structuredSince.value) out.since = structuredSince.value
|
||||
const authors = splitAuthors(structuredAuthors.value)
|
||||
if (authors.length) out.discord_authors = authors
|
||||
return out
|
||||
})
|
||||
|
||||
watch([structuredVideos, structuredSince], () => {
|
||||
watch([structuredVideos, structuredSince, structuredAuthors], () => {
|
||||
if (configTab.value === 'structured') {
|
||||
jsonText.value = JSON.stringify(config.value, null, 2)
|
||||
jsonError.value = ''
|
||||
@@ -128,8 +156,7 @@ watch(jsonText, (txt) => {
|
||||
}
|
||||
jsonError.value = ''
|
||||
// Reflect recognized keys into the structured view.
|
||||
structuredVideos.value = parsed.videos !== false
|
||||
structuredSince.value = parsed.since ?? ''
|
||||
takeConfig(parsed)
|
||||
} catch {
|
||||
jsonError.value = 'Invalid JSON'
|
||||
}
|
||||
@@ -146,14 +173,13 @@ watch(() => props.modelValue, async (open) => {
|
||||
url.value = props.source.url
|
||||
enabled.value = props.source.enabled
|
||||
const co = props.source.config_overrides || {}
|
||||
structuredVideos.value = co.videos !== false
|
||||
structuredSince.value = co.since ?? ''
|
||||
takeConfig(co)
|
||||
jsonText.value = JSON.stringify(co, null, 2)
|
||||
artistChoice.value = { id: props.source.artist_id, name: props.source.artist_name }
|
||||
} else {
|
||||
platform.value = platformsStore.list[0]?.key || 'patreon'
|
||||
url.value = ''; enabled.value = true
|
||||
structuredVideos.value = true; structuredSince.value = ''
|
||||
takeConfig({})
|
||||
jsonText.value = '{}'
|
||||
artistChoice.value = props.initialArtist
|
||||
? { id: props.initialArtist.id, name: props.initialArtist.name }
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
<div class="fc-source-row__url-wrap">
|
||||
<a :href="source.url" target="_blank" rel="noopener" class="fc-source-row__url"
|
||||
@click.stop>
|
||||
{{ source.url }}
|
||||
{{ source.display_name || source.url }}
|
||||
</a>
|
||||
<!-- Edit sits next to the source identity (operator-requested), not in
|
||||
the action cluster where it was easy to fat-finger Remove. -->
|
||||
|
||||
@@ -108,7 +108,7 @@
|
||||
v-if="item.singleSource"
|
||||
:href="item.singleSource.url" target="_blank" rel="noopener"
|
||||
class="fc-subs__sub-url" @click.stop
|
||||
>{{ item.singleSource.url }}</a>
|
||||
>{{ item.singleSource.display_name || item.singleSource.url }}</a>
|
||||
</template>
|
||||
|
||||
<template #item.platforms="{ item }">
|
||||
@@ -517,6 +517,7 @@ const filteredGroups = computed(() => {
|
||||
g.sources.some(
|
||||
(s) =>
|
||||
(s.url || '').toLowerCase().includes(q) ||
|
||||
(s.display_name || '').toLowerCase().includes(q) ||
|
||||
(s.platform || '').toLowerCase().includes(q),
|
||||
),
|
||||
)
|
||||
|
||||
@@ -153,9 +153,11 @@ def test_an_embeds_identity_ignores_its_signature():
|
||||
def embed(sig):
|
||||
return {"type": "image", "image": {"proxy_url": f"https://media/p/x.png?ex={sig}"}}
|
||||
|
||||
one = DiscordClient.extract_media(_msg(9, embeds=[embed("a")]))[0].media_id
|
||||
two = DiscordClient.extract_media(_msg(9, embeds=[embed("b")]))[0].media_id
|
||||
assert one == two and len(one) <= 33
|
||||
art = [{"url": "https://cdn/a/1.png"}]
|
||||
one = DiscordClient.extract_media(_msg(9, attachments=art, embeds=[embed("a")]))[1]
|
||||
two = DiscordClient.extract_media(_msg(9, attachments=art, embeds=[embed("b")]))[1]
|
||||
assert one.kind == two.kind == "embed"
|
||||
assert one.media_id == two.media_id and len(one.media_id) <= 33
|
||||
|
||||
|
||||
def test_post_seams():
|
||||
@@ -165,11 +167,52 @@ def test_post_seams():
|
||||
|
||||
|
||||
def test_a_text_only_message_is_not_a_post():
|
||||
"""gallery-dl never made one: chat lines would bury the drops."""
|
||||
"""Chat lines would bury the drops."""
|
||||
assert DiscordClient.post_record_key(_msg(6, content="brb")) is None
|
||||
assert DiscordClient.post_meta(_msg(1))["date"].startswith("2026-09-20")
|
||||
|
||||
|
||||
def test_a_message_without_an_attached_image_is_chat():
|
||||
"""Operator 2026-09-28: only content with an image attached. A lone
|
||||
archive, a link preview or a Tenor GIF takes nothing and records nothing."""
|
||||
chat = [
|
||||
_msg(1, content="the stash", attachments=[
|
||||
{"url": "https://cdn/a/Links_Stash.rar", "content_type": "application/x-rar"},
|
||||
]),
|
||||
_msg(2, content="look", embeds=[
|
||||
{"type": "image", "image": {"proxy_url": "https://media/p/x.png"}},
|
||||
]),
|
||||
_msg(3, content="lol", embeds=[
|
||||
{"type": "gifv", "video": {"proxy_url": "https://media/p/t.mp4"}},
|
||||
]),
|
||||
_msg(4, content="wip", attachments=[{"url": "https://cdn/a/wip.psd"}]),
|
||||
]
|
||||
for m in chat:
|
||||
assert DiscordClient.extract_media(m) == []
|
||||
assert DiscordClient.post_record_key(m) is None
|
||||
|
||||
|
||||
def test_an_attached_image_takes_every_file_numbered_as_gallery_dl_does():
|
||||
"""The gate decides WHETHER a message is taken, never which of its files:
|
||||
the rar beside the image keeps its number, so on-disk names still match."""
|
||||
m = _msg(7, attachments=[
|
||||
{"url": "https://cdn/a/pack.rar"},
|
||||
{"url": "https://cdn/a/noext", "content_type": "image/png"},
|
||||
])
|
||||
items = DiscordClient.extract_media(m)
|
||||
assert [(i.num, i.filename) for i in items] == [(1, "pack"), (2, "noext")]
|
||||
assert DiscordClient.post_record_key(m) == ("message:7", "7")
|
||||
video = _msg(8, attachments=[{"url": "https://cdn/a/clip.MP4?ex=1"}])
|
||||
assert len(DiscordClient.extract_media(video)) == 1
|
||||
|
||||
|
||||
def test_a_forwarded_image_counts_as_attached():
|
||||
m = _msg(9, message_snapshots=[
|
||||
{"message": {"type": 0, "attachments": [{"url": "https://cdn/a/fwd.png"}]}},
|
||||
])
|
||||
assert DiscordClient.post_record_key(m) == ("message:9", "9")
|
||||
|
||||
|
||||
# -- the walk --------------------------------------------------------------------
|
||||
|
||||
def test_a_channel_pages_newest_first_and_skips_system_messages():
|
||||
@@ -206,6 +249,50 @@ def test_messages_carry_server_and_channel_metadata():
|
||||
assert meta["parent"] == "Art"
|
||||
|
||||
|
||||
def _one_channel(messages):
|
||||
return {
|
||||
"/guilds/1": _ok({"id": "1", "name": "Todding's Server"}),
|
||||
"/guilds/1/channels": _ok([
|
||||
{"id": "4", "type": 4, "name": "Todding"},
|
||||
{"id": "2", "type": 0, "name": "banana-land", "parent_id": "4"},
|
||||
]),
|
||||
("/channels/2/messages", None): _ok(messages),
|
||||
("/channels/2/threads/search", 0): _ok({"threads": []}),
|
||||
}
|
||||
|
||||
|
||||
def test_only_from_takes_the_creators_messages_and_no_one_elses():
|
||||
"""Operator 2026-09-28: in the creator's server, another member posting a
|
||||
meme is not the creator's art. Matched by id, username or display name."""
|
||||
creator = {"id": "7", "username": "todding", "global_name": "Todding"}
|
||||
other = {"id": "8", "username": "jakeboii", "global_name": "Jake Boii"}
|
||||
msgs = [_msg(3, author=creator), _msg(2, author=other), _msg(1, author=creator)]
|
||||
url = "https://discord.com/channels/1/2"
|
||||
|
||||
everyone = _client(_one_channel([dict(m) for m in msgs]))
|
||||
assert [mid for mid, _ in _ids(everyone, url)] == ["3", "2", "1"]
|
||||
for who in (["Todding"], ["TODDING "], ["7"]):
|
||||
client = _client(_one_channel([dict(m) for m in msgs]))
|
||||
client.only_from(who)
|
||||
assert [mid for mid, _ in _ids(client, url)] == ["3", "1"]
|
||||
client = _client(_one_channel([dict(m) for m in msgs]))
|
||||
client.only_from([])
|
||||
assert len(_ids(client, url)) == 3
|
||||
|
||||
|
||||
def test_the_source_label_names_the_server_and_channel_the_walk_read():
|
||||
client = _client(_one_channel([]))
|
||||
url = "https://discord.com/channels/1/2"
|
||||
assert client.source_label(url) is None # nothing walked yet
|
||||
list(client.iter_posts(url))
|
||||
assert client.source_label(url) == "Todding's Server · #banana-land"
|
||||
assert client.source_label("https://discord.com/channels/1") == "Todding's Server"
|
||||
client._channels["9"] = {"channel": "wip", "is_thread": True, "parent": "banana-land"}
|
||||
assert client.source_label("https://discord.com/channels/1/9") == (
|
||||
"Todding's Server · #banana-land › wip"
|
||||
)
|
||||
|
||||
|
||||
def test_a_server_walks_text_then_threads_newest_created_first_and_skips_private():
|
||||
routes = {
|
||||
"/guilds/1": _ok({"id": "1", "name": "S"}),
|
||||
|
||||
@@ -283,6 +283,27 @@ async def test_scroll_item_shape_minimal(db):
|
||||
assert "description_full" not in item
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_discord_post_names_its_channel(db):
|
||||
"""#4481: the card says which channel a message came from. Read from the
|
||||
record's `channel`; never on another platform, even with the same key."""
|
||||
artist = await _seed_artist(db, "todding-ch")
|
||||
dsrc = await _seed_source(db, artist.id, "discord", "https://discord.com/channels/1/2")
|
||||
psrc = await _seed_source(db, artist.id, "patreon", "https://p/todding-ch")
|
||||
now = datetime.now(UTC)
|
||||
d = await _seed_post(db, dsrc.id, external_id="D1", post_date=now)
|
||||
d.raw_metadata = {"category": "discord", "channel": "banana-land"}
|
||||
blank = await _seed_post(db, dsrc.id, external_id="D2", post_date=now)
|
||||
blank.raw_metadata = {"category": "discord", "channel": " "}
|
||||
p = await _seed_post(db, psrc.id, external_id="P1", post_date=now)
|
||||
p.raw_metadata = {"channel": "not-discord"}
|
||||
await db.commit()
|
||||
|
||||
page = await PostFeedService(db).scroll(cursor=None, limit=10, artist_id=artist.id)
|
||||
by_id = {it["external_post_id"]: it["channel"] for it in page["items"]}
|
||||
assert by_id == {"D1": "banana-land", "D2": None, "P1": None}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_scroll_surfaces_translation_fields(db):
|
||||
# #143: a translated post exposes the translated title/description + source
|
||||
|
||||
Reference in New Issue
Block a user