AI Implementation feature(1087): Mastodon Post Composition and Publishing #6

Merged
m0rph3us1987 merged 2 commits from feature-1087-1785868532240 into staging 2026-08-04 18:57:46 +00:00
9 changed files with 425 additions and 26 deletions
Showing only changes of commit 038e10b87d - Show all commits
+1 -1
View File
@@ -5,7 +5,7 @@ MASTODON_ACCESS_TOKEN=replace-me
VISIBILITY=public VISIBILITY=public
SITE_URL=https://blog.example.com SITE_URL=https://blog.example.com
HASHTAGS=#throwback,#10backward HASHTAGS=#throwback,#10backward
THROWNBACK_PREFIX=Throwback: THROWNBACK_PREFIX=Heute vor 10 Jahren:
MAX_RETRIES=3 MAX_RETRIES=3
RUN_AT=09:00 RUN_AT=09:00
TZ=Europe/Berlin TZ=Europe/Berlin
+1 -1
View File
@@ -26,7 +26,7 @@ OPTIONAL_KEYS: tuple[str, ...] = ("BLOG_REPO_URL", "BLOG_DIR")
DEFAULTS = { DEFAULTS = {
"VISIBILITY": "public", "VISIBILITY": "public",
"THROWNBACK_PREFIX": "Throwback:", "THROWNBACK_PREFIX": "Heute vor 10 Jahren:",
"MAX_RETRIES": "3", "MAX_RETRIES": "3",
"TZ": "Europe/Berlin", "TZ": "Europe/Berlin",
"RUN_AT": "09:00", "RUN_AT": "09:00",
+19 -19
View File
@@ -13,17 +13,17 @@ from .logging_setup import (
log_run_summary, log_run_summary,
log_startup, log_startup,
) )
from .matching import iter_anniversary_paths from .matching import MatchedPost, find_anniversary_matches
from .publishing import publish_mastodon
from .state import PostedStore from .state import PostedStore
def _iter_candidates(config: Config) -> Iterable[str]: def _iter_candidates(config: Config) -> Iterable[MatchedPost]:
"""Yield candidate post identifiers (the relative path under """Yield candidate :class:`MatchedPost` objects whose anniversary is
``_posts/blog/``) for posts whose anniversary is exactly 10 years exactly 10 years before today.
before today.
""" """
post_root = config.blog_dir / "_posts" / "blog" post_root = config.blog_dir / "_posts" / "blog"
return iter_anniversary_paths(post_root, config.site_url) return find_anniversary_matches(post_root, config.site_url)
def _run_once(config: Config) -> tuple[int, int, int, int, list[str]]: def _run_once(config: Config) -> tuple[int, int, int, int, list[str]]:
@@ -40,24 +40,24 @@ def _run_once(config: Config) -> tuple[int, int, int, int, list[str]]:
store = PostedStore(config.data_dir) store = PostedStore(config.data_dir)
scanned = 0 candidates = list(_iter_candidates(config))
matched = 0 scanned = len(candidates)
posted = 0 matched = len(candidates)
skipped = 0
posted_ids: list[str] = []
for candidate_id in _iter_candidates(config): unposted: list[MatchedPost] = []
scanned += 1 for match in candidates:
matched += 1 if store.is_posted(match.path):
if store.is_posted(candidate_id):
skipped += 1
continue continue
posted_ids.append(candidate_id) unposted.append(match)
posted += 1
if posted_ids: skipped = matched - len(unposted)
posted_ids = [m.path for m in unposted]
if unposted:
publish_mastodon(config, unposted)
store.mark_posted_many(posted_ids) store.mark_posted_many(posted_ids)
posted = len(unposted)
return scanned, matched, posted, skipped, posted_ids return scanned, matched, posted, skipped, posted_ids
+18 -1
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
import calendar import calendar
import logging import logging
import re import re
import unicodedata
from dataclasses import dataclass from dataclasses import dataclass
from datetime import date, datetime from datetime import date, datetime
from pathlib import Path from pathlib import Path
@@ -15,6 +16,8 @@ _LOGGER = logging.getLogger("tenbackward.matching")
_FILENAME_PATTERN = re.compile(r"^(\d{4})-(\d{2})-(\d{2})-(.+)$") _FILENAME_PATTERN = re.compile(r"^(\d{4})-(\d{2})-(\d{2})-(.+)$")
_NON_ALNUM_RE = re.compile(r"[^a-z0-9]+")
_DEFAULT_TZ = "Europe/Berlin" _DEFAULT_TZ = "Europe/Berlin"
@@ -26,6 +29,19 @@ class MatchedPost:
url: str url: str
def slugify_title(title: str) -> str:
"""Return a URL-safe slug derived from ``title``.
Used by the matcher to derive the slug component of
:attr:`MatchedPost.url` from the resolved post title.
"""
normalized = unicodedata.normalize("NFKD", title)
ascii_only = normalized.encode("ascii", "ignore").decode("ascii")
lowered = ascii_only.lower()
dashed = _NON_ALNUM_RE.sub("-", lowered)
return dashed.strip("-")
def _today_in_berlin(tz_name: str = _DEFAULT_TZ) -> date: def _today_in_berlin(tz_name: str = _DEFAULT_TZ) -> date:
return datetime.now(ZoneInfo(tz_name)).date() return datetime.now(ZoneInfo(tz_name)).date()
@@ -152,7 +168,7 @@ def _process_file(
path=rel, path=rel,
title=title, title=title,
date=post_date, date=post_date,
url=_build_url(site_url, post_date, fn_slug), url=_build_url(site_url, post_date, slugify_title(title)),
) )
@@ -206,4 +222,5 @@ __all__ = [
"MatchedPost", "MatchedPost",
"find_anniversary_matches", "find_anniversary_matches",
"iter_anniversary_paths", "iter_anniversary_paths",
"slugify_title",
] ]
+141
View File
@@ -0,0 +1,141 @@
"""Mastodon publishing helpers and the API success boundary.
The :func:`publish_mastodon` function is the single boundary between the
pipeline orchestrator and the Mastodon HTTP API. Status composition and
length validation are pure and isolated from network access so they can
be tested without a server.
"""
from __future__ import annotations
from typing import Callable
from mastodon import Mastodon
from .config import Config
from .matching import MatchedPost, slugify_title
MASTODON_STATUS_LIMIT = 500
class PublishError(RuntimeError):
"""Raised when status composition, validation, or the Mastodon API
call fails. The pipeline orchestrator treats this like any other
pipeline error and lets the retry budget decide whether to give up.
"""
def _normalize_hashtags(hashtags: str) -> str:
parts = [token.strip() for token in (hashtags or "").split(",")]
parts = [token for token in parts if token]
return " ".join(parts)
def build_status_text(
prefix: str,
posts: list[MatchedPost],
hashtags: str,
) -> str:
"""Compose a single Mastodon status string for ``posts``.
The format is::
{prefix}
{title1}
{url1}
{title2}
{url2}
...
{hashtags}
Posts are sorted by ``(date, path)`` for deterministic output
regardless of the upstream ordering.
"""
if not posts:
raise PublishError("empty_posts: cannot compose status without posts")
ordered = sorted(posts, key=lambda m: (m.date, m.path))
blocks: list[str] = []
for match in ordered:
blocks.append(f"{match.title}\n{match.url}")
body = "\n\n".join(blocks)
lines: list[str] = [prefix.strip(), body]
tag_line = _normalize_hashtags(hashtags)
if tag_line:
lines.append(tag_line)
return "\n\n".join(lines) + "\n"
def validate_status(status: str, limit: int = MASTODON_STATUS_LIMIT) -> None:
"""Raise :class:`PublishError` when ``status`` exceeds ``limit``.
Never truncates: the spec requires failing safely rather than
shortening content.
"""
if len(status) > limit:
raise PublishError(
f"status_too_long: len={len(status)} limit={limit}"
)
def _post_status_via_mastodon_py(
status: str,
*,
base_url: str,
access_token: str,
visibility: str,
client_factory: Callable[..., Mastodon] | None = None,
) -> None:
factory = client_factory if client_factory is not None else Mastodon
client = factory(access_token=access_token, api_base_url=base_url)
client.status_post(status, visibility=visibility)
def publish_mastodon(
config: Config,
posts: list[MatchedPost],
*,
client_factory: Callable[..., Mastodon] | None = None,
) -> str:
"""Compose, validate, and publish ``posts`` to Mastodon.
Returns the composed status text on success. Raises
:class:`PublishError` on any failure (composition, length, or API).
The original exception is chained via ``raise ... from exc`` so the
caller can inspect the underlying cause.
"""
try:
status = build_status_text(
config.throwback_prefix, posts, config.hashtags
)
validate_status(status)
_post_status_via_mastodon_py(
status,
base_url=config.mastodon_base_url,
access_token=config.mastodon_access_token,
visibility=config.visibility,
client_factory=client_factory,
)
except PublishError:
raise
except Exception as exc: # noqa: BLE001 — third-party boundary
raise PublishError(
f"publish_failed: {type(exc).__name__}: {exc}"
) from exc
return status
__all__ = [
"MASTODON_STATUS_LIMIT",
"PublishError",
"build_status_text",
"publish_mastodon",
"slugify_title",
"validate_status",
]
+1 -1
View File
@@ -142,7 +142,7 @@ def test_load_config_applies_optional_defaults(env_setup, monkeypatch) -> None:
monkeypatch.delenv("TZ", raising=False) monkeypatch.delenv("TZ", raising=False)
config = load_config() config = load_config()
assert config.throwback_prefix == "Throwback:" assert config.throwback_prefix == "Heute vor 10 Jahren:"
assert config.max_retries == 3 assert config.max_retries == 3
assert config.tz == "Europe/Berlin" assert config.tz == "Europe/Berlin"
assert config.visibility == "public" assert config.visibility == "public"
+3
View File
@@ -11,6 +11,7 @@ from tenbackward.matching import (
MatchedPost, MatchedPost,
find_anniversary_matches, find_anniversary_matches,
iter_anniversary_paths, iter_anniversary_paths,
slugify_title,
) )
@@ -64,6 +65,7 @@ def test_match_uses_frontmatter_date_and_title(tmp_path: Path) -> None:
assert match.title == "Foo" assert match.title == "Foo"
assert match.date == date(2014, 8, 4) assert match.date == date(2014, 8, 4)
assert match.url == "https://chaospott.de/2014/08/04/foo/" assert match.url == "https://chaospott.de/2014/08/04/foo/"
assert match.url.endswith(slugify_title(match.title) + "/")
def test_match_falls_back_to_filename_when_frontmatter_missing(tmp_path: Path) -> None: def test_match_falls_back_to_filename_when_frontmatter_missing(tmp_path: Path) -> None:
@@ -87,6 +89,7 @@ def test_match_falls_back_to_filename_when_frontmatter_missing(tmp_path: Path) -
assert match.title == "no-frontmatter" assert match.title == "no-frontmatter"
assert match.date == date(2015, 3, 10) assert match.date == date(2015, 3, 10)
assert match.url == "https://chaospott.de/2015/03/10/no-frontmatter/" assert match.url == "https://chaospott.de/2015/03/10/no-frontmatter/"
assert match.url.endswith(slugify_title(match.title) + "/")
def test_unpublished_post_is_skipped(tmp_path: Path) -> None: def test_unpublished_post_is_skipped(tmp_path: Path) -> None:
+191
View File
@@ -0,0 +1,191 @@
from __future__ import annotations
import os
from datetime import date
from pathlib import Path
from typing import Callable
import pytest
from tenbackward.config import load_config
from tenbackward.matching import MatchedPost
from tenbackward.publishing import (
MASTODON_STATUS_LIMIT,
PublishError,
build_status_text,
publish_mastodon,
slugify_title,
validate_status,
)
from tenbackward.state import PostedStore
def _make_match(path: str, title: str, day: int = 4) -> MatchedPost:
return MatchedPost(
path=path,
title=title,
date=date(2016, 8, day),
url=f"https://blog.example.com/2016/08/{day:02d}/{path.split('-', 3)[-1].removesuffix('.md')}/",
)
class _FakeMastodon:
"""Captures ``status_post`` calls and can be configured to raise."""
def __init__(self, *, raise_on_post: Exception | None = None) -> None:
self.raise_on_post = raise_on_post
self.calls: list[tuple[str, dict]] = []
def __call__(self, *, access_token: str, api_base_url: str) -> "_FakeMastodon":
self.access_token = access_token
self.api_base_url = api_base_url
return self
def status_post(self, status: str, **kwargs) -> None:
self.calls.append((status, kwargs))
if self.raise_on_post is not None:
raise self.raise_on_post
@pytest.fixture()
def full_config(env_setup, tmp_path: Path):
os.environ["DATA_DIR"] = str(tmp_path)
return load_config()
def test_slugify_title_basic_ascii() -> None:
assert slugify_title("Hello, World!") == "hello-world"
def test_slugify_title_collapses_repeats_and_trims() -> None:
assert slugify_title("!!!Foo---Bar???") == "foo-bar"
assert slugify_title("---trim---me---") == "trim-me"
def test_build_status_text_single_post_layout() -> None:
match = _make_match("2016/2016-08-04-foo.md", "Foo")
status = build_status_text(
"Heute vor 10 Jahren:", [match], "#throwback,#10backward"
)
assert status == (
"Heute vor 10 Jahren:\n\n"
"Foo\nhttps://blog.example.com/2016/08/04/foo/\n\n"
"#throwback #10backward\n"
)
def test_build_status_text_multiple_posts_combined() -> None:
match_a = _make_match("2016/2016-08-04-a.md", "Alpha")
match_b = _make_match("2016/2016-08-04-b.md", "Bravo")
status = build_status_text("P:", [match_a, match_b], "#x")
assert "Alpha" in status
assert "Bravo" in status
assert match_a.url in status
assert match_b.url in status
assert status.count("Alpha") == 1
assert status.count("Bravo") == 1
assert status.count(match_a.url) == 1
assert status.count(match_b.url) == 1
def test_build_status_text_empty_posts_raises() -> None:
with pytest.raises(PublishError):
build_status_text("P:", [], "#x")
def test_build_status_text_preserves_deterministic_order() -> None:
match_a = _make_match("2016/2016-08-05-a.md", "Alpha", day=5)
match_b = _make_match("2016/2016-08-04-b.md", "Bravo", day=4)
forward = build_status_text("P:", [match_a, match_b], "#x")
reverse = build_status_text("P:", [match_b, match_a], "#x")
assert forward == reverse
assert forward.index("Bravo") < forward.index("Alpha")
def test_validate_status_within_limit() -> None:
validate_status("a" * 480)
def test_validate_status_over_limit_raises_and_does_not_truncate() -> None:
long = "a" * (MASTODON_STATUS_LIMIT + 1)
with pytest.raises(PublishError):
validate_status(long)
assert len(long) == MASTODON_STATUS_LIMIT + 1
def test_publish_mastodon_success_invokes_client_with_composed_status(
full_config,
) -> None:
match = _make_match("2016/2016-08-04-foo.md", "Foo")
fake = _FakeMastodon()
result = publish_mastodon(full_config, [match], client_factory=fake)
assert result.endswith("\n")
assert fake.calls, "status_post must be invoked"
posted_status, posted_kwargs = fake.calls[0]
assert posted_status == result
assert posted_kwargs == {"visibility": "public"}
assert fake.api_base_url == "https://mastodon.example"
assert fake.access_token == "test-token"
def test_publish_mastodon_api_failure_raises_and_does_not_persist(
full_config, tmp_path: Path
) -> None:
match = _make_match("2016/2016-08-04-foo.md", "Foo")
fake = _FakeMastodon(raise_on_post=RuntimeError("boom"))
with pytest.raises(PublishError) as exc_info:
publish_mastodon(full_config, [match], client_factory=fake)
assert "publish_failed" in str(exc_info.value)
assert isinstance(exc_info.value.__cause__, RuntimeError)
store = PostedStore(tmp_path)
assert not store.is_posted(match.path)
def test_publish_mastodon_over_limit_raises_before_api_call(
full_config,
) -> None:
long_title = "T" * (MASTODON_STATUS_LIMIT + 1)
match = MatchedPost(
path="2016/2016-08-04-foo.md",
title=long_title,
date=date(2016, 8, 4),
url="https://blog.example.com/2016/08/04/foo/",
)
fake = _FakeMastodon()
with pytest.raises(PublishError):
publish_mastodon(full_config, [match], client_factory=fake)
assert fake.calls == []
def test_publish_mastodon_combined_status_for_multiple_matches(full_config) -> None:
match_a = _make_match("2016/2016-08-04-a.md", "Alpha")
match_b = _make_match("2016/2016-08-04-b.md", "Bravo")
fake = _FakeMastodon()
publish_mastodon(full_config, [match_a, match_b], client_factory=fake)
assert len(fake.calls) == 1
status, _ = fake.calls[0]
assert "Alpha" in status
assert "Bravo" in status
assert match_a.url in status
assert match_b.url in status
def test_publish_mastodon_uses_config_visibility(env_setup, tmp_path: Path) -> None:
os.environ["VISIBILITY"] = "unlisted"
os.environ["DATA_DIR"] = str(tmp_path)
config = load_config()
match = _make_match("2016/2016-08-04-foo.md", "Foo")
fake = _FakeMastodon()
publish_mastodon(config, [match], client_factory=fake)
assert fake.calls[0][1] == {"visibility": "unlisted"}
+49 -2
View File
@@ -8,9 +8,22 @@ from pathlib import Path
import pytest import pytest
from datetime import date
from tenbackward import main as main_module from tenbackward import main as main_module
from tenbackward.logging_setup import JsonFormatter from tenbackward.logging_setup import JsonFormatter
from tenbackward.main import main from tenbackward.main import main
from tenbackward.matching import MatchedPost
from tenbackward.publishing import PublishError
def _match(path: str) -> MatchedPost:
return MatchedPost(
path=path,
title=path,
date=date(2016, 8, 4),
url=f"https://blog.example.com/2016/08/04/{path.split('-', 3)[-1].removesuffix('.md')}/",
)
@pytest.fixture() @pytest.fixture()
@@ -38,7 +51,11 @@ def _seed_state(data_dir: Path, ids: list[str]) -> None:
def test_run_emits_one_info_summary_on_success(env_setup, data_dir, capture_logger, monkeypatch) -> None: def test_run_emits_one_info_summary_on_success(env_setup, data_dir, capture_logger, monkeypatch) -> None:
monkeypatch.setenv("DATA_DIR", str(data_dir)) monkeypatch.setenv("DATA_DIR", str(data_dir))
monkeypatch.setattr(main_module, "_iter_candidates", lambda config: ["new-1", "already-1"]) def _stub_publish(config, posts, **_kwargs):
return "stubbed"
monkeypatch.setattr(main_module, "_iter_candidates", lambda config: [_match("new-1"), _match("already-1")])
monkeypatch.setattr(main_module, "publish_mastodon", _stub_publish)
_seed_state(data_dir, ["already-1"]) _seed_state(data_dir, ["already-1"])
@@ -100,7 +117,12 @@ def test_run_distinguishes_posted_from_skipped_via_ids(env_setup, data_dir, capt
monkeypatch.setattr( monkeypatch.setattr(
main_module, main_module,
"_iter_candidates", "_iter_candidates",
lambda config: ["alpha", "beta", "gamma"], lambda config: [_match("alpha"), _match("beta"), _match("gamma")],
)
monkeypatch.setattr(
main_module,
"publish_mastodon",
lambda config, posts, **_kwargs: "stubbed",
) )
_seed_state(data_dir, ["beta"]) _seed_state(data_dir, ["beta"])
@@ -114,3 +136,28 @@ def test_run_distinguishes_posted_from_skipped_via_ids(env_setup, data_dir, capt
assert "beta" not in payload["posted_ids"] assert "beta" not in payload["posted_ids"]
assert payload["skipped"] == 1 assert payload["skipped"] == 1
assert payload["posted"] == 2 assert payload["posted"] == 2
def test_run_publish_failure_leaves_state_untouched(
env_setup, data_dir, capture_logger, monkeypatch
) -> None:
monkeypatch.setenv("DATA_DIR", str(data_dir))
def _boom(config, posts, **_kwargs):
raise PublishError("publish_failed: stub")
monkeypatch.setattr(main_module, "_iter_candidates", lambda config: [_match("alpha")])
monkeypatch.setattr(main_module, "publish_mastodon", _boom)
rc = main()
assert rc != 0
from tenbackward.state import load_posted
assert load_posted(data_dir) == []
error_lines = [
line
for line in _run_lines(capture_logger)
if line.get("level") == "ERROR"
]
assert any(line.get("event") == "pipeline_error" for line in error_lines)