From 9b9061666f8b0c5978a2e619437866e8474a7918 Mon Sep 17 00:00:00 2001 From: Pierre-Henry Soria Date: Sat, 19 Sep 2026 04:14:23 +1000 Subject: [PATCH 1/5] fix: keep recent articles and drop cross-feed duplicates Two defects in the RSS collection path: The per-feed entry slice ran before the date filter, so a feed that is not sorted newest-first, or that pins an old post at the top, could drop every recent article. Filter on the cutoff first, then limit. The same link often arrives through several feeds (aggregators such as Hacker News republish posts that are already collected directly), and nothing rejected the repeat, so it became a separate page in the journal. Deduplicate on a normalised URL, ignoring case, a trailing slash and the tracking parameters that differ between feeds. Also remove the global cap, which multiplied the per-feed limit by the number of feeds and therefore could never bind. --- src/rss_aggregator.py | 66 +++++++++++++++++++++++++++++++++---------- 1 file changed, 51 insertions(+), 15 deletions(-) diff --git a/src/rss_aggregator.py b/src/rss_aggregator.py index 1b03cce..b8789b5 100644 --- a/src/rss_aggregator.py +++ b/src/rss_aggregator.py @@ -11,6 +11,7 @@ import time import re import html +from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit @dataclass class Article: @@ -50,9 +51,19 @@ def collect_articles(self) -> List[Article]: # Trier par date de publication (plus récent en premier) all_articles.sort(key=lambda x: x.published, reverse=True) - # Limiter le nombre total d'articles - max_total = self.config.max_articles_per_feed * len(self.config.rss_feeds) - return all_articles[:max_total] + # Écarter les doublons: un même lien revient souvent via plusieurs flux + # (agrégateurs). Le tri ci-dessus garde la version la plus récente. + unique_articles = [] + seen_urls = set() + for article in all_articles: + key = self.normalise_url(article.url) + if key in seen_urls: + print(f"↔️ Doublon ignoré: {article.title}") + continue + seen_urls.add(key) + unique_articles.append(article) + + return unique_articles def process_feed(self, feed_url: str) -> List[Article]: """Traiter un seul flux RSS""" @@ -80,19 +91,16 @@ def process_feed(self, feed_url: str) -> List[Article]: # Date limite (articles récents uniquement) cutoff_date = datetime.now(timezone.utc) - timedelta(days=self.config.days_lookback) - for entry in feed.entries[:self.config.max_articles_per_feed]: + # Filtrer sur la date avant de limiter: certains flux ne sont pas triés + # par date et épinglent d'anciennes entrées en tête. + recent_entries = [] + for entry in feed.entries: + published = self.parse_published(entry) + if published >= cutoff_date: + recent_entries.append((entry, published)) + + for entry, published in recent_entries[:self.config.max_articles_per_feed]: try: - # Parser la date de publication - published = datetime.now(timezone.utc) - if hasattr(entry, 'published_parsed') and entry.published_parsed: - published = datetime(*entry.published_parsed[:6], tzinfo=timezone.utc) - elif hasattr(entry, 'updated_parsed') and entry.updated_parsed: - published = datetime(*entry.updated_parsed[:6], tzinfo=timezone.utc) - - # Ignorer les articles trop anciens - if published < cutoff_date: - continue - # Obtenir le contenu content = self.extract_content(entry) @@ -116,6 +124,34 @@ def process_feed(self, feed_url: str) -> List[Article]: return articles + def normalise_url(self, url: str) -> str: + """Normaliser une URL pour comparer deux liens équivalents""" + if not url: + return "" + + cleaned = url.strip() + + # Ignorer la casse du schéma et du domaine, ainsi qu'un slash final + parts = urlsplit(cleaned) + path = parts.path.rstrip('/') or '/' + + # Retirer les paramètres de suivi, qui diffèrent d'un flux à l'autre + query = urlencode([(name, value) for name, value in parse_qsl(parts.query) + if not name.lower().startswith('utm_') + and name.lower() not in {'fbclid', 'gclid', 'ref', 'source'}]) + + return urlunsplit((parts.scheme.lower(), parts.netloc.lower(), path, query, '')) + + def parse_published(self, entry) -> datetime: + """Lire la date de publication d'une entrée, en UTC""" + for attribute in ('published_parsed', 'updated_parsed'): + value = getattr(entry, attribute, None) + if value: + return datetime(*value[:6], tzinfo=timezone.utc) + + # Sans date exploitable, considérer l'entrée comme actuelle + return datetime.now(timezone.utc) + def extract_content(self, entry) -> str: """Extraire le contenu de l'entrée RSS""" content = "" From aff5f7da9c7c3f0997131fee2fd968c0200a1476 Mon Sep 17 00:00:00 2001 From: Pierre-Henry Soria Date: Sat, 19 Sep 2026 04:14:28 +1000 Subject: [PATCH 2/5] fix: print the collected article text, not its 200-char preview An article carried both the cleaned content (up to 1000 characters) and a summary that was only its first 200 characters. The PDF preferred the summary, so four fifths of the text gathered from every feed was discarded before reaching the Kindle. Render whichever field holds more text. A video summary is written by the model and has no content field to compare against, so it keeps being used as before. --- src/pdf_generator.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/src/pdf_generator.py b/src/pdf_generator.py index 66f2ef2..155628b 100644 --- a/src/pdf_generator.py +++ b/src/pdf_generator.py @@ -183,12 +183,12 @@ def add_content_item(self, story: List, item, index: int, styles): story.append(Paragraph(source_info, styles['Metadata'])) - # Contenu principal - content_text = "" - if hasattr(item, 'summary') and item.summary: - content_text = item.summary - elif hasattr(item, 'content') and item.content: - content_text = item.content + # Contenu principal: préférer le texte le plus complet. Le résumé d'un + # article n'est qu'une troncature de son contenu, alors que celui d'une + # vidéo est rédigé par l'IA et constitue la seule source disponible. + summary_text = getattr(item, 'summary', '') or '' + full_text = getattr(item, 'content', '') or '' + content_text = full_text if len(full_text) > len(summary_text) else summary_text if content_text: # Diviser le contenu en paragraphes si nécessaire From 9e40a0a8708e110258d3fc8967fa898f693cf658 Mon Sep 17 00:00:00 2001 From: Pierre-Henry Soria Date: Sat, 19 Sep 2026 04:14:34 +1000 Subject: [PATCH 3/5] test: cover the article selection and rendering fixes Four regressions, three of which fail against the previous code: a pinned old entry no longer hides the recent ones, a link shared by two feeds is collected once, and the PDF shows the full article text. The fourth pins the behaviour the rendering change had to preserve, that a video still shows its model-written summary. Update the counts quoted in the README to match. --- README.md | 4 ++-- test_system.py | 52 ++++++++++++++++++++++++++++++++++++++++++++++++-- 2 files changed, 52 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 3ed25d6..92c9b3c 100644 --- a/README.md +++ b/README.md @@ -12,7 +12,7 @@ restent dans l'environnement; les valeurs par défaut sauvegardées ne les copie pas dans `config.yaml`. ```bash -venv/bin/python test_system.py # 12 régressions hors ligne +venv/bin/python test_system.py # 16 régressions hors ligne venv/bin/python verify.py # présence des fichiers et imports uniquement venv/bin/python demo.py # PDF fictif séparé du vrai journal quotidien ``` @@ -26,7 +26,7 @@ Les notes YouTube utilisent seulement le **titre et la description RSS**, pas la transcription ou le contenu vidéo. Vérifiez les faits avant utilisation. Les articles RSS sont des extraits, pas une analyse complète des sources. -Validation locale : 12 tests, contrôle des dépendances, scripts shell et démo +Validation locale : 16 tests, contrôle des dépendances, scripts shell et démo PDF de six pages inspectée. Aucun appel réel OpenAI/SMTP ni livraison Kindle vérifié. La mise en page reste A4 avec une page par élément; elle n'est pas validée sur Kindle. Helvetica couvre le texte français de la démo, mais pas tous les diff --git a/test_system.py b/test_system.py index a084fee..ca79e2b 100644 --- a/test_system.py +++ b/test_system.py @@ -6,15 +6,17 @@ import tempfile import unittest from contextlib import redirect_stdout -from datetime import datetime, timezone +from datetime import datetime, timedelta, timezone from pathlib import Path from types import SimpleNamespace from unittest.mock import MagicMock, patch +from reportlab.platypus import Paragraph + from src.config import Config from src.pdf_generator import PDFGenerator from src.rss_aggregator import Article, RSSAggregator -from src.youtube_summarizer import YouTubeSummarizer +from src.youtube_summarizer import VideoSummary, YouTubeSummarizer from src.kindle_sender import KindleSender import main as application @@ -100,6 +102,52 @@ def test_rss_dates_are_utc(self): self.assertEqual(len(articles), 1) self.assertEqual(articles[0].published.tzinfo, timezone.utc) + def test_duplicate_links_across_feeds_appear_once(self): + self.config_path.write_text( + 'rss_feeds:\n - https://example.invalid/a\n - https://example.invalid/b\n') + date = datetime.now(timezone.utc).strftime('%a, %d %b %Y %H:%M:%S GMT') + feed = ('TestShared' + 'https://example.invalid/post?utm_source=feed' + f'{date}Text' + '').encode() + with patch('src.rss_aggregator.requests.get', return_value=MagicMock(content=feed)), \ + patch('src.rss_aggregator.time.sleep'), redirect_stdout(io.StringIO()): + articles = RSSAggregator(self.config()).collect_articles() + self.assertEqual(len(articles), 1) + + def test_recent_entries_survive_a_pinned_older_entry(self): + old = (datetime.now(timezone.utc) - timedelta(days=400)).strftime('%a, %d %b %Y %H:%M:%S GMT') + new = datetime.now(timezone.utc).strftime('%a, %d %b %Y %H:%M:%S GMT') + items = ''.join( + f'Pinned {index}https://example.invalid/old{index}' + f'{old}Old' for index in range(3)) + feed = ('Test' + items + + 'Recenthttps://example.invalid/new' + f'{new}Fresh' + '').encode() + with patch('src.rss_aggregator.requests.get', return_value=MagicMock(content=feed)): + articles = RSSAggregator(self.config()).process_feed('https://example.invalid/rss') + self.assertEqual([article.title for article in articles], ['Recent']) + + def test_pdf_renders_article_content_rather_than_its_truncation(self): + body = 'Phrase unique. ' + 'corps ' * 200 + article = Article('Titre', body, 'https://example.invalid/a', + datetime.now(), 'Source', summary=body[:200] + '...') + generator = PDFGenerator(self.config()) + story, styles = [], generator.create_custom_styles() + generator.add_content_item(story, article, 1, styles) + rendered = ' '.join(item.text for item in story if isinstance(item, Paragraph)) + self.assertIn(body.strip(), rendered) + + def test_pdf_still_uses_the_ai_summary_for_videos(self): + video = VideoSummary('Titre', 'Résumé rédigé par le modèle.', 'https://example.invalid/v', + datetime.now(), 'YouTube', 'Chaîne') + generator = PDFGenerator(self.config()) + story, styles = [], generator.create_custom_styles() + generator.add_content_item(story, video, 1, styles) + rendered = ' '.join(item.text for item in story if isinstance(item, Paragraph)) + self.assertIn('Résumé rédigé par le modèle.', rendered) + def test_smtp_uses_verified_tls_with_a_timeout(self): pdf = self.root / 'test.pdf'; pdf.write_bytes(b'%PDF-test') config = SimpleNamespace(kindle_email='reader@example.invalid', sender_email='sender@example.invalid', From b3d1d19aa7ef7578954eb05c0dc9291016b73940 Mon Sep 17 00:00:00 2001 From: Pierre-Henry Soria Date: Sat, 19 Sep 2026 06:15:56 +1000 Subject: [PATCH 4/5] ci: run the offline regressions on every push and pull request The README noted that no CI was configured. The suite needs no credentials and makes no network call, so it runs as-is on a hosted runner: dependencies from the lockfile on Python 3.12, then verify.py, the 16 regressions, and demo.py. The demo PDF is uploaded as an artifact so a rendering change can be inspected from the run. Read-only permissions, and no secrets are referenced; the repository is public and nothing here should reach a real feed, OpenAI or SMTP. --- .github/workflows/tests.yml | 43 +++++++++++++++++++++++++++++++++++++ 1 file changed, 43 insertions(+) create mode 100644 .github/workflows/tests.yml diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml new file mode 100644 index 0000000..e013845 --- /dev/null +++ b/.github/workflows/tests.yml @@ -0,0 +1,43 @@ +name: tests + +# Les régressions sont hors ligne: aucun secret, aucun appel RSS, OpenAI ou SMTP. +on: + push: + branches: [main] + pull_request: + workflow_dispatch: + +permissions: + contents: read + +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + cache: pip + cache-dependency-path: requirements.lock + + - name: Installer les dépendances verrouillées + run: python -m pip install -r requirements.lock + + - name: Vérifier la structure et les imports + run: python verify.py + + - name: Régressions hors ligne + run: python test_system.py + + - name: Générer le PDF de démonstration + run: python demo.py + + - name: Conserver le PDF de démonstration + uses: actions/upload-artifact@v4 + with: + name: demo-journal + path: output/demo_journal_*.pdf + if-no-files-found: error + retention-days: 7 From 4d5bb57c17f6692ae48b61ce6badee85e1ed691b Mon Sep 17 00:00:00 2001 From: Pierre-Henry Soria Date: Sat, 19 Sep 2026 06:16:07 +1000 Subject: [PATCH 5/5] docs: note that CI now replays the offline checks --- README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 92c9b3c..c762bae 100644 --- a/README.md +++ b/README.md @@ -30,8 +30,9 @@ Validation locale : 16 tests, contrôle des dépendances, scripts shell et démo PDF de six pages inspectée. Aucun appel réel OpenAI/SMTP ni livraison Kindle vérifié. La mise en page reste A4 avec une page par élément; elle n'est pas validée sur Kindle. Helvetica couvre le texte français de la démo, mais pas tous les -caractères Unicode possibles dans des flux tiers. Aucun workflow CI n'est encore -configuré. +caractères Unicode possibles dans des flux tiers. GitHub Actions rejoue ces +vérifications hors ligne sur chaque push et pull request; aucun envoi réel n'y +est testé. [Capture historique de la démo (juin 2025)](demo/screenshots/daily-learning-plan-script-generation.png)