fix(ci): run full pytest suite, repair 13 rotted tests

CI was running only test_plugin_contract.py and test_version_consistency.py
(2 of 84 test files), masking 13 rotted tests across 4 clusters. The suite is
fully offline-safe (1402 tests in ~7s without network), so the narrow scope
wasn't gating integration flakiness; it was just stale. validate.yml now runs
`uv run pytest` against the full suite.

Engine fix: store.findings_from_report is rerank-first. ranked_candidates is
the primary persistence path; hackernews/polymarket are unconditionally
supplemented from items_by_source because they rank poorly but matter for
watchlists. When ranked_candidates was empty (rerank failed or skipped),
reddit, x, and every other source were silently dropped. The supplement loop
now falls back to all sources only when ranked_candidates is empty; the normal
path is unchanged.

Test repairs:
- test_store.py (6) + test_watchlist_commands.py (2): cascade from the engine fix
- test_get_new_findings_filters_by_date (latent): local-time vs SQLite UTC
  flake — switched to datetime.now(timezone.utc)
- TestPollDeviceAuth (3): mock_time.time side_effect lists too short after
  impl added a last_reminder call — padded timeout test, pinned others to
  return_value=0 (loops terminate via urlopen, not the clock)
- test_bare_run_emits_web_promo: engine reads ~/.config/last30days/.env, so
  a contributor's saved EXA/PARALLEL key made grounding "available" and
  suppressed the web promo. Also missing X made the "x" promo preempt "web".
  Set LAST30DAYS_CONFIG_DIR="", subprocess cwd=tmpdir, XAI_API_KEY stub.
This commit is contained in:
Trevin Chow
2026-05-16 21:34:22 -07:00
parent 2e39ee8ce4
commit eb2d7a0f37
5 changed files with 57 additions and 31 deletions
+3 -3
View File
@@ -10,7 +10,7 @@ permissions:
contents: read
jobs:
plugin-contract:
tests:
runs-on: ubuntu-latest
steps:
- name: Checkout
@@ -22,5 +22,5 @@ jobs:
- name: Set up Python
run: uv python install 3.12
- name: Run plugin contract tests
run: uv run pytest tests/test_plugin_contract.py tests/test_version_consistency.py
- name: Run test suite
run: uv run pytest
+11 -8
View File
@@ -676,24 +676,28 @@ def findings_from_report(
Uses ranked candidates (post-rerank) when available for quality scores and explanations.
Supplements with raw items from items_by_source for HN/PM that didn't rank highly
but are valuable for watchlist persistence.
but are valuable for watchlist persistence. When ranked_candidates is empty
(degraded path — rerank failed or was skipped), falls back to supplementing
all sources from items_by_source so findings aren't silently dropped.
"""
findings = []
seen_urls = set()
# Phase 1: Process ranked candidates (high-quality data with explanations and corroboration)
for candidate in report.ranked_candidates:
finding = finding_from_candidate(candidate)
findings.append(finding)
findings.append(finding_from_candidate(candidate))
seen_urls.add(candidate.url)
# Phase 2: Add HN/PM items not already captured in ranked candidates
for source_name in ["hackernews", "polymarket"]:
supplement_sources = (
list(report.items_by_source)
if not report.ranked_candidates
else ["hackernews", "polymarket"]
)
for source_name in supplement_sources:
if source_name not in report.items_by_source:
continue
for item in report.items_by_source[source_name]:
if item.url in seen_urls:
continue # Already captured with rich data
continue
findings.append({
"source": source_name,
"source_url": item.url,
@@ -706,7 +710,6 @@ def findings_from_report(
})
seen_urls.add(item.url)
# Apply global limit after collecting all findings (fix: was per-source, now global)
return findings[:limit] if limit is not None else findings
+23 -4
View File
@@ -6,6 +6,7 @@ from __future__ import annotations
import os
import subprocess
import sys
import tempfile
import unittest
from pathlib import Path
@@ -27,13 +28,31 @@ class FooterNudgeSuppressionTests(unittest.TestCase):
"--emit=md",
*argv,
]
env = {**os.environ, "LAST30DAYS_SKIP_PREFLIGHT": "1"}
env = {
**os.environ,
"LAST30DAYS_SKIP_PREFLIGHT": "1",
# Skip ~/.config/last30days/.env so a contributor's saved
# BRAVE/EXA/SERPER/PARALLEL key doesn't make grounding "available"
# and suppress the promo we're checking for.
"LAST30DAYS_CONFIG_DIR": "",
# Pin X as available so _missing_sources_for_promo selects "web"
# (otherwise the "x" promo wins and the BRAVE_API_KEY string never
# appears).
"XAI_API_KEY": "test-stub",
}
# Strip any grounded-web keys the host might have so the promo path
# triggers deterministically in mock + no-backend.
# triggers deterministically in mock + no-backend. Also strip X cookie
# credentials so XAI_API_KEY is the unambiguous X backend.
for key in ("BRAVE_API_KEY", "EXA_API_KEY", "SERPER_API_KEY",
"PARALLEL_API_KEY", "OPENROUTER_API_KEY"):
"PARALLEL_API_KEY", "OPENROUTER_API_KEY",
"AUTH_TOKEN", "CT0", "LAST30DAYS_X_BACKEND"):
env.pop(key, None)
return subprocess.run(cmd, capture_output=True, text=True, env=env)
# Run from a tmpdir so _find_project_env() can't walk up into any
# .claude/last30days.env above the repo on the contributor's machine.
with tempfile.TemporaryDirectory() as tmp:
return subprocess.run(
cmd, capture_output=True, text=True, env=env, cwd=tmp,
)
def test_bare_run_emits_web_promo(self):
result = self._run(topic="OpenAI")
+8 -4
View File
@@ -200,8 +200,10 @@ class TestPollDeviceAuth:
@patch("lib.setup_wizard.urlopen")
def test_timeout_returns_none(self, mock_urlopen, mock_time):
"""Returns None when timeout is exceeded."""
# Simulate time passing beyond deadline
mock_time.time = MagicMock(side_effect=[0, 301])
# poll_device_auth calls time.time() for deadline init, last_reminder init,
# then once per while-loop iteration. Three values are enough for one check
# that exceeds the deadline.
mock_time.time = MagicMock(side_effect=[0, 0, 301])
mock_time.sleep = MagicMock()
result = setup_wizard.poll_device_auth("dc-123", interval=5, timeout=300)
@@ -211,7 +213,9 @@ class TestPollDeviceAuth:
@patch("lib.setup_wizard.urlopen")
def test_expired_token_returns_none(self, mock_urlopen, mock_time):
"""Returns None on expired_token error."""
mock_time.time = MagicMock(side_effect=[0, 0])
# Loop terminates via urlopen response, not the clock — pin time to 0
# so the deadline check stays a non-event regardless of call count.
mock_time.time = MagicMock(return_value=0)
mock_time.sleep = MagicMock()
expired_resp = MagicMock()
@@ -230,7 +234,7 @@ class TestPollDeviceAuth:
"""HTTP 400 during polling continues (authorization pending)."""
from urllib.error import HTTPError
mock_time.time = MagicMock(side_effect=[0, 0, 0])
mock_time.time = MagicMock(return_value=0)
mock_time.sleep = MagicMock()
success_resp = MagicMock()
+5 -5
View File
@@ -3,7 +3,7 @@
import json
import sqlite3
import tempfile
from datetime import datetime, timedelta
from datetime import datetime, timedelta, timezone
from pathlib import Path
import pytest
@@ -503,14 +503,14 @@ def test_get_new_findings_filters_by_date(temp_db, sample_report):
findings = store.findings_from_report(sample_report)
store.store_findings(run_id, topic["id"], findings)
# Get findings since tomorrow (should be empty)
tomorrow = (datetime.now() + timedelta(days=1)).strftime("%Y-%m-%d")
# Use UTC because store writes first_seen via SQLite's datetime('now') (UTC).
# Local-time math here would flake near midnight UTC.
tomorrow = (datetime.now(timezone.utc) + timedelta(days=1)).strftime("%Y-%m-%d")
new_findings = store.get_new_findings(topic["id"], since=tomorrow)
assert len(new_findings) == 0
# Get findings since yesterday (should have all)
yesterday = (datetime.now() - timedelta(days=1)).strftime("%Y-%m-%d")
yesterday = (datetime.now(timezone.utc) - timedelta(days=1)).strftime("%Y-%m-%d")
new_findings = store.get_new_findings(topic["id"], since=yesterday)
assert len(new_findings) == 4