Compare commits
89 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| f3df47c381 | |||
| aba6172032 | |||
| d07e4698e3 | |||
| 4c0282dd55 | |||
| 99909fca67 | |||
| 36c43d50b7 | |||
| ecf68347db | |||
| e2d9d705f6 | |||
| 9c09a67ac2 | |||
| 32da0bd6cb | |||
| 863c3bc145 | |||
| 9f39d10bc5 | |||
| 03043da407 | |||
| 8bab997854 | |||
| 375fd0bcc0 | |||
| 92d65723e4 | |||
| f794f82af5 | |||
| 791c0a57a0 | |||
| 211df0deaa | |||
| d7b3995da1 | |||
| e217db77cc | |||
| 8ea207b348 | |||
| b04212680d | |||
| 2c2cfb9e7e | |||
| 3276496f49 | |||
| 68ae74ff4f | |||
| 27c90504c0 | |||
| f4eb0af104 | |||
| 79b5d049ce | |||
| 0e2059661a | |||
| b1c5f8db82 | |||
| 89c5cb9d5d | |||
| 5d4f9ef2c5 | |||
| 01262f78c6 | |||
| 96a4a78faa | |||
| eb2d7a0f37 | |||
| 719cdef2fb | |||
| ac04b56acc | |||
| 42bfc6c76c | |||
| 5a2fe5279b | |||
| a717dd2b2c | |||
| 9f08bb68b5 | |||
| 602de1ebda | |||
| bf3a82a87e | |||
| c010feb8f8 | |||
| edea402b7c | |||
| 4d4ac97ffb | |||
| cd34966b4f | |||
| 85255be350 | |||
| 1aa120a420 | |||
| 306d8c2d73 | |||
| d0dcf751f1 | |||
| afd4b04d6d | |||
| 14d8f62e02 | |||
| 0fd532d249 | |||
| 8867a007ea | |||
| 8af8f06b06 | |||
| 01b5f3dc1e | |||
| 2e39ee8ce4 | |||
| 37033164da | |||
| 9fe4b8f130 | |||
| 73dc6b9996 | |||
| 1fd763e09f | |||
| e0f6ef845a | |||
| c918e18465 | |||
| 6fe0aca7ee | |||
| d9f606ff75 | |||
| 74a387b093 | |||
| 4a30923892 | |||
| 9fb19eae63 | |||
| d1cc29d338 | |||
| 095bcae915 | |||
| ded52062e6 | |||
| 164d7ae6ed | |||
| f1ce7533e6 | |||
| 0b939bf703 | |||
| 9f95efb215 | |||
| ff54c07a3b | |||
| e276c30477 | |||
| 2f277dfc66 | |||
| 6c2c55733c | |||
| 997708ad48 | |||
| c913e1cf89 | |||
| 54db014c7c | |||
| 4f6b86c456 | |||
| 80a1a47eef | |||
| c845f483d6 | |||
| dc934ddb6a | |||
| ed455ca036 |
@@ -11,7 +11,7 @@
|
||||
{
|
||||
"name": "last30days",
|
||||
"description": "Research any topic across Reddit, X, YouTube, TikTok, Instagram, Hacker News, Polymarket, GitHub, and 5+ more sources. AI agent scores by upvotes, likes, and real money - not editors.",
|
||||
"version": "3.2.0",
|
||||
"version": "3.2.4",
|
||||
"author": {
|
||||
"name": "Matt Van Horn",
|
||||
"url": "https://github.com/mvanhorn"
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "last30days",
|
||||
"version": "3.2.0",
|
||||
"version": "3.2.4",
|
||||
"description": "Research any topic across Reddit, X, YouTube, TikTok, Instagram, Hacker News, Polymarket, GitHub, and 5+ more sources. AI agent scores by upvotes, likes, and real money - not editors.",
|
||||
"author": {
|
||||
"name": "Matt Van Horn",
|
||||
|
||||
@@ -1,43 +0,0 @@
|
||||
{
|
||||
"name": "last30days",
|
||||
"version": "3.2.0",
|
||||
"description": "Research any topic across Reddit, X, YouTube, TikTok, Instagram, Hacker News, Polymarket, GitHub, and 5+ more sources. AI agent scores by upvotes, likes, and real money - not editors.",
|
||||
"author": {
|
||||
"name": "Matt Van Horn",
|
||||
"email": "mvanhorn@gmail.com",
|
||||
"url": "https://github.com/mvanhorn"
|
||||
},
|
||||
"homepage": "https://github.com/mvanhorn/last30days-skill",
|
||||
"repository": "https://github.com/mvanhorn/last30days-skill",
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"research",
|
||||
"reddit",
|
||||
"twitter",
|
||||
"youtube",
|
||||
"tiktok",
|
||||
"instagram",
|
||||
"trends",
|
||||
"polymarket",
|
||||
"github",
|
||||
"hacker-news"
|
||||
],
|
||||
"skills": "./skills/",
|
||||
"interface": {
|
||||
"displayName": "Last 30 Days",
|
||||
"shortDescription": "Research recent discussion across social and web sources",
|
||||
"longDescription": "Research any topic across Reddit, X, YouTube, TikTok, Instagram, Hacker News, Polymarket, GitHub, and 5+ more sources. AI agent scores by upvotes, likes, and real money - not editors.",
|
||||
"developerName": "Matt Van Horn",
|
||||
"category": "Research",
|
||||
"capabilities": [
|
||||
"Interactive",
|
||||
"Read",
|
||||
"Write"
|
||||
],
|
||||
"websiteURL": "https://github.com/mvanhorn/last30days-skill",
|
||||
"privacyPolicyURL": "https://docs.github.com/en/site-policy/privacy-policies/github-general-privacy-statement",
|
||||
"termsOfServiceURL": "https://docs.github.com/en/site-policy/github-terms/github-terms-of-service",
|
||||
"defaultPrompt": "Use Last 30 Days to research this topic from the last 30 days across Reddit, X, YouTube, and web.",
|
||||
"brandColor": "#FF6B35"
|
||||
}
|
||||
}
|
||||
@@ -13,7 +13,6 @@
|
||||
<!-- How did you verify this works? -->
|
||||
|
||||
- [ ] Ran `uv run python -m pytest -q --tb=short`
|
||||
- [ ] Ran `bash scripts/sync.sh` (if scripts/ changed)
|
||||
|
||||
## Related Issues
|
||||
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
name: Security
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
dependency-audit:
|
||||
name: Dependency audit
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v5
|
||||
|
||||
- name: Set up Python
|
||||
run: uv python install 3.12
|
||||
|
||||
- name: Export locked dependency set
|
||||
run: |
|
||||
uv export \
|
||||
--locked \
|
||||
--all-groups \
|
||||
--no-hashes \
|
||||
--format requirements.txt \
|
||||
--output-file /tmp/last30days-requirements.txt
|
||||
|
||||
# Advisory-first: visibility before enforcement. This repo handles API keys,
|
||||
# cookies, browser tokens, and local env files, so dependency CVEs should be
|
||||
# visible in CI logs even before the project has a clean blocking baseline.
|
||||
# Set continue-on-error: false once a clean baseline run is confirmed.
|
||||
- name: Run pip-audit against locked dependencies
|
||||
continue-on-error: true
|
||||
run: uvx --python 3.12 pip-audit -r /tmp/last30days-requirements.txt --progress-spinner=off
|
||||
|
||||
secret-scan:
|
||||
name: Secret scan
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout full history for diff-aware scanning
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# Advisory-first: this reports verified secrets in pull requests and pushes to
|
||||
# main, but does not block merges until maintainers confirm a clean baseline.
|
||||
# The TruffleHog action automatically scans the PR range for pull_request
|
||||
# events and the pushed commit range for push events.
|
||||
# Set continue-on-error: false once a clean baseline run is confirmed.
|
||||
# Contributor policy: never commit real secrets in fixtures, tests, docs, or
|
||||
# examples; use obvious dummy values and env-based auth patterns instead.
|
||||
- name: Run TruffleHog OSS secret scan
|
||||
if: github.event_name == 'pull_request' || github.event_name == 'push' || github.event_name == 'workflow_dispatch'
|
||||
uses: trufflesecurity/trufflehog@v3.95.2
|
||||
continue-on-error: true
|
||||
with:
|
||||
path: ./
|
||||
version: v3.95.2
|
||||
extra_args: --only-verified
|
||||
@@ -10,7 +10,7 @@ permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
plugin-contract:
|
||||
tests:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -22,5 +22,5 @@ jobs:
|
||||
- name: Set up Python
|
||||
run: uv python install 3.12
|
||||
|
||||
- name: Run plugin contract tests
|
||||
run: uv run pytest tests/test_plugin_contract.py tests/test_version_consistency.py
|
||||
- name: Run test suite
|
||||
run: uv run pytest
|
||||
|
||||
@@ -1 +1,38 @@
|
||||
@CLAUDE.md
|
||||
# last30days Skill
|
||||
|
||||
Agent Skills package for researching any topic across Reddit, X, YouTube, and web. Installable across Claude Code (most common host), Codex, Cursor, GitHub Copilot, Gemini CLI, and 50+ other [Agent Skills](https://agentskills.io) hosts. Python scripts with multi-source search aggregation.
|
||||
|
||||
## Structure
|
||||
- `skills/last30days/SKILL.md` — canonical skill definition
|
||||
- `skills/last30days/scripts/last30days.py` — main research engine
|
||||
- `skills/last30days/scripts/lib/` — search, enrichment, rendering modules
|
||||
- `skills/last30days/scripts/lib/vendor/bird-search/` — vendored X search client
|
||||
- `docs/solutions/` — documented solutions to past problems (bugs, best practices, workflow patterns), organized by category with YAML frontmatter (`module`, `tags`, `problem_type`)
|
||||
- `CONCEPTS.md` — shared domain vocabulary (Skill, Engine, Harness, Beta channel) — relevant when orienting to the codebase or discussing project terminology
|
||||
|
||||
## Orientation
|
||||
- This is an Agent Skills package, not a CLI tool. The product is the slash-command-invoked skill (`/last30days <topic>` in most harnesses); `scripts/last30days.py` is implementation. Claude Code is the most common host but not the only one — features must work across every harness the skill installs into.
|
||||
- Feature design starts from the slash-command UX. A new engine flag with no SKILL.md integration is incomplete — the model invoking the skill won't know the flag exists.
|
||||
- README and PR examples show `/last30days <topic>` first. Direct CLI invocation (`python3 scripts/last30days.py ...`) is a fallback for scripting, cron, and dev-time engine testing; label it as such, never as the primary path.
|
||||
- Slash commands don't pass shell mechanics through. `/last30days OpenClaw --emit=html | pbcopy` is invalid in any harness — either use the slash form (no flags or pipes; let the model translate user intent into engine flags) or use the direct CLI form (full `python3 ...` with explicit flags and a real shell).
|
||||
|
||||
## Commands
|
||||
```bash
|
||||
# Dev/fallback: direct engine invocation (scripting, cron, or engine testing only)
|
||||
python3 skills/last30days/scripts/last30days.py "test query" --emit=compact
|
||||
npx skills add . -g -y # one-time: symlink this repo into every detected harness's skill dir
|
||||
|
||||
## Rules
|
||||
- `lib/__init__.py` must be bare package marker (comment only, NO eager imports)
|
||||
- One-time setup: `npx skills add . -g -y` creates symlinks from each detected harness's skill dir to this repo. Edits in the working tree propagate live to every harness — no re-deploy step needed.
|
||||
- Git remote: origin = public (`mvanhorn/last30days-skill`)
|
||||
|
||||
## Security hygiene
|
||||
- Never commit real API keys, browser cookies, auth tokens, app passwords, access tokens, or `.env` contents.
|
||||
- Use the env-based auth patterns in `skills/last30days/scripts/lib/env.py`; tests and fixtures must use obvious dummy values only.
|
||||
- Keep examples safe by redacting secrets and avoiding copy/pasteable live credentials in docs, fixtures, and test data.
|
||||
- Do not weaken or disable the advisory security workflow (`.github/workflows/security.yml`) without explaining why in the PR description or review thread.
|
||||
|
||||
## Beta channel
|
||||
|
||||
Experimental changes get tested on `mvanhorn/last30days-skill-private`, which installs as a parallel `/last30days-beta` slash command. Beta-only changes never ship to public without a review PR here. Workflow guide lives at `BETA.md` in the private repo. Plan that established this setup: `docs/plans/2026-04-17-005-feat-beta-skill-from-private-repo-plan.md`.
|
||||
|
||||
@@ -7,6 +7,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- `LAST30DAYS_YOUTUBE_SSH_HOST` env var: when set, yt-dlp YouTube search invocations are routed through `ssh <host>` for residential-IP egress. Bypasses YouTube's bot-wall on datacenter IPs (Hetzner/DigitalOcean/AWS) where `ytsearch:` returns 0 results regardless of cookies (the IP fingerprint is checked first). The named host must be configured in `~/.ssh/config` and have yt-dlp installed. Host value is validated against `^[a-zA-Z0-9._-]+$` to reject SSH option-injection (e.g. a leading `-` masquerading as a flag). The transcript path is unchanged (uses the existing HTTP fallback when SSH-routing is on, since the timedtext API isn't bot-walled).
|
||||
|
||||
### Changed
|
||||
|
||||
- Replace the SKILL_ROOT resolver loops in Step 1 and comparison-mode with a single `SKILL_DIR` substitution pattern. The model templates the absolute path of the SKILL.md's own directory (which it always knows from the Read tool result); the bash block just validates that `scripts/last30days.py` lives there. Removes ~80 lines of bash across the two locations. Fixes a real bug: the previous resolver could pick a different install than the SKILL.md the model loaded from (spec-vs-engine divergence) and didn't enumerate harnesses like Hermes at all. The simplification works for any harness without enumeration because it just uses wherever SKILL.md was loaded from. STEP 0's marketplaces-stale-clone hop is unchanged.
|
||||
- Rename "Digg AI 1000" to just "Digg" in user-facing output (footer line, source label, inline-quote suffix, why_relevant, container attribution). Internal references to the upstream Digg AI 1000 product remain in code comments and docstrings.
|
||||
- Bump `POSTS_PER_CLUSTER` from 3 to 5 and the render-side display limit from 2 to 3 to match the per-source enrichment caps used by Reddit, HN, YouTube, TikTok, and GitHub. The previous 3/2 caps routinely truncated cluster context (e.g. dropped a Jason Calacanis quote tweet on a `cli-printing-press` run).
|
||||
- Rewrite SKILL.md path resolution. STEP 0 narrows from a global canonical-path enforcement to a Claude-Code-marketplaces-only stale-clone guard. Step 1 SKILL_ROOT resolver walks a single precedence list (Claude plugin cache, then `~/.codex/skills/`, `~/.agents/skills/`, repo checkout, `./.skills/last30days` for `npx skills add`, CWD, Gemini). Adds SKILL.md frontmatter fallback to `render.py::_skill_version` so the badge no longer prints `v?` on installs that don't include `.claude-plugin/plugin.json`.
|
||||
|
||||
- Switch SKILL.md's `--plan` and `--competitors-plan` invocation templates from inline single-quoted JSON to heredoc-written tmpfiles. Apostrophes in resolved context strings ("McDonald's", "people's choice", "developer's") previously closed the outer single-quote and broke shell parsing before the engine started — observed in a Codex run during PR #400 testing. The engine's `parse_plan()` / `parse_competitors_plan()` already supported file paths (via `os.path.isfile()` probe); only the template prose changed. Fixes [#403](https://github.com/mvanhorn/last30days-skill/issues/403).
|
||||
|
||||
### Removed
|
||||
|
||||
- **BREAKING for Codex native-plugin users:** `.codex-plugin/plugin.json` and the matching SKILL_ROOT resolver branch in SKILL.md Step 1. Codex users should install via `npx skills add mvanhorn/last30days-skill` or copy the skill to `~/.codex/skills/last30days/`.
|
||||
- **`skills/last30days/scripts/sync.sh`.** The maintainer dev-deploy script is gone. Every job it did has a better replacement: `npx skills add . -g -y` symlinks the working tree into every detected harness's skill dir (better than sync.sh's copy model — edits propagate live), `hermes skills install mvanhorn/last30days-skill --force` handles Hermes, `clawhub install last30days-official` handles OpenClaw, and the Claude marketplace cache target was a "test against the official install path" hack we shouldn't have been recommending in the first place. The `test_sync_cache_path_uses_skill_version` test was dropped along with it. CLAUDE.md, HERMES_SETUP.md, the PR template, and a render.py docstring were updated to drop references; CHANGELOG and historical docs (release notes, plan files) keep their existing mentions as accurate history.
|
||||
|
||||
## [3.2.0] - 2026-05-09
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,25 +1 @@
|
||||
# last30days Skill
|
||||
|
||||
Claude Code skill for researching any topic across Reddit, X, YouTube, and web.
|
||||
Python scripts with multi-source search aggregation.
|
||||
|
||||
## Structure
|
||||
- `skills/last30days/SKILL.md` — canonical skill definition
|
||||
- `skills/last30days/scripts/last30days.py` — main research engine
|
||||
- `skills/last30days/scripts/lib/` — search, enrichment, rendering modules
|
||||
- `skills/last30days/scripts/lib/vendor/bird-search/` — vendored X search client
|
||||
|
||||
## Commands
|
||||
```bash
|
||||
python3 skills/last30days/scripts/last30days.py "test query" --emit=compact
|
||||
bash skills/last30days/scripts/sync.sh
|
||||
```
|
||||
|
||||
## Rules
|
||||
- `lib/__init__.py` must be bare package marker (comment only, NO eager imports)
|
||||
- After edits: run `bash skills/last30days/scripts/sync.sh` to deploy
|
||||
- Git remote: origin = public (`mvanhorn/last30days-skill`)
|
||||
|
||||
## Beta channel
|
||||
|
||||
Experimental changes get tested on `mvanhorn/last30days-skill-private`, which installs as a parallel `/last30days-beta` slash command. Beta-only changes never ship to public without a review PR here. Workflow guide lives at `BETA.md` in the private repo. Plan that established this setup: `docs/plans/2026-04-17-005-feat-beta-skill-from-private-repo-plan.md`.
|
||||
@AGENTS.md
|
||||
|
||||
+23
@@ -0,0 +1,23 @@
|
||||
# Concepts
|
||||
|
||||
Shared vocabulary for `last30days-skill`. Terms here have a precise project-specific meaning — distinct enough from their general technical sense that a new contributor would need them defined to follow conversations, PR descriptions, or the SKILL.md contract.
|
||||
|
||||
## The package
|
||||
|
||||
### Skill
|
||||
|
||||
A self-contained agent-instructions package consisting of a `SKILL.md` prose contract plus a sibling `scripts/` directory containing the executable code the SKILL.md invokes. The package conforms to the [Agent Skills](https://agentskills.io) open format and installs across every major harness (Claude Code, Codex, Cursor, GitHub Copilot, Gemini CLI, and 50+ others) via `npx skills add`, harness-native plugin installers, or per-harness skill directories. A Skill is the unit of distribution; the Skill is the product.
|
||||
|
||||
### Engine
|
||||
|
||||
The Python script (`scripts/last30days.py`) the Skill's SKILL.md invokes to do the actual research work. The Engine and SKILL.md have a contract: SKILL.md tells the model which flags to pass (`--plan`, `--competitors-plan`, `--x-handle`, `--subreddits`, `--emit=compact`, etc.), and the Engine produces a specific output shape (badge line, ranked evidence clusters, emoji-tree footer) that the model is contractually required to pass through. The Engine is implementation; the SKILL.md prose is the agent-facing surface.
|
||||
|
||||
### Harness
|
||||
|
||||
The agent runtime that loads Skills and invokes them on the user's behalf. Claude Code is the most common Harness for this Skill but not the only one — Codex, Cursor, GitHub Copilot, Gemini CLI, and the rest of the Agent Skills ecosystem also count. "Multi-harness" describes a Skill that works correctly across every Harness it installs into; features written without multi-harness awareness (e.g., engine flags with no SKILL.md integration, or paths hardcoded to one Harness's install layout) regress on Harnesses other than the one they were tested against.
|
||||
|
||||
## Distribution
|
||||
|
||||
### Beta channel
|
||||
|
||||
A parallel install of the Skill, sourced from the private `mvanhorn/last30days-skill-private` repo and installed as `/last30days-beta` rather than `/last30days`. The Beta channel exists so experimental changes can be tested by real users before they ship to the public `/last30days`. Promotion from Beta to public happens via a review PR against this (public) repo — Beta-only changes never ship to public without that PR. The Beta channel workflow guide lives in `BETA.md` in the private repo.
|
||||
+1
-1
@@ -23,7 +23,7 @@ v3 has full GitHub search: issues, PRs, person-mode profiles, project-mode repos
|
||||
### @thinkun
|
||||
[PR #116](https://github.com/mvanhorn/last30days-skill/pull/116) - Resilient Reddit, prevent enrichment timeout from discarding results
|
||||
v3 has parallel enrichment with per-item timeouts. No results are ever dropped.
|
||||
> _Add your bio, website, or anything you'd like here._
|
||||
> Thinker, technologist, AI expert, music-tinkerer. Founder of [Thinkun](https://thinkun.com). [@thinkun on GitHub](https://github.com/thinkun) · [@unthink on X](https://x.com/unthink)
|
||||
|
||||
### @thomasmktong
|
||||
[PR #124](https://github.com/mvanhorn/last30days-skill/pull/124) - Pure Python Reddit fallback
|
||||
|
||||
+12
-22
@@ -10,28 +10,20 @@ This guide covers installing last30days on Hermes AI Agent.
|
||||
|
||||
## Installation
|
||||
|
||||
### Option 1: Via sync.sh (Recommended)
|
||||
|
||||
```bash
|
||||
# Clone the repo
|
||||
git clone https://github.com/mvanhorn/last30days-skill.git
|
||||
cd last30days-skill
|
||||
|
||||
# Run the sync script
|
||||
bash skills/last30days/scripts/sync.sh
|
||||
hermes skills install mvanhorn/last30days-skill --force
|
||||
```
|
||||
|
||||
This will auto-detect Hermes and deploy to `~/.hermes/skills/research/last30days/`
|
||||
This pulls the latest release from GitHub and deploys to `~/.hermes/skills/research/last30days/`. `--force` reinstalls over any existing copy.
|
||||
|
||||
### Option 2: Manual Copy
|
||||
### Developer / live-edit alternative
|
||||
|
||||
If you're hacking on the skill locally and want edits to propagate to Hermes without re-installing, symlink your working tree:
|
||||
|
||||
```bash
|
||||
# Create directory
|
||||
mkdir -p ~/.hermes/skills/research/last30days
|
||||
|
||||
# Copy files
|
||||
cp skills/last30days/SKILL.md ~/.hermes/skills/research/last30days/
|
||||
cp -r skills/last30days/scripts ~/.hermes/skills/research/last30days/
|
||||
git clone https://github.com/mvanhorn/last30days-skill.git
|
||||
mkdir -p ~/.hermes/skills/research
|
||||
ln -s "$(pwd)/last30days-skill/skills/last30days" ~/.hermes/skills/research/last30days
|
||||
```
|
||||
|
||||
## Usage
|
||||
@@ -59,7 +51,7 @@ On first run, the skill will guide you through setup:
|
||||
|
||||
2. **Optional: ScrapeCreators**
|
||||
- Adds TikTok, Instagram, Reddit backup
|
||||
- 10,000 free API calls
|
||||
- 100 free credits (no expiration)
|
||||
- Sign up at scrapecreators.com
|
||||
|
||||
3. **Optional: API Keys**
|
||||
@@ -106,14 +98,12 @@ python3.12 scripts/last30days.py --diagnose
|
||||
|
||||
## Updating
|
||||
|
||||
To update to the latest version:
|
||||
|
||||
```bash
|
||||
cd last30days-skill
|
||||
git pull
|
||||
bash skills/last30days/scripts/sync.sh
|
||||
hermes skills install mvanhorn/last30days-skill --force
|
||||
```
|
||||
|
||||
If you symlinked your working tree (developer alternative above), just `git pull` in the repo — edits propagate live, no re-install step.
|
||||
|
||||
## Support
|
||||
|
||||
- Original repo: https://github.com/mvanhorn/last30days-skill
|
||||
|
||||
@@ -14,21 +14,18 @@
|
||||
|
||||
This README tracks the current v3 pipeline. The runtime skill spec lives in [SKILL.md](SKILL.md), which is the source of truth for the latest command and setup behavior.
|
||||
|
||||
Claude Code:
|
||||
**Claude Code (recommended — auto-updates via marketplace):**
|
||||
```
|
||||
/plugin marketplace add mvanhorn/last30days-skill
|
||||
```
|
||||
|
||||
OpenClaw:
|
||||
**Codex, Cursor, Copilot, Gemini CLI, or any of 50+ [Agent Skills](https://agentskills.io) hosts:**
|
||||
```
|
||||
clawhub install last30days-official
|
||||
npx skills add mvanhorn/last30days-skill -g
|
||||
```
|
||||
(`-g` installs globally for your user, available across all projects. Drop it to scope per-project.)
|
||||
|
||||
Hermes:
|
||||
```
|
||||
# The skill auto-deploys when you run sync.sh
|
||||
# Or manually copy to ~/.hermes/skills/research/last30days/
|
||||
```
|
||||
More install options (claude.ai web, OpenClaw, manual) in the [Install](#install) section below.
|
||||
|
||||
Zero config. Reddit, HN, Polymarket, and GitHub work immediately. Run it once and the setup wizard unlocks X, YouTube, TikTok, and more in 30 seconds.
|
||||
|
||||
@@ -68,7 +65,7 @@ If you're meeting with a CEO, have you read all their tweets and YouTube transcr
|
||||
| **Hacker News** | The developer consensus. 825 points, 899 comments. Where technical people actually argue. |
|
||||
| **Polymarket** | Not opinions. Odds. Backed by real money. 96% confidence on album sales. 4% on an acquisition. |
|
||||
| **GitHub** | For people: PR velocity, top repos by stars, release notes. For topics: issues and discussions. |
|
||||
| **Digg AI 1000** | Curated story clusters from ~1000 high-signal AI accounts on X, with attributable inline quotes (no X auth required). Auto-enabled when `digg-pp-cli` is on PATH. |
|
||||
| **Digg** | Curated story clusters from Digg's AI 1000 leaderboard (~1000 high-signal AI accounts on X), with attributable inline quotes (no X auth required). Auto-enabled when `digg-pp-cli` is on PATH. |
|
||||
| **Threads** | The post-Twitter text layer. Conversations from creators and brands. |
|
||||
| **Pinterest** | Visual discovery. Pins, saves, and comments on products and ideas. |
|
||||
| **Bluesky** | The decentralized social layer. AT Protocol posts from the post-Twitter migration. |
|
||||
@@ -155,8 +152,10 @@ Say "eli5 on" after any research run. The synthesis rewrites in plain language.
|
||||
|
||||
- **Free Reddit comments.** Public JSON gives you threads + top comments with upvote counts. No API key, no ScrapeCreators. Just works.
|
||||
- **YouTube transcripts that actually work.** Widened candidate pool 3x past music videos to reach talk/review content with captions.
|
||||
- **Threads, Pinterest, YouTube + TikTok comments.** Opt-in sources via ScrapeCreators. Set `INCLUDE_SOURCES=tiktok,instagram` and add threads, pinterest, youtube_comments, tiktok_comments for more. `youtube_comments` and `tiktok_comments` surface top comments with vote counts the same way Reddit does.
|
||||
- **Perplexity Sonar.** Grounded web search with citations via OpenRouter. Add `OPENROUTER_API_KEY` to unlock.
|
||||
- **TikTok, Instagram, Threads.** All three activate automatically once `SCRAPECREATORS_API_KEY` is set — same key, same per-call cost. Suppress any of them with `EXCLUDE_SOURCES=tiktok,instagram,threads` (any comma-separated subset).
|
||||
- **Pinterest.** Per-query opt-in (visual pins, narrow utility): the model passes `--search=pinterest` for the runs that need it. Requires `SCRAPECREATORS_API_KEY`.
|
||||
- **YouTube + TikTok comments.** Persistent opt-in via `INCLUDE_SOURCES=youtube_comments,tiktok_comments` because each video pulls N extra ScrapeCreators calls on top of the base search. Surface top comments with vote counts the same way Reddit does.
|
||||
- **Perplexity Sonar.** Grounded web search with citations via OpenRouter. Add `OPENROUTER_API_KEY` and `INCLUDE_SOURCES=perplexity` (it's a separate paid API — opt-in keeps you from being surprise-billed).
|
||||
- **Polymarket noise filtering.** Common-word disambiguation prevents "Apple" from matching "Will Apple release a car?"
|
||||
- **Resilient Reddit.** Timeout budgets and runtime fallback. One slow thread doesn't kill the whole run.
|
||||
- **Fun judge v2.** Humor scoring baked into the narrative. Reddit's cleverest one-liners mixed into the synthesis where they fit, not dumped in a separate section.
|
||||
@@ -168,12 +167,61 @@ Say "eli5 on" after any research run. The synthesis rewrites in plain language.
|
||||
|
||||
## Install
|
||||
|
||||
| Surface | Install |
|
||||
|---------|---------|
|
||||
| **claude.ai** (web) | [Download `last30days.skill`](https://github.com/mvanhorn/last30days-skill/releases/latest/download/last30days.skill) and upload via Settings > Capabilities > Skills > + |
|
||||
| **Claude Code** | `/plugin marketplace add mvanhorn/last30days-skill` |
|
||||
| **OpenClaw** | `clawhub install last30days-official` |
|
||||
| **Gemini CLI** | Clone then `gemini extensions install ./last30days-skill` (see below) |
|
||||
| Surface | Install | Updates |
|
||||
|---------|---------|---------|
|
||||
| **Claude Code** (recommended) | `/plugin marketplace add mvanhorn/last30days-skill` | Auto via marketplace, or `claude plugin update last30days@last30days-skill` |
|
||||
| **Codex, Cursor, Copilot, Gemini CLI, GitHub Copilot, or any of 50+ [Agent Skills](https://agentskills.io) hosts** | `npx skills add mvanhorn/last30days-skill -g` | `npx skills update last30days -g` |
|
||||
| **claude.ai** (web) | [Download `last30days.skill`](https://github.com/mvanhorn/last30days-skill/releases/latest/download/last30days.skill) and upload via Settings > Capabilities > Skills > + | Re-download and re-upload |
|
||||
| **OpenClaw** | `clawhub install last30days-official` | `clawhub update last30days-official` |
|
||||
|
||||
### Claude Code (recommended)
|
||||
|
||||
```
|
||||
/plugin marketplace add mvanhorn/last30days-skill
|
||||
```
|
||||
|
||||
Recommended because the Claude Code marketplace handles updates for you — the plugin cache is versioned and auto-refreshes when a new release publishes. Run `claude plugin update last30days@last30days-skill` to force a check.
|
||||
|
||||
If you'd rather use the agent-skills install path on Claude Code, that's also supported:
|
||||
|
||||
```
|
||||
npx skills add mvanhorn/last30days-skill -g -a claude-code
|
||||
```
|
||||
|
||||
The native plugin and the `npx skills` install can coexist; Claude Code dedupes the slash command.
|
||||
|
||||
### Codex, Cursor, Copilot, Gemini CLI, and other Agent Skills hosts
|
||||
|
||||
Install via the open [Agent Skills](https://agentskills.io) CLI — supports 50+ harnesses including `codex`, `cursor`, `github-copilot`, `gemini-cli`, `claude-code`, `windsurf`, `cline`, `continue`, `roo`, `aider-desk`, `opencode`, `goose`, and more (full list on the [vercel-labs/skills repo](https://github.com/vercel-labs/skills)).
|
||||
|
||||
```bash
|
||||
npx skills add mvanhorn/last30days-skill -g
|
||||
```
|
||||
|
||||
The `-g` (global) flag installs to your user directory so the skill is available across all projects. Without `-g`, `npx skills` installs project-locally into `./.skills/` (committed with the repo). For a research-the-world tool, global is what you want.
|
||||
|
||||
By default this installs for whichever harness `npx skills` detects. To target a specific one (or multiple):
|
||||
|
||||
```bash
|
||||
npx skills add mvanhorn/last30days-skill -g -a codex
|
||||
npx skills add mvanhorn/last30days-skill -g -a cursor
|
||||
npx skills add mvanhorn/last30days-skill -g -a gemini-cli
|
||||
npx skills add mvanhorn/last30days-skill -g -a codex -a cursor
|
||||
```
|
||||
|
||||
Update later with:
|
||||
|
||||
```bash
|
||||
npx skills update last30days -g
|
||||
```
|
||||
|
||||
Or update everything you've installed globally via `npx skills`:
|
||||
|
||||
```bash
|
||||
npx skills update -g
|
||||
```
|
||||
|
||||
List and remove with `npx skills list -g` and `npx skills remove last30days -g`.
|
||||
|
||||
### claude.ai (web)
|
||||
|
||||
@@ -181,15 +229,7 @@ Say "eli5 on" after any research run. The synthesis rewrites in plain language.
|
||||
2. Go to [claude.ai Settings > Capabilities > Skills](https://claude.ai/settings/capabilities)
|
||||
3. Click the `+` button in the Skills panel and drop the file in
|
||||
|
||||
Enable "Code execution and file creation" under Capabilities first - skills won't run without it.
|
||||
|
||||
### Claude Code
|
||||
|
||||
```
|
||||
/plugin marketplace add mvanhorn/last30days-skill
|
||||
```
|
||||
|
||||
Update later with `claude plugin update last30days@last30days-skill`.
|
||||
Enable "Code execution and file creation" under Capabilities first — skills won't run without it.
|
||||
|
||||
### OpenClaw
|
||||
|
||||
@@ -197,22 +237,14 @@ Update later with `claude plugin update last30days@last30days-skill`.
|
||||
clawhub install last30days-official
|
||||
```
|
||||
|
||||
### Gemini CLI
|
||||
|
||||
Gemini CLI v0.9.0 has an upstream installer bug that can fail with `Configuration file not found at /tmp/gemini-extensionXXXXXX/gemini-extension.json` ([upstream issue](https://github.com/google-gemini/gemini-cli/issues/11452)). Workaround:
|
||||
|
||||
```bash
|
||||
git clone https://github.com/mvanhorn/last30days-skill
|
||||
gemini extensions install ./last30days-skill
|
||||
```
|
||||
|
||||
### Manual (developer)
|
||||
|
||||
```bash
|
||||
git clone https://github.com/mvanhorn/last30days-skill.git ~/.claude/skills/last30days
|
||||
git clone https://github.com/mvanhorn/last30days-skill.git
|
||||
ln -s "$(pwd)/last30days-skill/skills/last30days" ~/.claude/skills/last30days
|
||||
```
|
||||
|
||||
Or build the claude.ai `.skill` file from source: `bash skills/last30days/scripts/build-skill.sh` produces `dist/last30days.skill`.
|
||||
The symlink keeps the install in sync with your working tree as you edit — no re-copy needed. For `claude.ai`, build the `.skill` file from source: `bash skills/last30days/scripts/build-skill.sh` produces `dist/last30days.skill`.
|
||||
|
||||
Reddit (with comments), Hacker News, Polymarket, and GitHub work immediately. Zero configuration. Run `/last30days` once and the setup wizard unlocks more sources in 30 seconds.
|
||||
|
||||
@@ -226,10 +258,28 @@ These platforms don't have relationships with each other. X doesn't know what Re
|
||||
| X / Twitter | Log into x.com in any browser | Free |
|
||||
| YouTube | `brew install yt-dlp` | Free |
|
||||
| Bluesky | App password from bsky.app | Free |
|
||||
| TikTok + Instagram + Threads + Pinterest + YouTube comments | ScrapeCreators key | 10,000 free calls |
|
||||
| TikTok + Instagram + Threads + Pinterest + YouTube comments | ScrapeCreators key | 100 free credits, then PAYG |
|
||||
| Perplexity Sonar | OpenRouter key | Pay as you go |
|
||||
| Web search | Brave Search key | 2,000 free queries/month |
|
||||
|
||||
### macOS Keychain (optional)
|
||||
|
||||
On macOS you can store keys in the system Keychain instead of a `.env` file. The skill picks them up automatically as the lowest-priority source — `.env` files and process environment still win on collision.
|
||||
|
||||
```bash
|
||||
# Interactive setup — prompts for each known key, skip with empty input
|
||||
skills/last30days/scripts/setup-keychain.sh
|
||||
|
||||
# Or store a single key by hand
|
||||
security add-generic-password -a "$USER" -s last30days-XAI_API_KEY -w "xai-..."
|
||||
|
||||
# Inspect / clean up
|
||||
skills/last30days/scripts/setup-keychain.sh --list
|
||||
skills/last30days/scripts/setup-keychain.sh --delete XAI_API_KEY
|
||||
```
|
||||
|
||||
Items are stored under service name `last30days-<KEY>` for the current user. On non-Darwin platforms the loader is a no-op, so there is no behaviour change for Linux/Windows users.
|
||||
|
||||
## How it works
|
||||
|
||||
1. **You type a topic.** Person, company, product, technology, "X vs Y." Anything.
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
# last30days Skill Specification
|
||||
|
||||
## Overview
|
||||
|
||||
`last30days` is a Claude Code skill that researches a given topic across Reddit and X (Twitter) using the OpenAI Responses API and xAI Responses API respectively. It enforces a strict 30-day recency window, popularity-aware ranking, and produces actionable outputs including best practices, a prompt pack, and a reusable context snippet. OpenAI auth can come from `OPENAI_API_KEY` or Codex login credentials.
|
||||
|
||||
The skill operates in three modes depending on available API keys: **reddit-only** (OpenAI key), **x-only** (xAI key), or **both** (full cross-validation). It uses automatic model selection to stay current with the latest models from both providers, with optional pinning for stability.
|
||||
|
||||
## Architecture
|
||||
|
||||
The orchestrator (`last30days.py`) coordinates discovery, enrichment, normalization, scoring, deduplication, and rendering. Each concern is isolated in `scripts/lib/`:
|
||||
|
||||
- **env.py**: Load API keys from `~/.config/last30days/.env` and Codex auth from `~/.codex/auth.json`
|
||||
- **dates.py**: Date range calculation and confidence scoring
|
||||
- **cache.py**: 24-hour TTL caching keyed by topic + date range
|
||||
- **http.py**: stdlib-only HTTP client with retry logic
|
||||
- **models.py**: Auto-selection of OpenAI/xAI models with 7-day caching
|
||||
- **openai_reddit.py**: OpenAI Responses API + web_search for Reddit
|
||||
- **xai_x.py**: xAI Responses API + x_search for X
|
||||
- **reddit_enrich.py**: Fetch Reddit thread JSON for real engagement metrics
|
||||
- **hackernews.py**: Hacker News search via Algolia API (free, no auth)
|
||||
- **polymarket.py**: Polymarket prediction market search via Gamma API (free, no auth)
|
||||
- **normalize.py**: Convert raw API responses to canonical schema
|
||||
- **score.py**: Compute popularity-aware scores (relevance + recency + engagement)
|
||||
- **dedupe.py**: Near-duplicate detection via text similarity
|
||||
- **render.py**: Generate markdown and JSON outputs
|
||||
- **schema.py**: Type definitions and validation
|
||||
|
||||
## Embedding in Other Skills
|
||||
|
||||
Other skills can import the research context in several ways:
|
||||
|
||||
### Inline Context Injection
|
||||
```markdown
|
||||
## Recent Research Context
|
||||
!python3 ~/.claude/skills/last30days/scripts/last30days.py "your topic" --emit=context
|
||||
```
|
||||
|
||||
### Read from File
|
||||
```markdown
|
||||
## Research Context
|
||||
!cat ~/.local/share/last30days/out/last30days.context.md
|
||||
```
|
||||
|
||||
### Get Path for Dynamic Loading
|
||||
```bash
|
||||
CONTEXT_PATH=$(python3 ~/.claude/skills/last30days/scripts/last30days.py "topic" --emit=path)
|
||||
cat "$CONTEXT_PATH"
|
||||
```
|
||||
|
||||
### JSON for Programmatic Use
|
||||
```bash
|
||||
python3 ~/.claude/skills/last30days/scripts/last30days.py "topic" --emit=json > research.json
|
||||
```
|
||||
|
||||
## CLI Reference
|
||||
|
||||
```
|
||||
python3 ~/.claude/skills/last30days/scripts/last30days.py <topic> [options]
|
||||
|
||||
Options:
|
||||
--refresh Bypass cache and fetch fresh data
|
||||
--mock Use fixtures instead of real API calls
|
||||
--emit=MODE Output mode: compact|json|md|context|path (default: compact)
|
||||
--sources=MODE Source selection: auto|reddit|x|both (default: auto)
|
||||
```
|
||||
|
||||
## Output Files
|
||||
|
||||
All outputs are written to `~/.local/share/last30days/out/`:
|
||||
|
||||
- `report.md` - Human-readable full report
|
||||
- `report.json` - Normalized data with scores
|
||||
- `last30days.context.md` - Compact reusable snippet for other skills
|
||||
- `raw_openai.json` - Raw OpenAI API response
|
||||
- `raw_xai.json` - Raw xAI API response
|
||||
- `raw_reddit_threads_enriched.json` - Enriched Reddit thread data
|
||||
@@ -1,47 +0,0 @@
|
||||
# last30days Implementation Tasks
|
||||
|
||||
## Setup & Configuration
|
||||
- [x] Create directory structure
|
||||
- [x] Write SPEC.md
|
||||
- [x] Write TASKS.md
|
||||
- [x] Write SKILL.md with proper frontmatter
|
||||
|
||||
## Core Library Modules
|
||||
- [x] scripts/lib/env.py - Environment and API key loading
|
||||
- [x] scripts/lib/dates.py - Date range and confidence utilities
|
||||
- [x] scripts/lib/cache.py - TTL-based caching
|
||||
- [x] scripts/lib/http.py - HTTP client with retry
|
||||
- [x] scripts/lib/models.py - Auto model selection
|
||||
- [x] scripts/lib/schema.py - Data structures
|
||||
- [x] scripts/lib/openai_reddit.py - OpenAI Responses API
|
||||
- [x] scripts/lib/xai_x.py - xAI Responses API
|
||||
- [x] scripts/lib/reddit_enrich.py - Reddit thread JSON fetcher
|
||||
- [x] scripts/lib/normalize.py - Schema normalization
|
||||
- [x] scripts/lib/score.py - Popularity scoring
|
||||
- [x] scripts/lib/dedupe.py - Near-duplicate detection
|
||||
- [x] scripts/lib/render.py - Output rendering
|
||||
|
||||
## Main Script
|
||||
- [x] scripts/last30days.py - CLI orchestrator
|
||||
|
||||
## Fixtures
|
||||
- [x] fixtures/openai_sample.json
|
||||
- [x] fixtures/xai_sample.json
|
||||
- [x] fixtures/reddit_thread_sample.json
|
||||
- [x] fixtures/models_openai_sample.json
|
||||
- [x] fixtures/models_xai_sample.json
|
||||
|
||||
## Tests
|
||||
- [x] tests/test_dates.py
|
||||
- [x] tests/test_cache.py
|
||||
- [x] tests/test_models.py
|
||||
- [x] tests/test_score.py
|
||||
- [x] tests/test_dedupe.py
|
||||
- [x] tests/test_normalize.py
|
||||
- [x] tests/test_render.py
|
||||
|
||||
## Validation
|
||||
- [x] Run tests in mock mode
|
||||
- [x] Demo --emit=compact
|
||||
- [x] Demo --emit=context
|
||||
- [x] Verify file tree
|
||||
@@ -1,4 +1,7 @@
|
||||
---
|
||||
|
||||
> **NOTE (added 2026-05-16):** This plan references `bash scripts/sync.sh`. That script was deleted in [PR #405](https://github.com/mvanhorn/last30days-skill/pull/405); the install workflow is now `npx skills add . -g -y` (symlinks the working tree across every detected harness). For context on why sync.sh went away, see [docs/solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md](../solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md). The decisions captured in this plan remain accurate; only the deploy mechanism changed.
|
||||
|
||||
title: "feat: --competitors flag for auto-discovered comparison fan-out"
|
||||
type: feat
|
||||
status: active
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
---
|
||||
|
||||
> **NOTE (added 2026-05-16):** This plan references `bash scripts/sync.sh`. That script was deleted in [PR #405](https://github.com/mvanhorn/last30days-skill/pull/405); the install workflow is now `npx skills add . -g -y` (symlinks the working tree across every detected harness). For context on why sync.sh went away, see [docs/solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md](../solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md). The decisions captured in this plan remain accurate; only the deploy mechanism changed.
|
||||
|
||||
title: "fix: per-entity resolution, default-2, and stale-path guard for --competitors"
|
||||
type: fix
|
||||
status: active
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
---
|
||||
|
||||
> **NOTE (added 2026-05-16):** This plan references `bash scripts/sync.sh`. That script was deleted in [PR #405](https://github.com/mvanhorn/last30days-skill/pull/405); the install workflow is now `npx skills add . -g -y` (symlinks the working tree across every detected harness). For context on why sync.sh went away, see [docs/solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md](../solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md). The decisions captured in this plan remain accurate; only the deploy mechanism changed.
|
||||
|
||||
title: "feat: vs mode runs N full passes and --competitors is vs with auto-discovery"
|
||||
type: feat
|
||||
status: active
|
||||
|
||||
@@ -1,4 +1,7 @@
|
||||
---
|
||||
|
||||
> **NOTE (added 2026-05-16):** This plan references `bash scripts/sync.sh`. That script was deleted in [PR #405](https://github.com/mvanhorn/last30days-skill/pull/405); the install workflow is now `npx skills add . -g -y` (symlinks the working tree across every detected harness). For context on why sync.sh went away, see [docs/solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md](../solutions/workflow-issues/release-consistency-test-cascade-2026-05-16.md). The decisions captured in this plan remain accurate; only the deploy mechanism changed.
|
||||
|
||||
title: "fix: comparison title says (/Last30Days) instead of (Last 30 Days)"
|
||||
type: fix
|
||||
status: active
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
---
|
||||
title: Search-quality eval is manual by default, not a CI gate on every PR
|
||||
date: 2026-05-10
|
||||
category: docs/solutions/architecture
|
||||
module: skills/last30days/scripts/evaluate_search_quality.py
|
||||
problem_type: design_decision
|
||||
component: ci_policy
|
||||
severity: low
|
||||
applies_when:
|
||||
- a contributor proposes wiring search-quality eval into PR CI
|
||||
- a change affects retrieval, ranking, grounding, or synthesis quality and a reviewer asks "why aren't we testing this in CI?"
|
||||
- someone is deciding whether a new evaluator-style script belongs in the default CI workflow
|
||||
related_components:
|
||||
- search_quality_evaluation
|
||||
- ci_workflow
|
||||
- llm_judging
|
||||
tags:
|
||||
- ci-policy
|
||||
- eval
|
||||
- design-decision
|
||||
- cost-vs-signal
|
||||
- non-determinism
|
||||
- manual-gates
|
||||
---
|
||||
|
||||
# Search-quality eval is manual by default, not a CI gate on every PR
|
||||
|
||||
## Context
|
||||
|
||||
`skills/last30days/scripts/evaluate_search_quality.py` compares a baseline revision against a candidate revision across a fixed pool of reviewer topics. It produces two flavors of metrics: deterministic overlap (Jaccard, retention) and LLM-judged quality scores. The natural impulse on seeing an evaluator script is to wire it into CI on every PR — "regression catcher, run it automatically." We deliberately don't.
|
||||
|
||||
Three properties of this particular evaluator make CI-on-every-PR the wrong default:
|
||||
|
||||
1. **Live API access.** The candidate revision typically needs the engine to actually run, which means real ScrapeCreators calls, real reddit fetches, real YouTube searches. CI runs would either need production credentials or a record/replay fixture set that drifts almost immediately as external APIs change shape.
|
||||
|
||||
2. **Cost and latency.** A full eval pass runs the pipeline N times across reviewer topics. Multiplied by every PR (including doc-only PRs), the spend is meaningful and the wall-clock pushes CI from ~30s to many minutes.
|
||||
|
||||
3. **Non-determinism in the judging path.** The LLM-judged metrics are valuable for review but depend on judge-model behavior on a given day. A flaky eval that fails 1 PR in 20 because the judge re-scored an item differently is a worse CI signal than no eval at all — it teaches contributors to retry rather than read the result.
|
||||
|
||||
The deterministic overlap metrics are useful regression signals but they are not the same as user-facing correctness. A change that improves overlap can degrade synthesis quality; a change that drops overlap can be a deliberate improvement. So even the deterministic side isn't safe to auto-fail on.
|
||||
|
||||
## Guidance
|
||||
|
||||
### 1. Keep search-quality eval available, just not automatic
|
||||
|
||||
The script stays runnable by maintainers and contributors. The pattern is:
|
||||
|
||||
```bash
|
||||
LAST30DAYS_PYTHON=python3.13 \
|
||||
python3 skills/last30days/scripts/evaluate_search_quality.py \
|
||||
--baseline main --candidate HEAD
|
||||
```
|
||||
|
||||
Reviewers can request a manual eval run when a PR is in the retrieval/ranking/synthesis path and the risk warrants it. Contributors can run it locally before submitting if they want signal upfront.
|
||||
|
||||
### 2. Standard PR CI gates remain deterministic and contract-shaped
|
||||
|
||||
`pytest` (offline-safe), plugin-contract checks, version-consistency contracts, ruff/lint. Anything that returns the same answer twice for the same input. Quality-of-output assessment lives outside that loop.
|
||||
|
||||
### 3. The middle ground is `workflow_dispatch`, not auto-PR-gating
|
||||
|
||||
If maintainers want a GitHub-triggered eval that doesn't make every PR pay the live-API cost, the right shape is a manually-dispatched workflow (or a label-triggered one) — not a `pull_request:` workflow that runs unconditionally. That keeps the cost knob in human hands.
|
||||
|
||||
### 4. Revisit if the eval can ever be made offline-deterministic
|
||||
|
||||
The blocker is the live-API + non-determinism combination. If a future iteration of the script can compute meaningful Jaccard/retention metrics against static fixtures (no live API calls, no LLM judging), the decision flips and it becomes a candidate for default CI. The decision below tracks that condition; revisit when it's met.
|
||||
|
||||
## What this means in practice
|
||||
|
||||
- Don't merge PRs that wire `evaluate_search_quality.py` into the default `validate.yml` workflow.
|
||||
- Do merge PRs that add `workflow_dispatch` triggers or label-gated runs.
|
||||
- When reviewing a retrieval/ranking change, request a manual eval if the diff suggests it could regress quality — don't expect CI to catch it.
|
||||
|
||||
## Links
|
||||
|
||||
- `skills/last30days/scripts/evaluate_search_quality.py` — the evaluator script
|
||||
- `docs/search-quality-eval.md` — user-facing usage documentation
|
||||
- `.github/workflows/validate.yml` — the default CI workflow (deterministic gates only)
|
||||
|
||||
---
|
||||
|
||||
*Adapted from a draft ADR proposed by @hnshah in [#374](https://github.com/mvanhorn/last30days-skill/pull/374), restructured into the `docs/solutions/` convention. The original ADR text correctly identified the constraint; this version adds the "why workflow_dispatch is the middle ground" framing and the revisit-condition.*
|
||||
@@ -0,0 +1,219 @@
|
||||
---
|
||||
title: Release-time consistency tests cause cascade CI failures across all open PRs
|
||||
date: 2026-05-16
|
||||
category: docs/solutions/workflow-issues
|
||||
module: ci-release-engineering
|
||||
problem_type: workflow_issue
|
||||
component: testing_framework
|
||||
severity: high
|
||||
applies_when:
|
||||
- a test asserts consistency between two release-time artifacts (e.g., SKILL.md version and a hardcoded pin in a shell script)
|
||||
- one artifact is updated as part of a version bump and the other requires a manual lockstep update
|
||||
- multiple long-lived PRs are open simultaneously against the same base branch
|
||||
symptoms:
|
||||
- every open PR's CI fails after a version bump even though the PRs are unrelated to versioning
|
||||
- the failing test references a stale hardcoded value that was not updated alongside the bumped version
|
||||
- PR authors must rebase and manually fix an artifact they did not touch
|
||||
root_cause: missing_workflow_step
|
||||
resolution_type: code_fix
|
||||
related_components:
|
||||
- development_workflow
|
||||
- documentation
|
||||
tags:
|
||||
- ci
|
||||
- release-engineering
|
||||
- consistency-test
|
||||
- version-pin
|
||||
- cascade-failure
|
||||
- test-design
|
||||
- workflow
|
||||
---
|
||||
|
||||
# Release-time consistency tests cause cascade CI failures across all open PRs
|
||||
|
||||
## Context
|
||||
|
||||
A `tests/test_version_consistency.py::test_sync_cache_path_uses_skill_version` test was added to enforce that the version string embedded in `skills/last30days/scripts/sync.sh` (a hardcoded plugin-cache path segment) matched the version frontmatter in `skills/last30days/SKILL.md`. The intention was sound: the cache path had to stay in lockstep with the skill version or the sync would silently pull stale files.
|
||||
|
||||
The test worked as designed until a release shipped. At that point it turned into a cascade-failure machine:
|
||||
|
||||
1. A release PR bumps `SKILL.md` version (e.g., 3.2.0 → 3.2.1) **and** bumps the `sync.sh` pin. That PR's CI is green.
|
||||
2. The release PR merges to `main`.
|
||||
3. Every PR that was open at merge time was branched from pre-release `main`. Those PRs have `SKILL.md` 3.2.1 (inherited via merge-base with `main`) but their branch never touched `sync.sh`.
|
||||
4. CI for those PRs runs the consistency test against the new `main` — `SKILL.md` says 3.2.1, `sync.sh` still says 3.2.0 — and fails.
|
||||
5. All open PRs are now red simultaneously, with a failure that has nothing to do with their changes.
|
||||
|
||||
This affected at least five PRs during the 2026-05-13 to 2026-05-15 window: PR #400 (caught during rebase, required a manual pin bump), PRs #390 and #392 (OpenClaw `SCRAPECREATORS_API_KEY` fix, both stalled for the same stale-pin reason), and at least two others. A follow-up hotfix PR (#397 — `fix(sync): bump cache target to 3.2.1 to match SKILL.md`) was required just to unblock the queue.
|
||||
|
||||
The permanent fix was PR #405: delete `sync.sh` entirely (the install workflow made it redundant) and drop `test_sync_cache_path_uses_skill_version`. Once both were gone, no version-consistency cascade was possible.
|
||||
|
||||
## Guidance
|
||||
|
||||
### 1. Don't write consistency tests that read two files and assert one matches a substring derived from the other
|
||||
|
||||
This pattern looks safe but is not:
|
||||
|
||||
```python
|
||||
def test_sync_cache_path_uses_skill_version(self) -> None:
|
||||
sync_text = (SKILL_ROOT / "scripts" / "sync.sh").read_text(encoding="utf-8")
|
||||
version = _skill_version() # reads SKILL.md
|
||||
self.assertIn(
|
||||
f'last30days-skill/last30days/{version}"',
|
||||
sync_text, # asserts sync.sh contains that string
|
||||
)
|
||||
```
|
||||
|
||||
It encodes the assumption that both files are always updated together, in the same commit, on the same branch. That assumption breaks the moment two files have independent lifecycle owners — a versioned manifest and a deployment script are archetypal examples.
|
||||
|
||||
### 2. If the values genuinely need to stay in sync, derive one from the other at runtime
|
||||
|
||||
Remove the hardcoded pin from `sync.sh` and compute it:
|
||||
|
||||
```bash
|
||||
# sync.sh — derive version from SKILL.md at runtime, no pin to maintain
|
||||
SKILL_VERSION=$(grep -m1 '^version:' "$(dirname "$0")/../SKILL.md" \
|
||||
| sed 's/version:[[:space:]]*"\([^"]*\)"/\1/')
|
||||
CACHE_PATH="last30days-skill/last30days/${SKILL_VERSION}"
|
||||
```
|
||||
|
||||
Now there is only one source of truth (`SKILL.md`). The test that asserted they matched becomes vacuous and should be deleted. If `SKILL.md` is wrong, the sync itself will fail loudly — which is better feedback than a CI gate on a different PR.
|
||||
|
||||
### 3. If two values must stay independent for legitimate reasons, update them together and make the test self-skip if either source is missing
|
||||
|
||||
If separate versioning is genuinely required (e.g., SKILL.md versions for harness consumers, sync.sh versions a private artifact store with its own cadence), update both in the same PR — never staggered — and write the test to self-skip rather than error when either file is absent:
|
||||
|
||||
```python
|
||||
def test_sync_cache_path_uses_skill_version(self) -> None:
|
||||
sync_sh = SKILL_ROOT / "scripts" / "sync.sh"
|
||||
if not sync_sh.exists():
|
||||
self.skipTest("sync.sh not present; skipping pin consistency check")
|
||||
sync_text = sync_sh.read_text(encoding="utf-8")
|
||||
version = _skill_version()
|
||||
self.assertIn(
|
||||
f'last30days-skill/last30days/{version}"',
|
||||
sync_text,
|
||||
)
|
||||
```
|
||||
|
||||
Self-skipping means deleting the file is a non-event in CI — no cascading red, no hotfix PR to the queue.
|
||||
|
||||
### 4. Run consistency tests against the merge-base diff, not main
|
||||
|
||||
If you keep a two-file consistency test, scope it so it only fails when the PR itself modifies one of the two files but not the other. A GitHub Actions step can do this:
|
||||
|
||||
```yaml
|
||||
- name: Check sync.sh version pin consistency
|
||||
run: |
|
||||
BASE=$(git merge-base HEAD origin/main)
|
||||
SKILL_CHANGED=$(git diff --name-only "$BASE" HEAD | grep -c 'SKILL\.md' || true)
|
||||
SYNC_CHANGED=$(git diff --name-only "$BASE" HEAD | grep -c 'sync\.sh' || true)
|
||||
if [ "$SKILL_CHANGED" -gt 0 ] && [ "$SYNC_CHANGED" -eq 0 ]; then
|
||||
echo "SKILL.md version bumped but sync.sh pin was not updated"
|
||||
exit 1
|
||||
fi
|
||||
```
|
||||
|
||||
This only fires when your PR touched `SKILL.md` and left `sync.sh` alone — never because a release merged to `main` after you branched.
|
||||
|
||||
### 5. Ask whether you actually need this test
|
||||
|
||||
If the values are wrong, downstream tooling will fail loudly: the sync will fetch the wrong artifact, installs will break, or the harness will reject the version. A test that exists only to catch a human-bookkeeping error at release time adds cascade-fail risk without offering a meaningfully earlier signal. Weigh that cost before adding any two-file consistency gate.
|
||||
|
||||
## Why This Matters
|
||||
|
||||
The damage from a stale-pin consistency test is asymmetric. It:
|
||||
|
||||
- Fails on every open PR simultaneously the moment a release lands on `main` — not just the PR that forgot to update the pin.
|
||||
- Produces a failure message that points at a line in a test file with no obvious relationship to the PR's actual changes.
|
||||
- Requires either a hotfix PR (touching a file the failing PRs have no business touching) or a manual rebase of every affected branch.
|
||||
- Blocks work that has already been reviewed and approved.
|
||||
|
||||
In this repo the effect was measurable: at least five PRs stalled across a two-day window, one hotfix PR was shipped just to unblock the queue, and multiple authors spent time debugging a failure completely unrelated to their changes.
|
||||
|
||||
The broader principle is that tests which gate on *bookkeeping consistency between files* impose their maintenance cost on every contributor, every time, even when those contributors did nothing wrong. That cost compounds with team size and release cadence.
|
||||
|
||||
## When to Apply
|
||||
|
||||
Apply this guidance whenever you find yourself:
|
||||
|
||||
- Writing a test that reads two files and asserts that a string in one matches a value derived from the other.
|
||||
- Adding a CI step labeled "consistency check," "sync check," or "pin check" where the check compares a hardcoded value against a computed one from a separate file.
|
||||
- Working in a repo where a versioned manifest (e.g., `SKILL.md`, `package.json`, `pyproject.toml`) and a deployment artifact (e.g., a shell script, a Dockerfile, a Helm values file) are both maintained by hand.
|
||||
- Reviewing a PR that touches only one of two "paired" files and fails a consistency test for the other.
|
||||
|
||||
It does *not* apply to tests that read a single source of truth and validate its internal structure (e.g., asserting that `SKILL.md`'s frontmatter version is double-quoted, or that `package.json`'s `version` field is a valid semver string). Those tests have one file and one assertion; they cannot cascade across branches.
|
||||
|
||||
## Examples
|
||||
|
||||
### Before — the pattern that caused the cascade
|
||||
|
||||
Original `tests/test_version_consistency.py` (deleted in commit `9fb19ea`):
|
||||
|
||||
```python
|
||||
import re
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SKILL_ROOT = ROOT / "skills" / "last30days"
|
||||
|
||||
|
||||
def _skill_version() -> str:
|
||||
text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
||||
match = re.search(r'^version:\s*"([^"]+)"\s*$', text, re.MULTILINE)
|
||||
if not match:
|
||||
raise AssertionError("SKILL.md version frontmatter not found")
|
||||
return match.group(1)
|
||||
|
||||
|
||||
class TestVersionConsistency(unittest.TestCase):
|
||||
def test_sync_cache_path_uses_skill_version(self) -> None:
|
||||
sync_text = (SKILL_ROOT / "scripts" / "sync.sh").read_text(encoding="utf-8")
|
||||
version = _skill_version() # source 1: SKILL.md frontmatter
|
||||
self.assertIn( # assertion: sync.sh must contain
|
||||
f'last30days-skill/last30days/{version}"',
|
||||
sync_text, # source 2: hardcoded string in sync.sh
|
||||
)
|
||||
```
|
||||
|
||||
`sync.sh` contained a line like:
|
||||
|
||||
```bash
|
||||
PLUGIN_CACHE="$HOME/.cache/last30days-skill/last30days/3.2.0"
|
||||
```
|
||||
|
||||
When SKILL.md bumped to `3.2.1` in a release PR, `sync.sh` was updated in the same PR and CI stayed green. But every PR branched before that release still had `sync.sh` at `3.2.0`. Their CI failed immediately, with an assertion error pointing at the test, not at the release PR.
|
||||
|
||||
### After — what we did: delete both
|
||||
|
||||
PR #405 deleted `sync.sh` (the install workflow replaced it) and dropped `test_sync_cache_path_uses_skill_version` in the same change. No consistency gate, no pin to maintain, no cascade possible.
|
||||
|
||||
### After — what we could have done instead: derive at runtime
|
||||
|
||||
If `sync.sh` had still been needed, the right fix would have been to remove the hardcoded version from the script and derive it from `SKILL.md`:
|
||||
|
||||
```bash
|
||||
#!/usr/bin/env bash
|
||||
# sync.sh — no hardcoded version; reads SKILL.md as single source of truth
|
||||
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
SKILL_VERSION=$(grep -m1 '^version:' "${SCRIPT_DIR}/../SKILL.md" \
|
||||
| sed 's/version:[[:space:]]*"\([^"]*\)"/\1/')
|
||||
|
||||
if [ -z "$SKILL_VERSION" ]; then
|
||||
echo "error: could not parse version from SKILL.md" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
PLUGIN_CACHE="$HOME/.cache/last30days-skill/last30days/${SKILL_VERSION}"
|
||||
# ... rest of sync logic
|
||||
```
|
||||
|
||||
With this in place, `test_sync_cache_path_uses_skill_version` has no reason to exist — there is nothing to assert. Delete it. If the version parsing breaks, `sync.sh` itself exits non-zero with a clear message.
|
||||
|
||||
## Related
|
||||
|
||||
- **PR #397** (merged) — `fix(sync): bump cache target to 3.2.1 to match SKILL.md`. The hotfix that unblocked the cascade temporarily by bumping the pin.
|
||||
- **PR #400** (merged) — caught the same cascade during rebase; had to bump the pin to clear CI.
|
||||
- **PR #390** (closed) and **PR #392** (rebased + merged) — OpenClaw `SCRAPECREATORS_API_KEY` fix; both blocked by the cascade until rebased onto post-#405 main.
|
||||
- **PR #405** (merged) — the permanent fix: deleted `sync.sh` + `test_sync_cache_path_uses_skill_version` together.
|
||||
- **PR #412** (merged) — adjacent work that consolidated SKILL.md version parsing into `lib/skill_meta.py`, reducing future drift risk by giving the version field one canonical reader.
|
||||
@@ -0,0 +1,4 @@
|
||||
{
|
||||
"triggerOnUpdates": true,
|
||||
"statusCheck": true
|
||||
}
|
||||
@@ -97,7 +97,20 @@ if [[ -n "$HAS_BSKY" ]]; then
|
||||
SOURCE_COUNT=$((SOURCE_COUNT + 1))
|
||||
fi
|
||||
if [[ -n "$HAS_SCRAPECREATORS" ]]; then
|
||||
SOURCE_COUNT=$((SOURCE_COUNT + 3)) # Reddit comments + TikTok + Instagram
|
||||
# Start with Reddit comments + TikTok + Instagram, subtract any in EXCLUDE_SOURCES.
|
||||
# Normalise EXCLUDED (lowercase + collapse whitespace around commas + strip outer
|
||||
# whitespace) so the matching mirrors pipeline.py's .strip().lower() parsing.
|
||||
SC_ADD=3
|
||||
EXCLUDED="${ENV_EXCLUDE_SOURCES:-${EXCLUDE_SOURCES:-}}"
|
||||
EXCLUDED_NORM=$(printf '%s' "$EXCLUDED" | tr '[:upper:]' '[:lower:]' \
|
||||
| sed -E 's/[[:space:]]*,[[:space:]]*/,/g; s/^[[:space:]]+//; s/[[:space:]]+$//')
|
||||
if [[ ",$EXCLUDED_NORM," == *",tiktok,"* ]]; then
|
||||
SC_ADD=$((SC_ADD - 1))
|
||||
fi
|
||||
if [[ ",$EXCLUDED_NORM," == *",instagram,"* ]]; then
|
||||
SC_ADD=$((SC_ADD - 1))
|
||||
fi
|
||||
SOURCE_COUNT=$((SOURCE_COUNT + SC_ADD))
|
||||
fi
|
||||
|
||||
if [[ -n "$HAS_SCRAPECREATORS" ]]; then
|
||||
@@ -107,6 +120,6 @@ else
|
||||
# Setup done but missing ScrapeCreators — recommend it
|
||||
echo "/last30days: Ready — ${SOURCE_COUNT} sources active."
|
||||
echo " Tip: Add ScrapeCreators for Reddit comments + TikTok + Instagram."
|
||||
echo " 10,000 free API calls, no credit card — scrapecreators.com"
|
||||
echo " 100 free credits, no credit card — scrapecreators.com"
|
||||
echo " last30days has no affiliation with any API provider."
|
||||
fi
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 2.4 MiB |
+3
-5
@@ -1,16 +1,14 @@
|
||||
[project]
|
||||
name = "last30days-skill"
|
||||
version = "3.2.0"
|
||||
version = "3.2.4"
|
||||
description = "Multi-source last-30-days research skill"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
dependencies = [
|
||||
"requests>=2.32,<3",
|
||||
]
|
||||
dependencies = []
|
||||
|
||||
[dependency-groups]
|
||||
dev = [
|
||||
"pytest>=9,<10",
|
||||
"pytest>=9.0.3,<10",
|
||||
"pytest-cov>=7,<8",
|
||||
]
|
||||
|
||||
|
||||
+106
-65
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: last30days
|
||||
version: "3.2.0"
|
||||
version: "3.2.4"
|
||||
description: "Research what people actually say about any topic in the last 30 days. Pulls posts and engagement from Reddit, X, YouTube, TikTok, Hacker News, Polymarket, GitHub, and the web."
|
||||
argument-hint: 'last30days nvidia earnings reaction | last30days AI video tools | last30days what users want in react'
|
||||
allowed-tools: Bash, Read, Write, AskUserQuestion, WebSearch
|
||||
@@ -13,9 +13,9 @@ metadata:
|
||||
openclaw:
|
||||
emoji: "📰"
|
||||
requires:
|
||||
env:
|
||||
- SCRAPECREATORS_API_KEY
|
||||
env: []
|
||||
optionalEnv:
|
||||
- SCRAPECREATORS_API_KEY
|
||||
- OPENAI_API_KEY
|
||||
- XAI_API_KEY
|
||||
- OPENROUTER_API_KEY
|
||||
@@ -60,23 +60,32 @@ metadata:
|
||||
- clawhub
|
||||
---
|
||||
|
||||
# STEP 0: CANONICAL PATH SELF-CHECK — RUN BEFORE READING BELOW
|
||||
# STEP 0: STALE-CLONE SELF-CHECK — RUN BEFORE READING BELOW
|
||||
|
||||
Before reading anything else in this file, verify that the SKILL.md you just loaded is the versioned cache copy, not a stale marketplace clone:
|
||||
Before reading anything else in this file, check whether you loaded SKILL.md from the one known stale-clone location: Claude Code's marketplaces directory.
|
||||
|
||||
**Run this check:**
|
||||
|
||||
```bash
|
||||
CANONICAL_ROOT="$HOME/.claude/plugins/cache/last30days-skill/last30days"
|
||||
CANONICAL_LATEST=$(ls -d "$CANONICAL_ROOT"/*/ 2>/dev/null | sort -V | tail -1)
|
||||
CANONICAL_LATEST="${CANONICAL_LATEST%/}"
|
||||
CLAUDE_CACHE_LATEST=$(find "$HOME/.claude/plugins/cache/last30days-skill/last30days" -mindepth 1 -maxdepth 1 -type d 2>/dev/null | sort -V | tail -1)
|
||||
# Two cache layouts ship in the wild — nested ({version}/skills/last30days/SKILL.md)
|
||||
# and flat ({version}/SKILL.md). Resolve to whichever shape actually exists.
|
||||
CLAUDE_CACHE_SKILL_MD=""
|
||||
if [ -n "$CLAUDE_CACHE_LATEST" ]; then
|
||||
if [ -f "$CLAUDE_CACHE_LATEST/skills/last30days/SKILL.md" ]; then
|
||||
CLAUDE_CACHE_SKILL_MD="$CLAUDE_CACHE_LATEST/skills/last30days/SKILL.md"
|
||||
elif [ -f "$CLAUDE_CACHE_LATEST/SKILL.md" ]; then
|
||||
CLAUDE_CACHE_SKILL_MD="$CLAUDE_CACHE_LATEST/SKILL.md"
|
||||
fi
|
||||
fi
|
||||
echo "CLAUDE_CACHE_SKILL_MD=$CLAUDE_CACHE_SKILL_MD"
|
||||
```
|
||||
|
||||
If the SKILL.md file you just read is NOT under `$CANONICAL_LATEST/`, STOP and re-read `$CANONICAL_LATEST/SKILL.md` before proceeding. Do not run `--help`, do not plan, do not invoke the engine against any other path.
|
||||
If the SKILL.md path you just Read contains `/.claude/plugins/marketplaces/` AND `$CLAUDE_CACHE_SKILL_MD` is non-empty, STOP and re-read `$CLAUDE_CACHE_SKILL_MD` before proceeding. Otherwise the SKILL.md you have is fine — continue.
|
||||
|
||||
**Why:** `~/.claude/plugins/marketplaces/last30days-skill/` is a git clone Claude Code auto-restores to `origin/main` on session start. It can lag the versioned cache by one or more releases. Three 2026-04-22 test runs (Linear, Coinbase) loaded SKILL.md from `marketplaces/`, ran `--help` from the same stale path, did not see the `--competitors` flag that existed in the cache, and fell back to a manual comparison plan. Result: 2 of 3 windows never invoked the feature they were asked to test.
|
||||
**Why this specific check:** `~/.claude/plugins/marketplaces/last30days-skill/` is a git clone Claude Code auto-restores to `origin/main` on session start. It can lag the versioned cache by one or more releases. Three 2026-04-22 test runs (Linear, Coinbase) loaded SKILL.md from `marketplaces/`, ran `--help` from the same stale path, did not see the `--competitors` flag that existed in the cache, and fell back to a manual comparison plan. Result: 2 of 3 windows never invoked the feature they were asked to test. STEP 0 defends against that one Claude Code-specific bug.
|
||||
|
||||
**How to self-check:** the file path you used in your last Read tool call should match `$CANONICAL_LATEST/SKILL.md`. If it contains `marketplaces/` or any other prefix, that is the stale-path failure mode. Re-read from `$CANONICAL_LATEST/SKILL.md` and restart this contract from the top.
|
||||
|
||||
The same pinned resolver appears later in Step 1 for the engine Bash invocation. That guard is necessary but insufficient — by the time you reach Step 1, you may have already internalized an out-of-date flag list from the stale SKILL.md above it. This STEP 0 runs first so the CONTRACT itself is read from the right file.
|
||||
**Other install paths are fine:** `~/.codex/skills/`, `~/.agents/skills/`, an `npx skills add` install dir, or a repo checkout are all valid load points - the resolver in Step 1 picks them up. Do NOT abort or hop on those paths.
|
||||
|
||||
---
|
||||
|
||||
@@ -88,7 +97,7 @@ You are inside the `/last30days` SKILL. This is a specific research tool with a
|
||||
|
||||
**How v3.0.7 fixes it:** three structural anchors.
|
||||
1. **The MANDATORY first-line badge** (`🌐 last30days v{VERSION} · synced {YYYY-MM-DD}`) at the top of every response is the LAW 2 / LAW 4 enforcement anchor. See "BADGE (MANDATORY, FIRST LINE OF OUTPUT)" in the synthesis section.
|
||||
2. **The pinned SKILL_ROOT resolution** in the engine Bash calls always points to the public plugin cache, never `~/.openclaw/` or other stale copies.
|
||||
2. **The SKILL_DIR substitution** in the engine Bash calls uses the directory of the SKILL.md the model just Read — no resolver list, no precedence walk. Whichever install the harness loaded SKILL.md from is the install whose engine runs. Aligns spec-with-code and works for any harness without enumerating its install path.
|
||||
3. **This preface** tells you plainly: do NOT improvise. Follow SKILL.md top to bottom.
|
||||
|
||||
If you catch yourself about to write a `##` section header in a GENERAL-query body, a custom title line, a `Sources:` bullet list, a `for dir in ...` path-discovery loop, or a bare `python3 scripts/last30days.py "{TOPIC}"` engine call with no pre-flight flags — stop. Those are the exact failure modes the LAWs and this contract exist to prevent. The 10/10 beta validation from 2026-04-18 and the 0/8 public v3.0.6 regression from the same day had THE SAME MODEL and SIMILAR SKILL.md CONTENT; the delta is the three anchors this release restores. Read SKILL.md top to bottom before emitting your first response.
|
||||
@@ -105,7 +114,7 @@ These anchors used to live at line 1094 of this file. Three independent Opus 4.7
|
||||
🌐 last30days v{VERSION} · synced {YYYY-MM-DD}
|
||||
```
|
||||
|
||||
Replace `{VERSION}` with the installed plugin version (`jq -r '.version' "$SKILL_ROOT/../../.codex-plugin/plugin.json" 2>/dev/null || jq -r '.version' "$SKILL_ROOT/.claude-plugin/plugin.json"`) and `{YYYY-MM-DD}` with today's date. No other text on this line. One blank line after, then the synthesis begins.
|
||||
Replace `{VERSION}` with the installed plugin version (`jq -r '.version' "$SKILL_DIR/../../.claude-plugin/plugin.json" 2>/dev/null || awk '/^version:/{gsub(/"/,"",$2); print $2; exit}' "$SKILL_DIR/SKILL.md"`) and `{YYYY-MM-DD}` with today's date. No other text on this line. One blank line after, then the synthesis begins.
|
||||
|
||||
**Why the badge is MANDATORY:** it is the structural anchor for the canonical output shape. Without it the model drifts into blog-post narrative format with `##` section headers and invented titles, violating LAW 2 and LAW 4. The 2026-04-18 public v3.0.6 0/8 regression produced outputs with section headers like "The headline", "Why he is everywhere", "1. gstack dominates", "The 'Homecoming' peak". Direct cause: this anchor was absent. Do NOT skip the badge. Do NOT describe it. Do NOT paraphrase it. Emit it verbatim as line 1.
|
||||
|
||||
@@ -177,13 +186,13 @@ The self-evolving loop is the sticky use case. Every 15 tool calls Hermes pauses
|
||||
Cron-scheduled autonomous briefings are the most-cited concrete workflow. r/TunisiaTech's "Use cases of OpenClaw, Hermes Agent" thread says it plainly: "Currently I have daily cron jobs for news briefing, but I know there's much more I can do."
|
||||
```
|
||||
|
||||
**LAW 7 - YOU ARE THE PLANNER. `--plan` IS MANDATORY ON NAMED-ENTITY TOPICS.** If you are the reasoning model hosting this skill (Claude Code, Codex, Hermes, Gemini, or any agent runtime that invoked `/last30days`), YOU generate the JSON query plan. You do not need an API key, "LLM provider" credentials, or an external planning service - you ARE the LLM. The `--plan` flag exists precisely so a reasoning model generates its own plan upstream and passes it to the engine. The engine's internal planner and deterministic fallback are headless/cron paths only; on any reasoning-model path, bypass them by passing `--plan '$JSON'`.
|
||||
**LAW 7 - YOU ARE THE PLANNER. `--plan` IS MANDATORY ON NAMED-ENTITY TOPICS.** If you are the reasoning model hosting this skill (Claude Code, Codex, Hermes, Gemini, or any agent runtime that invoked `/last30days`), YOU generate the JSON query plan. You do not need an API key, "LLM provider" credentials, or an external planning service - you ARE the LLM. The `--plan` flag exists precisely so a reasoning model generates its own plan upstream and passes it to the engine. The engine's internal planner and deterministic fallback are headless/cron paths only; on any reasoning-model path, bypass them by passing `--plan "$QUERY_PLAN_FILE"` (the path to a tmpfile you wrote via heredoc — see Step 1 for the pattern; never inline `--plan '$JSON'`, apostrophes in search/ranking strings break shell parsing).
|
||||
|
||||
Named-entity topics (capitalized proper nouns, product names, person names, project names, or any topic that would benefit from handle resolution in Step 0.55) REQUIRE `--plan`. Your invocation of `scripts/last30days.py` MUST contain `--plan '$JSON'`. A bare `python3 scripts/last30days.py "$TOPIC" --emit=compact` on a named-entity topic is a LAW 7 violation. Before you invoke Bash, self-check: does my command contain `--plan`? If no, STOP and generate a plan first (see Step 0.75 for the schema).
|
||||
Named-entity topics (capitalized proper nouns, product names, person names, project names, or any topic that would benefit from handle resolution in Step 0.55) REQUIRE `--plan`. Your invocation of `scripts/last30days.py` MUST contain `--plan "$QUERY_PLAN_FILE"` (or any path the engine can read). A bare `python3 scripts/last30days.py "$TOPIC" --emit=compact` on a named-entity topic is a LAW 7 violation. Before you invoke Bash, self-check: does my command contain `--plan`? If no, STOP and generate a plan first (see Step 0.75 for the schema).
|
||||
|
||||
**Observed LAW 7 violation (2026-04-19, Hermes Agent Use Cases Run 1):** the model called the engine bare with no `--plan`, no pre-flight handle resolution. The engine emitted a stderr warning ("No --plan and no LLM provider configured. Using deterministic fallback...") which the model read as a capability constraint ("I don't have a key, I can't do LLM stuff") instead of as what it actually was: a reminder that the reasoning model skipped its own planning step. The misread came from the word "provider" - the engine uses "provider" to mean "the key for the engine's INTERNAL planner," but the model parsed it as "I need a provider to plan at all." You do not. You ARE the provider. Run 2 of the same topic (2026-04-19, framed as "best workflows") with the same model and same cache generated the plan itself via `--plan` and produced clean results - the delta was this step.
|
||||
|
||||
**Self-check before Bash:** re-read your pending `scripts/last30days.py` command. Does it contain `--plan '$JSON'`? If no, and the topic is a named entity, STOP. Return to Step 0.75 and generate the plan. Do not interpret the word "provider" in any engine message as "you need credentials" - you are the provider.
|
||||
**Self-check before Bash:** re-read your pending `scripts/last30days.py` command. Does it contain `--plan "$QUERY_PLAN_FILE"` (or another path the engine can read)? If no, and the topic is a named entity, STOP. Return to Step 0.75 and generate the plan, then write it to a tmpfile per the Step 1 pattern. Do not interpret the word "provider" in any engine message as "you need credentials" - you are the provider.
|
||||
|
||||
**LAW 8 - EVERY CITATION IN THE NARRATIVE IS AN INLINE MARKDOWN LINK `[name](url)`. NEVER A RAW URL STRING. NEVER A PLAIN NAME WHEN A URL IS AVAILABLE.** Applies to every query type. In the "What I learned:" narrative, in KEY PATTERNS, and in the COMPARISON body sections, every cited @handle, r/subreddit, publication, YouTube channel, TikTok creator, Instagram creator, and Polymarket market is wrapped as `[name](url)` at first mention. The URL comes from the raw research dump — every engine item carries a URL; WebSearch supplements carry URLs in their own output. Claude Code renders `[text](url)` as blue CMD-clickable text; the URL is hidden in the rendering, only the link text shows. The stats footer (emoji-tree block) is engine-emitted per LAW 5 and passes through verbatim — do NOT reformat its links yourself.
|
||||
|
||||
@@ -234,7 +243,7 @@ If your Bash call to `last30days.py` does NOT include the FULL pre-flight checkl
|
||||
|
||||
---
|
||||
|
||||
# last30days v3.2.0: Research Any Topic from the Last 30 Days
|
||||
# last30days v3.2.4: Research Any Topic from the Last 30 Days
|
||||
|
||||
> **Permissions overview:** Reads public web/platform data and optionally saves research briefings to `LAST30DAYS_MEMORY_DIR` (defaults to `~/Documents/Last30Days`). X/Twitter search uses optional user-provided tokens (AUTH_TOKEN/CT0 env vars). Bluesky search uses optional app password (BSKY_HANDLE/BSKY_APP_PASSWORD env vars - create at bsky.app/settings/app-passwords). All credential usage and data writes are documented in the [Security & Permissions](#security--permissions) section.
|
||||
|
||||
@@ -318,15 +327,14 @@ Common patterns:
|
||||
|
||||
- Always active: Reddit, Hacker News, Polymarket
|
||||
- If gh CLI is installed (check `which gh`): add GitHub
|
||||
- If digg-pp-cli is installed (check `which digg-pp-cli`): add Digg AI 1000
|
||||
- If digg-pp-cli is installed (check `which digg-pp-cli`): add Digg
|
||||
- If AUTH_TOKEN/CT0 or XAI_API_KEY or FROM_BROWSER is set, or xurl CLI is installed and authenticated: add X
|
||||
- If yt-dlp is installed (check `which yt-dlp`): add YouTube
|
||||
- If SCRAPECREATORS_API_KEY is set and INCLUDE_SOURCES contains tiktok: add TikTok
|
||||
- If SCRAPECREATORS_API_KEY is set and INCLUDE_SOURCES contains instagram: add Instagram
|
||||
- If SCRAPECREATORS_API_KEY is set and INCLUDE_SOURCES contains threads: add Threads
|
||||
- If SCRAPECREATORS_API_KEY is set and INCLUDE_SOURCES contains pinterest: add Pinterest
|
||||
- If SCRAPECREATORS_API_KEY is set: add TikTok, Instagram, Threads (suppress any of these via EXCLUDE_SOURCES)
|
||||
- If SCRAPECREATORS_API_KEY is set and the user explicitly requested pinterest for this query (e.g. via `--search=pinterest`): add Pinterest
|
||||
- If BSKY_HANDLE and BSKY_APP_PASSWORD are set: add Bluesky
|
||||
- If OPENROUTER_API_KEY is set: add Perplexity
|
||||
- If OPENROUTER_API_KEY is set and INCLUDE_SOURCES contains perplexity: add Perplexity
|
||||
- If EXCLUDE_SOURCES is set (comma-separated, case-insensitive): drop any matching source from the list above before displaying
|
||||
|
||||
Then display (use "and more" if 5+ sources, otherwise list all with Oxford comma):
|
||||
|
||||
@@ -583,18 +591,50 @@ When the user asks "X vs Y" (or "X vs Y vs Z"), the engine fans out N full `pipe
|
||||
|
||||
**Invocation:**
|
||||
```bash
|
||||
"${LAST30DAYS_PYTHON}" "${SKILL_ROOT}/scripts/last30days.py" "{TOPIC_A} vs {TOPIC_B} vs {TOPIC_C}" \
|
||||
# SKILL_DIR = absolute path of the directory containing THIS SKILL.md you just Read.
|
||||
# Substitute the actual path below — your harness told you where this file lives via
|
||||
# the Read tool result. Examples:
|
||||
# Read ~/.claude/skills/last30days/SKILL.md → SKILL_DIR=$HOME/.claude/skills/last30days
|
||||
# Read ~/.codex/skills/last30days/SKILL.md → SKILL_DIR=$HOME/.codex/skills/last30days
|
||||
# Read ~/.claude/plugins/cache/last30days-skill/last30days/3.2.4/skills/last30days/SKILL.md
|
||||
# → SKILL_DIR=$HOME/.claude/plugins/cache/last30days-skill/last30days/3.2.4/skills/last30days
|
||||
# scripts/last30days.py is always a direct child of SKILL_DIR (every install layout
|
||||
# packages SKILL.md and scripts/ as siblings).
|
||||
SKILL_DIR="<absolute path of the directory containing the SKILL.md you Read>"
|
||||
|
||||
if [ ! -f "$SKILL_DIR/scripts/last30days.py" ]; then
|
||||
echo "ERROR: scripts/last30days.py not found under SKILL_DIR=$SKILL_DIR" >&2
|
||||
echo "Re-check the directory of the SKILL.md you Read and substitute it as SKILL_DIR above." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Write the per-entity plan to a tmpfile and pass the path to the engine.
|
||||
# The engine's parse_competitors_plan() reads file paths transparently. This
|
||||
# avoids the inline-single-quoted-JSON apostrophe trap (resolved context
|
||||
# strings like "people's choice" or "McDonald's" otherwise close the outer
|
||||
# single-quote and break shell parsing before the engine is even invoked).
|
||||
# Trailing XXXXXX (no .json suffix) so BSD/macOS mktemp works the same as
|
||||
# GNU; BSD only substitutes X's at the end of the template.
|
||||
COMPETITORS_PLAN_FILE=$(mktemp "${TMPDIR:-/tmp}/last30days-competitors.XXXXXX")
|
||||
trap 'rm -f "$COMPETITORS_PLAN_FILE"' EXIT
|
||||
cat > "$COMPETITORS_PLAN_FILE" <<'PLAN_EOF'
|
||||
{
|
||||
"{TOPIC_B}": {"x_handle":"{TOPIC_B_HANDLE}","subreddits":["{TOPIC_B_SUB_1}","{TOPIC_B_SUB_2}"],"github_user":"{TOPIC_B_GH}","context":"{TOPIC_B_CONTEXT}"},
|
||||
"{TOPIC_C}": {"x_handle":"{TOPIC_C_HANDLE}","subreddits":["{TOPIC_C_SUB_1}"],"github_user":"{TOPIC_C_GH}","context":"{TOPIC_C_CONTEXT}"}
|
||||
}
|
||||
PLAN_EOF
|
||||
|
||||
"${LAST30DAYS_PYTHON}" "${SKILL_DIR}/scripts/last30days.py" "{TOPIC_A} vs {TOPIC_B} vs {TOPIC_C}" \
|
||||
--emit=compact \
|
||||
--save-dir="${LAST30DAYS_MEMORY_DIR}" \
|
||||
--save-suffix=v3 \
|
||||
--x-handle={TOPIC_A_HANDLE} \
|
||||
--subreddits={TOPIC_A_SUBS} \
|
||||
--competitors-plan '{
|
||||
"{TOPIC_B}": {"x_handle":"{TOPIC_B_HANDLE}","subreddits":["{TOPIC_B_SUB_1}","{TOPIC_B_SUB_2}"],"github_user":"{TOPIC_B_GH}","context":"{TOPIC_B_CONTEXT}"},
|
||||
"{TOPIC_C}": {"x_handle":"{TOPIC_C_HANDLE}","subreddits":["{TOPIC_C_SUB_1}"],"github_user":"{TOPIC_C_GH}","context":"{TOPIC_C_CONTEXT}"}
|
||||
}'
|
||||
--competitors-plan "$COMPETITORS_PLAN_FILE"
|
||||
```
|
||||
|
||||
**The quoted heredoc marker `'PLAN_EOF'` is load-bearing** — quoting suppresses shell interpolation so apostrophes, `$`, backticks, etc. pass through verbatim. If you ever switch to an unquoted `<<PLAN_EOF`, every variable reference and apostrophe inside the JSON becomes a parse hazard.
|
||||
|
||||
Topic A (the main topic, first in the vs-string) uses outer `--x-handle`, `--x-related`, `--subreddits`, `--github-user`, `--github-repo`, `--tiktok-*`, `--ig-creators` as usual. Topics B and C get their targeting from `--competitors-plan` entries (keyed by entity name, case-insensitive).
|
||||
|
||||
**Step 0.55 for N entities.** The same pre-research protocol that applies to a single-entity topic applies to EACH entity in a vs-run. For N=3, that means 3 WebSearches for X handles, 3 for subreddits, 3 for GitHub, 3 for news context — or equivalent batched queries. A `## Resolved Entities` block with dashes for any entity means you skipped Step 0.55 for that one. Re-run with a corrected plan.
|
||||
@@ -826,7 +866,7 @@ Only show lines for platforms where something was resolved. Skip empty lines. On
|
||||
- For how_to: prioritize YouTube (tutorials) and Reddit (guides)
|
||||
- Primary subquery weight = 1.0, secondary = 0.6-0.8, peripheral = 0.3-0.5
|
||||
|
||||
**Available sources (include ALL in primary subquery):** reddit, x, youtube, tiktok, instagram, hackernews, polymarket. Optional: bluesky, truthsocial, threads, pinterest, grounding (web search - only if user has Brave/Exa/Serper key), digg (Digg AI 1000 clusters - only if `digg-pp-cli` is on PATH)
|
||||
**Available sources (include ALL in primary subquery):** reddit, x, youtube, tiktok, instagram, hackernews, polymarket. Optional: bluesky, truthsocial, threads, pinterest, grounding (web search - only if user has Brave/Exa/Serper key), digg (Digg clusters - only if `digg-pp-cli` is on PATH)
|
||||
|
||||
**Intent → freshness_mode mapping:**
|
||||
- breaking_news, prediction → `strict_recent`
|
||||
@@ -869,44 +909,45 @@ Store your plan as `QUERY_PLAN_JSON` - you'll pass it to the script in the next
|
||||
**IMPORTANT: Include `--x-handle={RESOLVED_HANDLE}` in the command. For comparison mode: Pass `--x-handle={TOPIC_A_HANDLE}` to the first pass, `--x-handle={TOPIC_B_HANDLE}` to the second pass, and both to the head-to-head pass. Also include `--subreddits={RESOLVED_SUBREDDITS}`, `--tiktok-hashtags={RESOLVED_HASHTAGS}`, `--tiktok-creators={RESOLVED_TIKTOK_CREATORS}`, and `--ig-creators={RESOLVED_IG_CREATORS}` from Step 0.55. Omit any flag where the value was not resolved (empty).**
|
||||
|
||||
```bash
|
||||
# PIN SKILL_ROOT to an installed plugin cache first (highest-version dir wins on upgrade).
|
||||
# Prefer Codex's skill package path when installed as a Codex plugin. Keep the Claude
|
||||
# plugin-root fallback for other hosts, then fall back to a repo checkout.
|
||||
SKILL_ROOT="$(ls -d "$HOME/.codex/plugins/cache/"*/last30days/*/skills/last30days/ 2>/dev/null | sort -V | tail -1)"
|
||||
SKILL_ROOT="${SKILL_ROOT%/}"
|
||||
# SKILL_DIR = absolute path of the directory containing THIS SKILL.md you just Read.
|
||||
# Substitute the actual path below — your harness told you where this file lives via
|
||||
# the Read tool result. Examples:
|
||||
# Read ~/.claude/skills/last30days/SKILL.md → SKILL_DIR=$HOME/.claude/skills/last30days
|
||||
# Read ~/.codex/skills/last30days/SKILL.md → SKILL_DIR=$HOME/.codex/skills/last30days
|
||||
# Read ~/.claude/plugins/cache/last30days-skill/last30days/3.2.4/skills/last30days/SKILL.md
|
||||
# → SKILL_DIR=$HOME/.claude/plugins/cache/last30days-skill/last30days/3.2.4/skills/last30days
|
||||
# scripts/last30days.py is always a direct child of SKILL_DIR (every install layout
|
||||
# packages SKILL.md and scripts/ as siblings).
|
||||
SKILL_DIR="<absolute path of the directory containing the SKILL.md you Read>"
|
||||
|
||||
# Fallback for Claude plugin cache.
|
||||
if [ -z "$SKILL_ROOT" ] || [ ! -f "$SKILL_ROOT/scripts/last30days.py" ]; then
|
||||
CLAUDE_PLUGIN_ROOT="$(ls -d "$HOME/.claude/plugins/cache/last30days-skill/last30days/"*/ 2>/dev/null | sort -V | tail -1)"
|
||||
CLAUDE_PLUGIN_ROOT="${CLAUDE_PLUGIN_ROOT%/}"
|
||||
if [ -n "$CLAUDE_PLUGIN_ROOT" ]; then
|
||||
if [ -f "$CLAUDE_PLUGIN_ROOT/skills/last30days/scripts/last30days.py" ]; then
|
||||
SKILL_ROOT="$CLAUDE_PLUGIN_ROOT/skills/last30days"
|
||||
elif [ -f "$CLAUDE_PLUGIN_ROOT/scripts/last30days.py" ]; then
|
||||
SKILL_ROOT="$CLAUDE_PLUGIN_ROOT"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# Fallback for repo checkout / Gemini / local development hosts where the plugin cache does not exist.
|
||||
if [ -z "$SKILL_ROOT" ] || [ ! -f "$SKILL_ROOT/scripts/last30days.py" ]; then
|
||||
for dir in "." "./skills/last30days" "${CLAUDE_PLUGIN_ROOT:-}" "${GEMINI_EXTENSION_DIR:-}"; do
|
||||
[ -n "$dir" ] && [ -f "$dir/scripts/last30days.py" ] && SKILL_ROOT="$dir" && break
|
||||
done
|
||||
fi
|
||||
|
||||
if [ -z "${SKILL_ROOT:-}" ] || [ ! -f "$SKILL_ROOT/scripts/last30days.py" ]; then
|
||||
echo "ERROR: Could not find scripts/last30days.py in Codex/Claude plugin cache or repo checkout" >&2
|
||||
echo "Expected Codex: $HOME/.codex/plugins/cache/{MARKETPLACE}/last30days/{VERSION}/skills/last30days/scripts/last30days.py" >&2
|
||||
echo "Expected Claude: $HOME/.claude/plugins/cache/last30days-skill/last30days/{VERSION}/skills/last30days/scripts/last30days.py" >&2
|
||||
if [ ! -f "$SKILL_DIR/scripts/last30days.py" ]; then
|
||||
echo "ERROR: scripts/last30days.py not found under SKILL_DIR=$SKILL_DIR" >&2
|
||||
echo "Re-check the directory of the SKILL.md you Read and substitute it as SKILL_DIR above." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
"${LAST30DAYS_PYTHON}" "${SKILL_ROOT}/scripts/last30days.py" $ARGUMENTS --emit=compact --save-dir="${LAST30DAYS_MEMORY_DIR}" --save-suffix=v3
|
||||
"${LAST30DAYS_PYTHON}" "${SKILL_DIR}/scripts/last30days.py" $ARGUMENTS --emit=compact --save-dir="${LAST30DAYS_MEMORY_DIR}" --save-suffix=v3
|
||||
```
|
||||
|
||||
**If you ran Steps 0.55 and 0.75 (agent planning), add these flags:**
|
||||
- `--plan 'QUERY_PLAN_JSON'` (replace with actual JSON from Step 0.75)
|
||||
**If you ran Steps 0.55 and 0.75 (agent planning), pass the plan via a tmpfile and add the targeting flags:**
|
||||
|
||||
```bash
|
||||
# Write QUERY_PLAN_JSON to a tmpfile before the engine invocation above.
|
||||
# parse_plan() reads file paths transparently; this avoids inline-JSON
|
||||
# shell-quoting hazards (apostrophes in search_query / ranking_query
|
||||
# strings break single-quoted command-line JSON). Trailing XXXXXX (no
|
||||
# .json suffix) for BSD/macOS portability — BSD mktemp only substitutes
|
||||
# X's at the end of the template.
|
||||
QUERY_PLAN_FILE=$(mktemp "${TMPDIR:-/tmp}/last30days-plan.XXXXXX")
|
||||
trap 'rm -f "$QUERY_PLAN_FILE"' EXIT
|
||||
cat > "$QUERY_PLAN_FILE" <<'PLAN_EOF'
|
||||
{QUERY_PLAN_JSON_FROM_STEP_0.75}
|
||||
PLAN_EOF
|
||||
```
|
||||
|
||||
Then add to the engine command:
|
||||
|
||||
- `--plan "$QUERY_PLAN_FILE"` (path to the file you just wrote)
|
||||
- `--x-handle={RESOLVED_HANDLE}` (from Step 0.5)
|
||||
- `--subreddits={RESOLVED_SUBREDDITS}` (from Step 0.55)
|
||||
- `--tiktok-hashtags={RESOLVED_HASHTAGS}` (from Step 0.55)
|
||||
@@ -1647,7 +1688,7 @@ Want another prompt? Just tell me what you're creating next.
|
||||
- Sends search queries to Algolia HN Search API (`hn.algolia.com`) for Hacker News story and comment discovery (free, no auth)
|
||||
- Sends search queries to Polymarket Gamma API (`gamma-api.polymarket.com`) for prediction market discovery (free, no auth)
|
||||
- Runs `yt-dlp` locally for YouTube search and transcript extraction (no API key, public data)
|
||||
- Sends search queries to ScrapeCreators API (`api.scrapecreators.com`) for TikTok and Instagram search, transcript/caption extraction (PAYG after 10,000 free API calls)
|
||||
- Sends search queries to ScrapeCreators API (`api.scrapecreators.com`) for TikTok and Instagram search, transcript/caption extraction (PAYG after 100 free credits)
|
||||
- Optionally sends search queries to Brave Search API, Parallel AI API, or OpenRouter API for web search
|
||||
- Fetches public Reddit thread data from `reddit.com` for engagement metrics
|
||||
- Stores research findings in local SQLite database (watchlist mode only)
|
||||
@@ -1660,7 +1701,7 @@ Want another prompt? Just tell me what you're creating next.
|
||||
- Does not log, cache, or write API keys to output files
|
||||
- Does not send data to any endpoint not listed above
|
||||
- Hacker News and Polymarket sources are always available (no API key, no binary dependency)
|
||||
- TikTok and Instagram sources require SCRAPECREATORS_API_KEY (10,000 free API calls, then PAYG). Reddit uses ScrapeCreators only as a backup when public Reddit is unavailable.
|
||||
- TikTok and Instagram sources require SCRAPECREATORS_API_KEY (100 free credits one-time, then PAYG). Reddit uses ScrapeCreators only as a backup when public Reddit is unavailable.
|
||||
- Can be invoked autonomously by agents via the Skill tool (runs inline, not forked); pass `--agent` for non-interactive report output
|
||||
|
||||
**Bundled scripts:** `scripts/last30days.py` (main research engine), `scripts/lib/` (search, enrichment, rendering modules), `scripts/lib/vendor/bird-search/` (vendored X search client, MIT licensed)
|
||||
|
||||
@@ -20,6 +20,7 @@ sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
from lib import env as envlib
|
||||
from lib import schema
|
||||
from lib.providers import GEMINI_FLASH_LITE
|
||||
|
||||
|
||||
SKILL_ROOT = Path(__file__).resolve().parents[1]
|
||||
@@ -43,7 +44,7 @@ def _load_default_topics() -> list[tuple[str, str]]:
|
||||
|
||||
DEFAULT_TOPICS = _load_default_topics()
|
||||
DEFAULT_SEARCH = ""
|
||||
DEFAULT_JUDGE_MODEL = "gemini-3.1-flash-lite-preview"
|
||||
DEFAULT_JUDGE_MODEL = GEMINI_FLASH_LITE
|
||||
GEMINI_API_URL = "https://generativelanguage.googleapis.com/v1beta/models/{model}:generateContent?key={api_key}"
|
||||
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
# ruff: noqa: E402
|
||||
"""last30days v3.0.0 CLI."""
|
||||
"""last30days CLI."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -97,11 +97,13 @@ def save_output(
|
||||
save_dir: str,
|
||||
suffix: str = "",
|
||||
synthesis_md: str | None = None,
|
||||
topic_override: str | None = None,
|
||||
rendered_content: str | None = None,
|
||||
) -> Path:
|
||||
from datetime import datetime
|
||||
path = Path(save_dir).expanduser().resolve()
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
slug = slugify(report.topic)
|
||||
slug = slugify(topic_override or report.topic)
|
||||
extension = "json" if emit == "json" else "html" if emit == "html" else "md"
|
||||
raw_label = "raw-html" if emit == "html" else "raw"
|
||||
suffix_part = f"-{suffix}" if suffix else ""
|
||||
@@ -110,7 +112,9 @@ def save_output(
|
||||
out_path = path / f"{slug}-{raw_label}{suffix_part}-{datetime.now().strftime('%Y-%m-%d')}.{extension}"
|
||||
# Markdown saves keep the complete debug artifact. JSON and HTML preserve
|
||||
# their requested wire format so file extensions match their content.
|
||||
if emit in {"json", "html"}:
|
||||
if rendered_content is not None:
|
||||
content = rendered_content
|
||||
elif emit in {"json", "html"}:
|
||||
content = emit_output(report, emit, synthesis_md=synthesis_md)
|
||||
else:
|
||||
content = render.render_full(report)
|
||||
@@ -171,6 +175,10 @@ def emit_comparison_output(
|
||||
raise SystemExit(f"Unsupported emit mode: {emit}")
|
||||
|
||||
|
||||
def comparison_topic(entity_reports: list[tuple[str, schema.Report]]) -> str:
|
||||
return " vs ".join(label for label, _ in entity_reports)
|
||||
|
||||
|
||||
def compute_save_path_display(save_dir: str, topic: str, suffix: str, emit: str) -> str:
|
||||
"""Compute the user-friendly save path string that will be shown in the footer.
|
||||
|
||||
@@ -533,6 +541,13 @@ def main() -> int:
|
||||
|
||||
config = env.get_config()
|
||||
|
||||
# Surface SSH-routing config as an env var so library modules (e.g.
|
||||
# youtube_yt) can read it without taking a config dependency. This
|
||||
# routes yt-dlp through `ssh <host>` to bypass YouTube's bot-wall on
|
||||
# datacenter IPs (see lib/youtube_yt.py for details).
|
||||
if config.get("LAST30DAYS_YOUTUBE_SSH_HOST") and "LAST30DAYS_YOUTUBE_SSH_HOST" not in os.environ:
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = config["LAST30DAYS_YOUTUBE_SSH_HOST"]
|
||||
|
||||
# Handle setup subcommand
|
||||
topic = " ".join(args.topic).strip()
|
||||
if topic.lower() == "setup":
|
||||
@@ -873,10 +888,15 @@ def main() -> int:
|
||||
pass
|
||||
|
||||
fun_level = config.get("FUN_LEVEL", "medium").lower()
|
||||
# Comparison HTML is the one case where the saved file's title and content
|
||||
# have to be overridden away from the leading entity's report. Compute the
|
||||
# gate once so the footer-display and save-output paths can't disagree.
|
||||
is_comparison_html = bool(entity_reports) and args.emit == "html"
|
||||
footer_save_path = None
|
||||
if args.save_dir:
|
||||
save_topic_for_display = comparison_topic(entity_reports) if is_comparison_html else report.topic
|
||||
footer_save_path = compute_save_path_display(
|
||||
args.save_dir, report.topic, args.save_suffix or "", args.emit
|
||||
args.save_dir, save_topic_for_display, args.save_suffix or "", args.emit
|
||||
)
|
||||
|
||||
# Signal to render_compact whether pre-research flags were supplied.
|
||||
@@ -917,6 +937,8 @@ def main() -> int:
|
||||
args.save_dir,
|
||||
suffix=args.save_suffix or "",
|
||||
synthesis_md=synthesis_md,
|
||||
topic_override=comparison_topic(entity_reports) if is_comparison_html else None,
|
||||
rendered_content=rendered if is_comparison_html else None,
|
||||
)
|
||||
sys.stderr.write(f"[last30days] Saved output to {save_path}\n")
|
||||
# Competitor / vs-mode: also save a per-entity raw file for each peer.
|
||||
|
||||
@@ -9,6 +9,7 @@ import json
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from . import http, log, subproc
|
||||
@@ -17,6 +18,11 @@ from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from .relevance import token_overlap_relevance as _compute_relevance
|
||||
|
||||
# How many times to retry the bird-search subprocess when stdout is non-JSON
|
||||
# (typically an HTML anti-bot interstitial from Twitter's edge).
|
||||
MAX_JSON_DECODE_RETRIES = 2
|
||||
JSON_DECODE_RETRY_DELAY = 5.0 # seconds between retry attempts
|
||||
|
||||
|
||||
def _first_of(*values):
|
||||
"""Return first value that is not None."""
|
||||
@@ -148,16 +154,14 @@ def get_bird_status() -> Dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def _run_bird_search(query: str, count: int, timeout: int) -> Dict[str, Any]:
|
||||
"""Run a search using the vendored bird-search.mjs module.
|
||||
def _invoke_bird_subprocess(query: str, count: int, timeout: int):
|
||||
"""Invoke the vendored bird-search.mjs subprocess once.
|
||||
|
||||
Args:
|
||||
query: Full search query string (including since: filter)
|
||||
count: Number of results to request
|
||||
timeout: Timeout in seconds
|
||||
|
||||
Returns:
|
||||
Raw Bird JSON response or error dict.
|
||||
Returns (result, error_dict). If error_dict is non-None, treat it as the
|
||||
final result and do not retry — those errors are terminal (timeout,
|
||||
spawn failure). If error_dict is None, the subprocess ran to completion
|
||||
and `result` is the SubprocResult; the caller decides whether to retry
|
||||
based on the result.stdout content.
|
||||
"""
|
||||
cmd = [
|
||||
"node", str(_BIRD_SEARCH_MJS),
|
||||
@@ -184,9 +188,9 @@ def _run_bird_search(query: str, count: int, timeout: int) -> Dict[str, Any]:
|
||||
on_pid=_register,
|
||||
)
|
||||
except subproc.SubprocTimeout:
|
||||
return {"error": f"Search timed out after {timeout}s", "items": []}
|
||||
return None, {"error": f"Search timed out after {timeout}s", "items": []}
|
||||
except Exception as e:
|
||||
return {"error": str(e), "items": []}
|
||||
return None, {"error": str(e), "items": []}
|
||||
finally:
|
||||
if pid_holder:
|
||||
try:
|
||||
@@ -195,22 +199,80 @@ def _run_bird_search(query: str, count: int, timeout: int) -> Dict[str, Any]:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if result.returncode != 0:
|
||||
error = result.stderr.strip() or "Bird search failed"
|
||||
return {"error": error, "items": []}
|
||||
return result, None
|
||||
|
||||
output = result.stdout.strip()
|
||||
if not output:
|
||||
return {"items": []}
|
||||
|
||||
try:
|
||||
parsed = json.loads(output)
|
||||
except json.JSONDecodeError as e:
|
||||
return {"error": f"Invalid JSON response: {e}", "items": []}
|
||||
def _run_bird_search(query: str, count: int, timeout: int) -> Dict[str, Any]:
|
||||
"""Run a search using the vendored bird-search.mjs module.
|
||||
|
||||
if isinstance(parsed, list):
|
||||
return {"items": parsed}
|
||||
return parsed
|
||||
Retries the subprocess on JSON-decode failure (typically a Twitter
|
||||
anti-bot HTML interstitial in stdout) up to MAX_JSON_DECODE_RETRIES
|
||||
times with JSON_DECODE_RETRY_DELAY seconds between attempts. Terminal
|
||||
errors (subprocess timeout, non-zero return code) are returned
|
||||
immediately without retry.
|
||||
|
||||
Args:
|
||||
query: Full search query string (including since: filter)
|
||||
count: Number of results to request
|
||||
timeout: Timeout in seconds (per attempt)
|
||||
|
||||
Returns:
|
||||
Raw Bird JSON response or error dict.
|
||||
"""
|
||||
last_decode_error: Optional[str] = None
|
||||
|
||||
for attempt in range(MAX_JSON_DECODE_RETRIES):
|
||||
result, terminal_error = _invoke_bird_subprocess(query, count, timeout)
|
||||
if terminal_error is not None:
|
||||
return terminal_error
|
||||
|
||||
if result.returncode != 0:
|
||||
error = result.stderr.strip() or "Bird search failed"
|
||||
return {"error": error, "items": []}
|
||||
|
||||
output = result.stdout.strip()
|
||||
if not output:
|
||||
return {"items": []}
|
||||
|
||||
try:
|
||||
parsed = json.loads(output)
|
||||
except json.JSONDecodeError as e:
|
||||
# Twitter's edge sometimes serves an HTML anti-bot interstitial
|
||||
# in place of JSON. Tag the failure shape so it's distinguishable
|
||||
# from "no results" in logs, then retry the subprocess.
|
||||
looks_html = output.lstrip().lower().startswith(("<!doctype", "<html", "<"))
|
||||
attempt_num = attempt + 1
|
||||
log_msg = (
|
||||
f"Bird search returned non-JSON stdout "
|
||||
f"(looks_html={looks_html}, attempt {attempt_num}/{MAX_JSON_DECODE_RETRIES}, "
|
||||
f"first 80 chars: {output[:80]!r})"
|
||||
)
|
||||
last_decode_error = str(e)
|
||||
if attempt_num < MAX_JSON_DECODE_RETRIES:
|
||||
log.source_log(
|
||||
"X/bird",
|
||||
f"{log_msg}; retrying in {JSON_DECODE_RETRY_DELAY:.0f}s",
|
||||
)
|
||||
time.sleep(JSON_DECODE_RETRY_DELAY)
|
||||
continue
|
||||
log.source_log("X/bird", log_msg)
|
||||
return {
|
||||
"error": (
|
||||
f"Invalid JSON response after {MAX_JSON_DECODE_RETRIES} attempts "
|
||||
f"(likely Twitter anti-bot interstitial): {e}"
|
||||
),
|
||||
"items": [],
|
||||
}
|
||||
|
||||
if isinstance(parsed, list):
|
||||
return {"items": parsed}
|
||||
return parsed
|
||||
|
||||
# Defensive fallthrough — loop should always return above.
|
||||
return {
|
||||
"error": f"Bird search exhausted retries: {last_decode_error}",
|
||||
"items": [],
|
||||
}
|
||||
|
||||
|
||||
def search_x(
|
||||
|
||||
@@ -44,8 +44,9 @@ ENRICH_CONFIG = {
|
||||
"deep": 5,
|
||||
}
|
||||
|
||||
# X posts pulled per enriched cluster.
|
||||
POSTS_PER_CLUSTER = 3
|
||||
# X posts pulled per enriched cluster. Matches the 5-comment cap used by
|
||||
# Reddit/HN/YouTube/TikTok/GitHub enrichment.
|
||||
POSTS_PER_CLUSTER = 5
|
||||
|
||||
SEARCH_TIMEOUT = 30
|
||||
POSTS_TIMEOUT = 15
|
||||
@@ -285,9 +286,9 @@ def parse_digg_response(
|
||||
"posts": [],
|
||||
"relevance": round(relevance, 2),
|
||||
"why_relevant": (
|
||||
f"Digg AI 1000 cluster (rank {rank}, {post_count} posts, {unique_authors} authors)"
|
||||
f"Digg cluster (rank {rank}, {post_count} posts, {unique_authors} authors)"
|
||||
if rank is not None
|
||||
else f"Digg AI 1000 cluster ({post_count} posts, {unique_authors} authors)"
|
||||
else f"Digg cluster ({post_count} posts, {unique_authors} authors)"
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
@@ -29,6 +29,23 @@ else:
|
||||
|
||||
CODEX_AUTH_FILE = Path(os.environ.get("CODEX_AUTH_FILE", str(Path.home() / ".codex" / "auth.json")))
|
||||
|
||||
# macOS Keychain integration: items stored with this service prefix are picked
|
||||
# up automatically on Darwin as the lowest-priority credential source.
|
||||
# Example: `security add-generic-password -a "$USER" -s last30days-XAI_API_KEY -w "xai-..."`.
|
||||
KEYCHAIN_SERVICE_PREFIX = "last30days-"
|
||||
|
||||
# Single source of truth for which credentials the Keychain loader looks up.
|
||||
# The setup-keychain.sh helper mirrors this list and is held in sync via
|
||||
# tests/test_env_keychain.py::test_keychain_keys_match_setup_script.
|
||||
KEYCHAIN_KEYS = (
|
||||
"OPENAI_API_KEY", "XAI_API_KEY", "GOOGLE_API_KEY", "GEMINI_API_KEY",
|
||||
"GOOGLE_GENAI_API_KEY", "SCRAPECREATORS_API_KEY", "APIFY_API_TOKEN",
|
||||
"AUTH_TOKEN", "CT0", "BSKY_HANDLE", "BSKY_APP_PASSWORD",
|
||||
"TRUTHSOCIAL_TOKEN", "BRAVE_API_KEY", "EXA_API_KEY", "SERPER_API_KEY",
|
||||
"OPENROUTER_API_KEY", "PARALLEL_API_KEY", "XQUIK_API_KEY",
|
||||
"XIAOHONGSHU_API_BASE",
|
||||
)
|
||||
|
||||
AuthSource = Literal["api_key", "codex", "none"]
|
||||
AuthStatus = Literal["ok", "missing", "expired", "missing_account_id"]
|
||||
|
||||
@@ -53,6 +70,10 @@ class OpenAIAuth:
|
||||
|
||||
def _check_file_permissions(path: Path) -> None:
|
||||
"""Warn to stderr if a secrets file has overly permissive permissions."""
|
||||
if os.name == "nt":
|
||||
# Windows reports synthesized POSIX mode bits that do not reflect NTFS ACLs.
|
||||
return
|
||||
|
||||
try:
|
||||
mode = path.stat().st_mode
|
||||
# Check if group or other can read (bits 0o044)
|
||||
@@ -91,6 +112,46 @@ def load_env_file(path: Path) -> dict[str, str]:
|
||||
return env
|
||||
|
||||
|
||||
def _load_keychain(keys: list[str]) -> dict[str, str]:
|
||||
"""Load credentials from macOS Keychain (no-op on other platforms).
|
||||
|
||||
Each key is looked up as a generic password with service name
|
||||
``f"{KEYCHAIN_SERVICE_PREFIX}{key}"`` for the current user. Missing items
|
||||
and lookup failures are silent — Keychain is the lowest-priority source
|
||||
and is meant to be additive over `.env` files and process environment.
|
||||
"""
|
||||
import platform
|
||||
if platform.system() != "Darwin":
|
||||
return {}
|
||||
|
||||
import shutil
|
||||
security = shutil.which("security")
|
||||
if not security:
|
||||
return {}
|
||||
|
||||
import subprocess
|
||||
import pwd
|
||||
# USER can be unset under sudo, in Docker without --env USER, or in some CI
|
||||
# runners; fall back to the OS user record so lookups still match items
|
||||
# stored by setup-keychain.sh (which uses $USER).
|
||||
user = os.environ.get("USER") or pwd.getpwuid(os.getuid()).pw_name
|
||||
env: dict[str, str] = {}
|
||||
for key in keys:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[security, "find-generic-password",
|
||||
"-a", user,
|
||||
"-s", f"{KEYCHAIN_SERVICE_PREFIX}{key}",
|
||||
"-w"],
|
||||
capture_output=True, text=True, timeout=5,
|
||||
)
|
||||
except (subprocess.TimeoutExpired, OSError):
|
||||
continue
|
||||
if result.returncode == 0 and result.stdout.strip():
|
||||
env[key] = result.stdout.strip()
|
||||
return env
|
||||
|
||||
|
||||
def _decode_jwt_payload(token: str) -> dict[str, Any] | None:
|
||||
"""Decode JWT payload without verification."""
|
||||
try:
|
||||
@@ -214,6 +275,7 @@ def get_config() -> dict[str, Any]:
|
||||
1. Environment variables (os.environ)
|
||||
2. .claude/last30days.env (per-project config)
|
||||
3. ~/.config/last30days/.env (global config)
|
||||
4. macOS Keychain items prefixed ``last30days-`` (Darwin only)
|
||||
"""
|
||||
# Load from global config file
|
||||
file_env = load_env_file(CONFIG_FILE) if CONFIG_FILE else {}
|
||||
@@ -222,9 +284,14 @@ def get_config() -> dict[str, Any]:
|
||||
project_env_path = _find_project_env()
|
||||
project_env = load_env_file(project_env_path) if project_env_path else {}
|
||||
|
||||
# Merge: project overrides global
|
||||
# Merge file sources: project > global
|
||||
merged_env = {**file_env, **project_env}
|
||||
|
||||
# Keychain is the lowest-priority source (Darwin only; no-op elsewhere).
|
||||
# Loaded before openai_auth so OPENAI_API_KEY can come from Keychain too.
|
||||
keychain_env = _load_keychain(list(KEYCHAIN_KEYS))
|
||||
merged_env = {**keychain_env, **merged_env}
|
||||
|
||||
openai_auth = get_openai_auth(merged_env)
|
||||
|
||||
# Build config: Codex/OpenAI auth + process.env > project .env > global .env
|
||||
@@ -265,16 +332,21 @@ def get_config() -> dict[str, Any]:
|
||||
('FROM_BROWSER', None),
|
||||
('SETUP_COMPLETE', None),
|
||||
('INCLUDE_SOURCES', ''),
|
||||
('EXCLUDE_SOURCES', ''),
|
||||
('LAST30DAYS_YOUTUBE_SSH_HOST', None),
|
||||
]
|
||||
|
||||
for key, default in keys:
|
||||
config[key] = os.environ.get(key) or merged_env.get(key, default)
|
||||
|
||||
# Track which config source was used
|
||||
# Track which config source was used (highest-priority file source wins
|
||||
# the label; keychain is only reported when nothing else is configured).
|
||||
if project_env_path:
|
||||
config['_CONFIG_SOURCE'] = f'project:{project_env_path}'
|
||||
elif CONFIG_FILE and CONFIG_FILE.exists():
|
||||
config['_CONFIG_SOURCE'] = f'global:{CONFIG_FILE}'
|
||||
elif keychain_env:
|
||||
config['_CONFIG_SOURCE'] = 'keychain'
|
||||
else:
|
||||
config['_CONFIG_SOURCE'] = 'env_only'
|
||||
|
||||
@@ -517,12 +589,12 @@ def _parse_include_sources(config: dict[str, Any]) -> set[str]:
|
||||
def is_threads_available(config: dict[str, Any]) -> bool:
|
||||
"""Check if Threads source is available.
|
||||
|
||||
Requires SCRAPECREATORS_API_KEY AND 'threads' in INCLUDE_SOURCES.
|
||||
Threads is an opt-in source - it is not activated by default.
|
||||
Returns True when SCRAPECREATORS_API_KEY is set. Threads runs alongside
|
||||
TikTok and Instagram as part of the SC family — same key, same per-call
|
||||
cost shape, so the same default-on rule applies. Suppress via
|
||||
EXCLUDE_SOURCES=threads.
|
||||
"""
|
||||
if not config.get('SCRAPECREATORS_API_KEY'):
|
||||
return False
|
||||
return 'threads' in _parse_include_sources(config)
|
||||
return bool(config.get('SCRAPECREATORS_API_KEY'))
|
||||
|
||||
|
||||
def is_instagram_available(config: dict[str, Any]) -> bool:
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import urllib.parse
|
||||
from datetime import datetime
|
||||
from urllib.parse import urlparse
|
||||
@@ -205,29 +206,90 @@ def web_search(
|
||||
backend = "parallel"
|
||||
else:
|
||||
return [], {}
|
||||
items: list[dict] = []
|
||||
artifact: dict = {}
|
||||
if backend == "brave":
|
||||
key = config.get("BRAVE_API_KEY")
|
||||
if not key:
|
||||
raise RuntimeError("BRAVE_API_KEY is required when web_backend='brave'")
|
||||
return brave_search(query, date_range, key)
|
||||
if backend == "exa":
|
||||
items, artifact = brave_search(query, date_range, key)
|
||||
elif backend == "exa":
|
||||
key = config.get("EXA_API_KEY")
|
||||
if not key:
|
||||
raise RuntimeError("EXA_API_KEY is required when web_backend='exa'")
|
||||
return exa_search(query, date_range, key)
|
||||
if backend == "serper":
|
||||
items, artifact = exa_search(query, date_range, key)
|
||||
elif backend == "serper":
|
||||
key = config.get("SERPER_API_KEY")
|
||||
if not key:
|
||||
raise RuntimeError("SERPER_API_KEY is required when web_backend='serper'")
|
||||
return serper_search(query, date_range, key)
|
||||
if backend == "parallel":
|
||||
items, artifact = serper_search(query, date_range, key)
|
||||
elif backend == "parallel":
|
||||
key = config.get("PARALLEL_API_KEY")
|
||||
if not key:
|
||||
raise RuntimeError("PARALLEL_API_KEY is required when web_backend='parallel'")
|
||||
return parallel_search(query, date_range, key)
|
||||
if backend != "none":
|
||||
items, artifact = parallel_search(query, date_range, key)
|
||||
elif backend != "none":
|
||||
raise ValueError(f"Unsupported web backend: {backend!r}")
|
||||
return [], {}
|
||||
else:
|
||||
return [], {}
|
||||
if items and not _reddit_excluded(config):
|
||||
items = _enrich_reddit_items(items)
|
||||
return items, artifact
|
||||
|
||||
|
||||
def _reddit_excluded(config: dict) -> bool:
|
||||
"""Return True when EXCLUDE_SOURCES contains 'reddit'.
|
||||
|
||||
Respects the same suppression knob the pipeline uses for source gating,
|
||||
so a user who set EXCLUDE_SOURCES=reddit doesn't get Reddit content
|
||||
smuggled back in via web-search URLs.
|
||||
"""
|
||||
raw = (config.get("EXCLUDE_SOURCES") or "").split(",")
|
||||
return any(s.strip().lower() == "reddit" for s in raw)
|
||||
|
||||
|
||||
def _enrich_reddit_items(items: list[dict]) -> list[dict]:
|
||||
"""Enrich web search results that are Reddit URLs with thread body and comments.
|
||||
|
||||
Claude Code's WebFetch blocks reddit.com, so the model can't retrieve
|
||||
Reddit content from web search results. This fetches it via the public
|
||||
JSON API (reddit.com/.../.json) which bypasses that restriction.
|
||||
|
||||
Callers should gate this with EXCLUDE_SOURCES=reddit handling (see
|
||||
`_reddit_excluded`) so a user who explicitly excluded Reddit doesn't
|
||||
get Reddit content via web-search URLs.
|
||||
"""
|
||||
from . import reddit_enrich
|
||||
from .reddit_enrich import RedditRateLimitError
|
||||
|
||||
for item in items:
|
||||
url = item.get("url", "")
|
||||
if "reddit.com" not in url or "/comments/" not in url:
|
||||
continue
|
||||
try:
|
||||
thread_data = reddit_enrich.fetch_thread_data(url, timeout=8)
|
||||
if not thread_data:
|
||||
continue
|
||||
parsed = reddit_enrich.parse_thread_data(thread_data)
|
||||
# selftext lives under parsed["submission"], not at the top level
|
||||
selftext = (parsed.get("submission") or {}).get("selftext", "")
|
||||
if selftext:
|
||||
item["snippet"] = selftext[:2000]
|
||||
comments = parsed.get("comments", [])
|
||||
top = reddit_enrich.get_top_comments(comments)
|
||||
if top:
|
||||
item["top_comments"] = [
|
||||
{"score": c.get("score", 0), "excerpt": (c.get("body") or "")[:200]}
|
||||
for c in top[:5]
|
||||
]
|
||||
item["enriched_via"] = "reddit_json_api"
|
||||
except RedditRateLimitError as exc:
|
||||
# Stop iterating to avoid flooding more 429s
|
||||
sys.stderr.write(f"[Web] Reddit rate-limited, halting enrichment: {exc}\n")
|
||||
break
|
||||
except Exception as exc:
|
||||
sys.stderr.write(f"[Web] Reddit enrichment failed for {url}: {exc}\n")
|
||||
return items
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -88,17 +88,26 @@ def search_hackernews(
|
||||
|
||||
# Use extracted core subject instead of raw topic for cleaner Algolia matching
|
||||
core = extract_core_subject(topic)
|
||||
_log(f"Searching for '{core}' (raw: '{topic}', since {from_date}, count={count})")
|
||||
# Hyphens and commas tokenize awkwardly in Algolia; flatten them so themed
|
||||
# queries like "ts-bun-node" or "claude, personal agents" become plain words.
|
||||
core_flat = _flatten_query_for_algolia(core)
|
||||
_log(f"Searching for '{core_flat}' (raw: '{topic}', since {from_date}, count={count})")
|
||||
|
||||
# Use relevance-sorted search with minimum engagement filter.
|
||||
# NOTE: restrictSearchableAttributes=title omitted intentionally — it would
|
||||
# miss Ask HN/Show HN threads where the topic appears in the body.
|
||||
params = {
|
||||
"query": core,
|
||||
"query": core_flat,
|
||||
"tags": "story",
|
||||
"numericFilters": f"created_at_i>{from_ts},created_at_i<{to_ts},points>2",
|
||||
"hitsPerPage": str(count),
|
||||
}
|
||||
# Algolia defaults to AND across query tokens, so a 4-5 word theme query
|
||||
# matches no stories. Mark all-but-the-first token as optional so Algolia
|
||||
# ranks by how many tokens match instead of requiring every one.
|
||||
tokens = core_flat.split()
|
||||
if len(tokens) > 1:
|
||||
params["optionalWords"] = " ".join(tokens[1:])
|
||||
|
||||
from urllib.parse import urlencode
|
||||
url = f"{ALGOLIA_SEARCH_URL}?{urlencode(params)}"
|
||||
@@ -117,28 +126,56 @@ def search_hackernews(
|
||||
return response
|
||||
|
||||
|
||||
def _title_matches_query(title: str, query: str, author: str = "") -> bool:
|
||||
"""Check if the query term appears in the title content, not just an HN prefix or author.
|
||||
_WORD_BOUNDARY_RE_CACHE: Dict[str, "re.Pattern[str]"] = {}
|
||||
|
||||
Returns True if the query (or any multi-word token) appears in the title
|
||||
after stripping "Tell HN:", "Show HN:", "Ask HN:", "Launch HN:" prefixes
|
||||
and ignoring the author name. Returns True when query is empty (no filter).
|
||||
|
||||
def _flatten_query_for_algolia(text: str) -> str:
|
||||
"""Normalise query for Algolia + post-filter comparison.
|
||||
|
||||
Multi-keyword theme queries frequently contain commas (delimiters) or
|
||||
hyphens (compound terms like ``ts-bun-node``); both tokenize awkwardly.
|
||||
Flatten them to spaces and collapse runs of whitespace so the search
|
||||
parameter and the post-filter operate on the same shape.
|
||||
"""
|
||||
return " ".join(text.replace(",", " ").replace("-", " ").split())
|
||||
|
||||
|
||||
def _title_matches_query(title: str, query: str, author: str = "") -> bool:
|
||||
"""Check if any query token appears as a whole word in the title.
|
||||
|
||||
Returns True when the query is empty (no filter), or when at least one
|
||||
query token matches as a whole word in the title after stripping
|
||||
"Tell HN:", "Show HN:", "Ask HN:", "Launch HN:" prefixes.
|
||||
|
||||
We previously required *every* token to appear (all-words), which killed
|
||||
every Algolia hit on multi-keyword themes like "claude, personal agents,
|
||||
agentic infra" because real HN titles never contain all five tokens
|
||||
verbatim. Relaxing to any-word matches Algolia's `optionalWords` behaviour
|
||||
in `search_hackernews`. Token-overlap relevance scoring at parse time
|
||||
demotes hits where only one weak token matched, so the loosened gate
|
||||
won't surface noise to the top of the ranking.
|
||||
|
||||
Word-boundary matching (rather than naive substring) prevents short
|
||||
tokens like ``ai`` or ``ts`` from matching unrelated words like
|
||||
``email`` or ``artists``.
|
||||
"""
|
||||
if not query:
|
||||
return True
|
||||
stripped = _HN_PREFIXES.sub("", title).strip()
|
||||
# Also check that the match isn't solely in the author's username
|
||||
check_text = stripped.lower()
|
||||
query_lower = query.lower()
|
||||
# Check each word of the query independently; all must appear somewhere
|
||||
# in the stripped title (not just the prefix).
|
||||
query_words = query_lower.split()
|
||||
# Normalise the query the same way search_hackernews does so post-filter
|
||||
# tokens line up with what Algolia actually saw.
|
||||
query_words = [w for w in _flatten_query_for_algolia(query.lower()).split() if w]
|
||||
if not query_words:
|
||||
return True
|
||||
for word in query_words:
|
||||
if word in check_text:
|
||||
continue
|
||||
# Word not found in stripped title — reject
|
||||
return False
|
||||
return True
|
||||
pattern = _WORD_BOUNDARY_RE_CACHE.get(word)
|
||||
if pattern is None:
|
||||
pattern = re.compile(rf"\b{re.escape(word)}\b")
|
||||
_WORD_BOUNDARY_RE_CACHE[word] = pattern
|
||||
if pattern.search(check_text):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def parse_hackernews_response(response: Dict[str, Any], query: str = "") -> List[Dict[str, Any]]:
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
import json
|
||||
import re
|
||||
import socket
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
@@ -22,9 +23,19 @@ def log(msg: str):
|
||||
MAX_RETRIES = 5
|
||||
MAX_429_RETRIES = 2
|
||||
RETRY_DELAY = 2.0
|
||||
# DNS resolution failures (gaierror) are transient — typically resolved by a
|
||||
# brief backoff and retry. Use a dedicated minimum attempt count + exponential
|
||||
# delays (1s, 2s, 4s) so callers that pass a small `retries` value still get a
|
||||
# meaningful chance to recover from a transient resolution failure.
|
||||
MIN_DNS_RETRIES = 3
|
||||
USER_AGENT = "last30days-skill/3.0 (Assistant Skill)"
|
||||
|
||||
|
||||
def _is_dns_failure(err: urllib.error.URLError) -> bool:
|
||||
"""Return True if a URLError was caused by DNS resolution (gaierror)."""
|
||||
return isinstance(getattr(err, "reason", None), socket.gaierror)
|
||||
|
||||
|
||||
class HTTPError(Exception):
|
||||
"""HTTP request error with status code."""
|
||||
def __init__(self, message: str, status_code: Optional[int] = None, body: Optional[str] = None):
|
||||
@@ -85,7 +96,13 @@ def request(
|
||||
|
||||
last_error = None
|
||||
rate_limit_count = 0
|
||||
for attempt in range(retries):
|
||||
# DNS failures get a dedicated minimum attempt count + exponential backoff.
|
||||
# `effective_retries` is the actual loop bound; we expand it on the first
|
||||
# gaierror if the caller passed a smaller `retries` value than MIN_DNS_RETRIES.
|
||||
effective_retries = retries
|
||||
dns_attempts = 0
|
||||
attempt = 0
|
||||
while attempt < effective_retries:
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as response:
|
||||
body = response.read().decode('utf-8')
|
||||
@@ -115,6 +132,8 @@ def request(
|
||||
if rate_limit_count >= max_429_retries:
|
||||
raise last_error
|
||||
|
||||
# HTTP errors respect the caller's original `retries`; only DNS
|
||||
# failures get the widened `effective_retries` budget.
|
||||
if attempt < retries - 1:
|
||||
if e.code == 429:
|
||||
# Respect Retry-After header, fall back to exponential backoff
|
||||
@@ -130,11 +149,43 @@ def request(
|
||||
else:
|
||||
delay = RETRY_DELAY * (2 ** attempt)
|
||||
time.sleep(delay)
|
||||
else:
|
||||
# Caller's original retry budget exhausted; an earlier DNS
|
||||
# failure may have widened `effective_retries`, but that
|
||||
# widening is DNS-only — don't grant extra HTTP attempts.
|
||||
break
|
||||
except urllib.error.URLError as e:
|
||||
log(f"URL Error: {e.reason}")
|
||||
last_error = HTTPError(f"URL Error: {e.reason}")
|
||||
if attempt < retries - 1:
|
||||
if _is_dns_failure(e):
|
||||
# DNS resolution failures are transient; expand the retry budget
|
||||
# to MIN_DNS_RETRIES if the caller passed fewer, and use
|
||||
# exponential backoff (1s, 2s, 4s, ...) instead of the linear
|
||||
# default. Counts DNS attempts separately so other URLError
|
||||
# causes don't bypass the regular retry budget.
|
||||
dns_attempts += 1
|
||||
if effective_retries < MIN_DNS_RETRIES:
|
||||
log(
|
||||
f"DNS resolution failed; expanding retry budget from "
|
||||
f"{effective_retries} to {MIN_DNS_RETRIES}"
|
||||
)
|
||||
effective_retries = MIN_DNS_RETRIES
|
||||
if attempt < effective_retries - 1:
|
||||
delay = 2 ** (dns_attempts - 1) # 1s, 2s, 4s, 8s, ...
|
||||
log(
|
||||
f"DNS resolution failure (attempt {dns_attempts}); "
|
||||
f"retrying in {delay:.1f}s"
|
||||
)
|
||||
time.sleep(delay)
|
||||
elif attempt < retries - 1:
|
||||
# Non-DNS URLError (e.g. ConnectionRefused) respects the
|
||||
# caller's original retry budget, not the DNS-widened bound.
|
||||
time.sleep(RETRY_DELAY * (attempt + 1))
|
||||
else:
|
||||
# Caller's original retry budget exhausted; an earlier DNS
|
||||
# failure widening `effective_retries` does not carry over
|
||||
# to non-DNS error paths.
|
||||
break
|
||||
except json.JSONDecodeError as e:
|
||||
log(f"JSON decode error: {e}")
|
||||
last_error = HTTPError(f"Invalid JSON response: {e}")
|
||||
@@ -144,7 +195,13 @@ def request(
|
||||
log(f"Connection error: {type(e).__name__}: {e}")
|
||||
last_error = HTTPError(f"Connection error: {type(e).__name__}: {e}")
|
||||
if attempt < retries - 1:
|
||||
# Socket errors respect the caller's original retry budget.
|
||||
time.sleep(RETRY_DELAY * (attempt + 1))
|
||||
else:
|
||||
# Original budget exhausted; DNS widening doesn't apply here.
|
||||
break
|
||||
|
||||
attempt += 1
|
||||
|
||||
if last_error:
|
||||
raise last_error
|
||||
|
||||
@@ -12,11 +12,6 @@ import sys
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional, Set
|
||||
|
||||
try:
|
||||
import requests as _requests
|
||||
except ImportError:
|
||||
_requests = None
|
||||
|
||||
from . import dates, http, log
|
||||
|
||||
SCRAPECREATORS_BASE = "https://api.scrapecreators.com"
|
||||
@@ -236,30 +231,17 @@ def _user_reels(
|
||||
"""
|
||||
_log(f"User reels: @{handle}")
|
||||
reels_url = f"{SCRAPECREATORS_BASE}/v1/instagram/user/reels"
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"handle": handle})
|
||||
url = f"{reels_url}?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"User reels error (urllib) for @{handle}: {e}")
|
||||
return []
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
reels_url,
|
||||
params={"handle": handle},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"User reels error for @{handle}: {e}")
|
||||
return []
|
||||
try:
|
||||
data = http.get(
|
||||
reels_url,
|
||||
params={"handle": handle},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"User reels error for @{handle}: {e}")
|
||||
return []
|
||||
|
||||
raw_items = data.get("items") or data.get("reels") or data.get("data") or []
|
||||
_log(f" -> {len(raw_items)} reels from @{handle}")
|
||||
@@ -293,31 +275,17 @@ def search_instagram(
|
||||
|
||||
_log(f"Searching Instagram for '{core_topic}' (depth={depth}, count={config['results_per_page']})")
|
||||
|
||||
if not _requests:
|
||||
_log("requests library not installed, falling back to urllib")
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"query": core_topic})
|
||||
url = f"{SCRAPECREATORS_BASE}/v2/instagram/reels/search?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error (urllib): {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_BASE}/v2/instagram/reels/search",
|
||||
params={"query": core_topic},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/v2/instagram/reels/search",
|
||||
params={"query": core_topic},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
|
||||
# Items are in the 'reels' array (ScrapeCreators v2 response)
|
||||
raw_items = data.get("reels") or data.get("items") or data.get("data") or []
|
||||
@@ -367,7 +335,7 @@ def fetch_captions(
|
||||
config = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])
|
||||
max_captions = config["max_captions"]
|
||||
|
||||
if not video_items or not token or not _requests:
|
||||
if not video_items or not token:
|
||||
return {}
|
||||
|
||||
top_items = video_items[:max_captions]
|
||||
@@ -392,26 +360,24 @@ def fetch_captions(
|
||||
if not url:
|
||||
continue
|
||||
try:
|
||||
resp = _requests.get(
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/v2/instagram/media/transcript",
|
||||
params={"url": url},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=15,
|
||||
retries=1,
|
||||
)
|
||||
if resp.status_code == 200:
|
||||
data = resp.json()
|
||||
transcripts = data.get("transcripts") or []
|
||||
if transcripts and isinstance(transcripts, list):
|
||||
# Combine all transcript segments
|
||||
transcript_text = " ".join(
|
||||
t.get("text", "") for t in transcripts
|
||||
if isinstance(t, dict) and t.get("text")
|
||||
)
|
||||
if transcript_text:
|
||||
words = transcript_text.split()
|
||||
if len(words) > CAPTION_MAX_WORDS:
|
||||
transcript_text = ' '.join(words[:CAPTION_MAX_WORDS]) + '...'
|
||||
captions[vid] = transcript_text
|
||||
transcripts = data.get("transcripts") or []
|
||||
if transcripts and isinstance(transcripts, list):
|
||||
transcript_text = " ".join(
|
||||
t.get("text", "") for t in transcripts
|
||||
if isinstance(t, dict) and t.get("text")
|
||||
)
|
||||
if transcript_text:
|
||||
words = transcript_text.split()
|
||||
if len(words) > CAPTION_MAX_WORDS:
|
||||
transcript_text = ' '.join(words[:CAPTION_MAX_WORDS]) + '...'
|
||||
captions[vid] = transcript_text
|
||||
except Exception as e:
|
||||
_log(f"Transcript fetch failed for {vid}: {e}")
|
||||
|
||||
|
||||
@@ -412,7 +412,7 @@ def _normalize_digg(
|
||||
Each cluster is one item. The TLDR carries the most useful body for
|
||||
rerank and synthesis. Top-ranked X posts attached at search time are
|
||||
passed through under metadata['posts'] so render can emit them as
|
||||
inline 'via Digg AI 1000' quotes.
|
||||
inline 'via Digg' quotes.
|
||||
"""
|
||||
title = str(item.get("title") or "").strip()
|
||||
tldr = str(item.get("tldr") or "").strip()
|
||||
@@ -428,7 +428,7 @@ def _normalize_digg(
|
||||
body=body,
|
||||
url=str(item.get("url") or f"https://di.gg/ai/{cluster_url_id}"),
|
||||
author="",
|
||||
container="Digg AI 1000",
|
||||
container="Digg",
|
||||
published_at=item.get("date"),
|
||||
date_confidence=_date_confidence(item, from_date, to_date, default="high"),
|
||||
engagement=item.get("engagement") or {},
|
||||
|
||||
@@ -11,11 +11,6 @@ import re
|
||||
import sys
|
||||
from typing import Any, Dict, List, Optional, Set
|
||||
|
||||
try:
|
||||
import requests as _requests
|
||||
except ImportError:
|
||||
_requests = None
|
||||
|
||||
from . import dates, http, log
|
||||
|
||||
SCRAPECREATORS_BASE = "https://api.scrapecreators.com/v1/pinterest"
|
||||
@@ -140,31 +135,17 @@ def search_pinterest(
|
||||
|
||||
_log(f"Searching Pinterest for '{core_topic}' (depth={depth}, count={config['results_per_page']})")
|
||||
|
||||
if not _requests:
|
||||
_log("requests library not installed, falling back to urllib")
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"keyword": core_topic})
|
||||
url = f"{SCRAPECREATORS_BASE}/search?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error (urllib): {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_BASE}/search",
|
||||
params={"keyword": core_topic},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/search",
|
||||
params={"keyword": core_topic},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
|
||||
# Extract items from response - try common SC response shapes
|
||||
raw_items = data.get("pins") or data.get("results") or data.get("data") or data.get("items") or []
|
||||
|
||||
@@ -128,6 +128,9 @@ def available_sources(config: dict[str, Any], requested_sources: list[str] | Non
|
||||
available.append("pinterest")
|
||||
if env.is_xquik_available(config):
|
||||
available.append("xquik")
|
||||
exclude = {s.strip().lower() for s in (config.get("EXCLUDE_SOURCES") or "").split(",") if s.strip()}
|
||||
if exclude:
|
||||
available = [s for s in available if s not in exclude]
|
||||
return available
|
||||
|
||||
|
||||
@@ -538,7 +541,7 @@ def _finalize_items_by_source(
|
||||
if source == "digg" and items:
|
||||
# Pull top-ranked X posts only for the survivors that will appear
|
||||
# in the brief. Spending the enrichment budget here (rather than
|
||||
# at retrieval time) keeps the inline 'via Digg AI 1000' quotes
|
||||
# at retrieval time) keeps the inline 'via Digg' quotes
|
||||
# paired with the clusters dedupe actually kept.
|
||||
digg.enrich_source_items(items, top_k=3)
|
||||
finalized[source] = items
|
||||
@@ -1078,7 +1081,7 @@ def _mock_stream_results(source: str, subquery: schema.SubQuery) -> tuple[list[d
|
||||
"digg": [
|
||||
{
|
||||
"id": "mock1abc",
|
||||
"title": f"Digg AI 1000 cluster about {subquery.search_query}",
|
||||
"title": f"Digg cluster about {subquery.search_query}",
|
||||
"url": "https://di.gg/ai/mock1abc",
|
||||
"tldr": f"Curated cluster summarizing recent {subquery.search_query} discussion across the AI 1000.",
|
||||
"author": "",
|
||||
|
||||
@@ -9,7 +9,7 @@ from typing import Any
|
||||
|
||||
from . import env, http, schema
|
||||
|
||||
GEMINI_FLASH_LITE = "gemini-3.1-flash-lite-preview"
|
||||
GEMINI_FLASH_LITE = "gemini-3.1-flash-lite"
|
||||
GEMINI_PRO = "gemini-3.1-pro-preview"
|
||||
OPENAI_DEFAULT = "gpt-5.4-nano"
|
||||
XAI_DEFAULT = "grok-4-1-fast"
|
||||
@@ -232,8 +232,8 @@ def _resolve_model_pins(config: dict[str, Any], depth: str, provider_name: str)
|
||||
rerank_model = config.get("LAST30DAYS_RERANK_MODEL") or default_rerank
|
||||
|
||||
if provider_name == "gemini":
|
||||
_require_gemini_31_preview(planner_model, role="planner")
|
||||
_require_gemini_31_preview(rerank_model, role="rerank")
|
||||
_require_gemini_31(planner_model, role="planner")
|
||||
_require_gemini_31(rerank_model, role="rerank")
|
||||
|
||||
return planner_model, rerank_model
|
||||
|
||||
@@ -344,11 +344,11 @@ def _resolve_x_backend(config: dict[str, Any]) -> str | None:
|
||||
return env.get_x_source(config)
|
||||
|
||||
|
||||
def _require_gemini_31_preview(model: str, *, role: str) -> None:
|
||||
if model.startswith("gemini-3.1-") and model.endswith("-preview"):
|
||||
def _require_gemini_31(model: str, *, role: str) -> None:
|
||||
if model.startswith("gemini-3.1-"):
|
||||
return
|
||||
raise RuntimeError(
|
||||
f"{role} must use a Gemini 3.1 preview model. Got: {model}"
|
||||
f"{role} must use a Gemini 3.1 model. Got: {model}"
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -8,25 +8,40 @@ from collections import Counter
|
||||
from datetime import date
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from . import dates, schema
|
||||
from . import dates, schema, skill_meta
|
||||
|
||||
|
||||
def _skill_version() -> str:
|
||||
"""Read plugin version from a plugin manifest if available.
|
||||
"""Read plugin version from .claude-plugin/plugin.json, falling back to SKILL.md frontmatter.
|
||||
|
||||
Tries nearest plugin.json by walking up from render.py's own location.
|
||||
Falls back to "?" if not found. This keeps the badge emission from
|
||||
crashing on non-plugin-cache installs (repo checkout, Gemini, Codex).
|
||||
Per-harness skill install dirs (`~/.claude/skills`, `~/.codex/skills`, `~/.agents/skills`,
|
||||
Hermes, etc.) do not always carry `.claude-plugin/plugin.json` — that file ships with
|
||||
plugin-cache installs but not with per-harness skill installs. SKILL.md frontmatter is
|
||||
the fallback that keeps the badge from emitting v? on those installs. Returns "?" only
|
||||
if no usable version string is found from either source (missing files, corrupt JSON,
|
||||
or SKILL.md without a version line).
|
||||
|
||||
A corrupt manifest at one ancestor does not shadow a valid manifest at a deeper one
|
||||
(continue, not break). SKILL.md parsing accepts double-quoted, single-quoted, or
|
||||
unquoted YAML version scalars (delegated to skill_meta.read_skill_version).
|
||||
"""
|
||||
here = pathlib.Path(__file__).resolve()
|
||||
for parent in [here.parent, *here.parents]:
|
||||
for manifest_dir in (".codex-plugin", ".claude-plugin"):
|
||||
candidate = parent / manifest_dir / "plugin.json"
|
||||
if candidate.is_file():
|
||||
try:
|
||||
return json.loads(candidate.read_text()).get("version", "?")
|
||||
except (json.JSONDecodeError, OSError):
|
||||
return "?"
|
||||
for parent in here.parents:
|
||||
manifest = parent / ".claude-plugin" / "plugin.json"
|
||||
if manifest.is_file():
|
||||
try:
|
||||
version = json.loads(manifest.read_text()).get("version")
|
||||
except (json.JSONDecodeError, OSError):
|
||||
continue
|
||||
if version:
|
||||
return version
|
||||
|
||||
# No usable manifest found at any ancestor — fall back to SKILL.md frontmatter.
|
||||
# First SKILL.md found in the walk is THIS skill's; never traverse past it.
|
||||
for parent in here.parents:
|
||||
skill_md = parent / "SKILL.md"
|
||||
if skill_md.is_file():
|
||||
return skill_meta.read_skill_version(skill_md) or "?"
|
||||
return "?"
|
||||
|
||||
|
||||
@@ -53,7 +68,7 @@ SOURCE_LABELS = {
|
||||
"xiaohongshu": "Xiaohongshu",
|
||||
"x": "X",
|
||||
"github": "GitHub",
|
||||
"digg": "Digg AI 1000",
|
||||
"digg": "Digg",
|
||||
"perplexity": "Perplexity",
|
||||
}
|
||||
|
||||
@@ -81,7 +96,7 @@ def render_compact(report: schema.Report, cluster_limit: int = 8, fun_level: str
|
||||
non_empty = [s for s, items in sorted(report.items_by_source.items()) if items]
|
||||
lines = [
|
||||
*_render_badge(),
|
||||
f"# last30days v3.0.0: {report.topic}",
|
||||
f"# last30days v{_skill_version()}: {report.topic}",
|
||||
"",
|
||||
*_assistant_safety_lines(),
|
||||
f"- Date range: {report.range_from} to {report.range_to}",
|
||||
@@ -587,7 +602,7 @@ def render_comparison_multi(
|
||||
|
||||
lines: list[str] = [
|
||||
*_render_badge(),
|
||||
f"# last30days v3.0.0: {synthesized_topic}",
|
||||
f"# last30days v{_skill_version()}: {synthesized_topic}",
|
||||
"",
|
||||
*_assistant_safety_lines(),
|
||||
f"- Comparison mode: {len(entities)} entities ({', '.join(entities)})",
|
||||
@@ -775,7 +790,7 @@ def render_full(report: schema.Report) -> str:
|
||||
# Start with the same header as compact
|
||||
non_empty = [s for s, items in sorted(report.items_by_source.items()) if items]
|
||||
lines = [
|
||||
f"# last30days v3.0.0: {report.topic}",
|
||||
f"# last30days v{_skill_version()}: {report.topic}",
|
||||
"",
|
||||
*_assistant_safety_lines(),
|
||||
f"- Date range: {report.range_from} to {report.range_to}",
|
||||
@@ -853,7 +868,7 @@ def render_full(report: schema.Report) -> str:
|
||||
tc_score = tc.get("score", "")
|
||||
attribution = _comment_attribution(item.source, tc.get("author"))
|
||||
lines.append(f" Top comment {attribution} ({tc_score} {vote_label}): {excerpt}")
|
||||
# Digg AI 1000: inline X-post quotes attached to the cluster.
|
||||
# Digg: inline X-post quotes attached to the cluster.
|
||||
for post in _digg_posts_for(item, limit=3):
|
||||
lines.append(f" > {_format_digg_quote(post)}")
|
||||
# Comment insights for Reddit
|
||||
@@ -1229,7 +1244,7 @@ _FOOTER_SOURCES: list[tuple[str, str, str, str, list[tuple[str, str]]]] = [
|
||||
("bluesky", "🦋", "Bluesky", "post", [("likes", "likes"), ("reposts", "reposts")]),
|
||||
("truthsocial", "🇺🇸", "Truth Social", "post", [("likes", "likes"), ("reposts", "reposts")]),
|
||||
("github", "🐙", "GitHub", "item", [("reactions", "reactions"), ("comments", "comments")]),
|
||||
("digg", "⛏️", "Digg AI 1000", "cluster", [("postCount", "posts"), ("uniqueAuthors", "authors")]),
|
||||
("digg", "⛏️", "Digg", "cluster", [("postCount", "posts"), ("uniqueAuthors", "authors")]),
|
||||
]
|
||||
|
||||
|
||||
@@ -1684,7 +1699,7 @@ def _comment_insight(item: schema.SourceItem | None) -> str | None:
|
||||
return str(insights[0]).strip() or None
|
||||
|
||||
|
||||
def _digg_posts_for(item: schema.SourceItem | None, limit: int = 2) -> list[dict]:
|
||||
def _digg_posts_for(item: schema.SourceItem | None, limit: int = 3) -> list[dict]:
|
||||
"""Return up to `limit` parsed Digg posts attached as enrichment to a cluster.
|
||||
|
||||
Returns an empty list for non-digg sources or clusters without enrichment.
|
||||
@@ -1704,17 +1719,17 @@ def _digg_posts_for(item: schema.SourceItem | None, limit: int = 2) -> list[dict
|
||||
|
||||
|
||||
def _format_digg_quote(post: dict, body_limit: int = 200) -> str:
|
||||
"""Format a Digg-attached X post as an inline 'via Digg AI 1000' quote line."""
|
||||
"""Format a Digg-attached X post as an inline 'via Digg' quote line."""
|
||||
handle = post.get("username") or ""
|
||||
x_url = post.get("x_url") or ""
|
||||
body = (post.get("body") or "").replace("\n", " ").strip()
|
||||
if len(body) > body_limit:
|
||||
body = body[: body_limit - 1].rstrip() + "…"
|
||||
if x_url and handle:
|
||||
return f"[@{handle}]({x_url}) via Digg AI 1000: {body}"
|
||||
return f"[@{handle}]({x_url}) via Digg: {body}"
|
||||
if handle:
|
||||
return f"@{handle} via Digg AI 1000: {body}"
|
||||
return f"via Digg AI 1000: {body}"
|
||||
return f"@{handle} via Digg: {body}"
|
||||
return f"via Digg: {body}"
|
||||
|
||||
|
||||
def _transcript_highlights(item: schema.SourceItem | None) -> list[str]:
|
||||
|
||||
@@ -335,8 +335,9 @@ def poll_device_auth(
|
||||
"""
|
||||
import sys
|
||||
|
||||
deadline = time.time() + timeout
|
||||
last_reminder = time.time()
|
||||
started_at = time.time()
|
||||
deadline = started_at + timeout
|
||||
last_reminder = started_at
|
||||
reminder_count = 0
|
||||
max_reminders = 4
|
||||
reminder_interval = 30 # seconds between reminders
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
"""SKILL.md metadata helpers — single source of truth for parsing skill frontmatter.
|
||||
|
||||
Centralizes the version regex that previously lived in render.py and was
|
||||
duplicated in tests/test_plugin_contract.py and tests/test_version_consistency.py.
|
||||
"""
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
# Matches `version: "x.y.z"`, `version: 'x.y.z'`, or `version: x.y.z` in YAML
|
||||
# frontmatter. Multiline so the pattern can be applied to a full SKILL.md text.
|
||||
# Three alternation groups — exactly one captures per successful match.
|
||||
_VERSION_RE = re.compile(
|
||||
r'''^version:\s*(?:"([^"]+)"|'([^']+)'|(\S+))\s*$''',
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
|
||||
def read_skill_version(skill_md_path: Path) -> str | None:
|
||||
"""Return the version string from a SKILL.md's frontmatter, or None.
|
||||
|
||||
Returns None if the file can't be read (missing, permission, decode error)
|
||||
or if no `version:` line is found. Accepts double-quoted, single-quoted,
|
||||
or unquoted YAML version scalars.
|
||||
"""
|
||||
try:
|
||||
text = skill_md_path.read_text()
|
||||
except (OSError, UnicodeDecodeError):
|
||||
return None
|
||||
match = _VERSION_RE.search(text)
|
||||
if not match:
|
||||
return None
|
||||
return match.group(1) or match.group(2) or match.group(3)
|
||||
@@ -152,35 +152,16 @@ def search_threads(
|
||||
_log(f"Searching for '{core_topic}' (depth={depth}, limit={config['results']})")
|
||||
|
||||
try:
|
||||
import requests as _requests
|
||||
except ImportError:
|
||||
_requests = None
|
||||
|
||||
if not _requests:
|
||||
_log("requests library not installed, falling back to urllib")
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"keyword": core_topic})
|
||||
url = f"{SCRAPECREATORS_BASE}/search?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error (urllib): {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_BASE}/search",
|
||||
params={"keyword": core_topic},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/search",
|
||||
params={"keyword": core_topic},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
|
||||
# Extract items from response (try common SC response shapes)
|
||||
raw_items = (
|
||||
|
||||
@@ -11,11 +11,6 @@ import re
|
||||
import sys
|
||||
from typing import Any, Dict, List, Optional, Set
|
||||
|
||||
try:
|
||||
import requests as _requests
|
||||
except ImportError:
|
||||
_requests = None
|
||||
|
||||
from . import dates, http, log
|
||||
|
||||
SCRAPECREATORS_BASE = "https://api.scrapecreators.com/v1/tiktok"
|
||||
@@ -214,30 +209,17 @@ def _hashtag_search(
|
||||
List of raw TikTok item dicts (aweme_info format).
|
||||
"""
|
||||
_log(f"Hashtag search: #{hashtag}")
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"hashtag": hashtag})
|
||||
url = f"{SCRAPECREATORS_BASE}/search/hashtag?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"Hashtag search error (urllib) for #{hashtag}: {e}")
|
||||
return []
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_BASE}/search/hashtag",
|
||||
params={"hashtag": hashtag},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"Hashtag search error for #{hashtag}: {e}")
|
||||
return []
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/search/hashtag",
|
||||
params={"hashtag": hashtag},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"Hashtag search error for #{hashtag}: {e}")
|
||||
return []
|
||||
|
||||
raw_items = data.get("aweme_list") or data.get("data") or []
|
||||
_log(f" -> {len(raw_items)} results for #{hashtag}")
|
||||
@@ -261,30 +243,17 @@ def _profile_videos(
|
||||
"""
|
||||
_log(f"Profile videos: @{handle}")
|
||||
profile_url = "https://api.scrapecreators.com/v3/tiktok/profile/videos"
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"handle": handle, "sort_by": "latest"})
|
||||
url = f"{profile_url}?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"Profile videos error (urllib) for @{handle}: {e}")
|
||||
return []
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
profile_url,
|
||||
params={"handle": handle, "sort_by": "latest"},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"Profile videos error for @{handle}: {e}")
|
||||
return []
|
||||
try:
|
||||
data = http.get(
|
||||
profile_url,
|
||||
params={"handle": handle, "sort_by": "latest"},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"Profile videos error for @{handle}: {e}")
|
||||
return []
|
||||
|
||||
raw_items = data.get("aweme_list") or data.get("data") or []
|
||||
_log(f" -> {len(raw_items)} videos from @{handle}")
|
||||
@@ -318,31 +287,17 @@ def search_tiktok(
|
||||
|
||||
_log(f"Searching TikTok for '{core_topic}' (depth={depth}, count={config['results_per_page']})")
|
||||
|
||||
if not _requests:
|
||||
_log("requests library not installed, falling back to urllib")
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"query": core_topic, "sort_by": "relevance"})
|
||||
url = f"{SCRAPECREATORS_BASE}/search/keyword?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error (urllib): {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_BASE}/search/keyword",
|
||||
params={"query": core_topic, "sort_by": "relevance"},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/search/keyword",
|
||||
params={"query": core_topic, "sort_by": "relevance"},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as e:
|
||||
_log(f"ScrapeCreators error: {e}")
|
||||
return {"items": [], "error": f"{type(e).__name__}: {e}"}
|
||||
|
||||
# Items are nested under aweme_info
|
||||
raw_entries = data.get("search_item_list") or data.get("data") or []
|
||||
@@ -397,7 +352,7 @@ def fetch_captions(
|
||||
config = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])
|
||||
max_captions = config["max_captions"]
|
||||
|
||||
if not video_items or not token or not _requests:
|
||||
if not video_items or not token:
|
||||
return {}
|
||||
|
||||
top_items = video_items[:max_captions]
|
||||
@@ -422,24 +377,23 @@ def fetch_captions(
|
||||
if not url:
|
||||
continue
|
||||
try:
|
||||
resp = _requests.get(
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/video/transcript",
|
||||
params={"url": url},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=15,
|
||||
retries=1,
|
||||
)
|
||||
if resp.status_code == 200:
|
||||
data = resp.json()
|
||||
transcript = data.get("transcript")
|
||||
transcript = data.get("transcript")
|
||||
if transcript:
|
||||
if isinstance(transcript, list):
|
||||
transcript = " ".join(str(s) for s in transcript)
|
||||
transcript = _clean_webvtt(transcript)
|
||||
if transcript:
|
||||
if isinstance(transcript, list):
|
||||
transcript = " ".join(str(s) for s in transcript)
|
||||
transcript = _clean_webvtt(transcript)
|
||||
if transcript:
|
||||
words = transcript.split()
|
||||
if len(words) > CAPTION_MAX_WORDS:
|
||||
transcript = ' '.join(words[:CAPTION_MAX_WORDS]) + '...'
|
||||
captions[vid] = transcript
|
||||
words = transcript.split()
|
||||
if len(words) > CAPTION_MAX_WORDS:
|
||||
transcript = ' '.join(words[:CAPTION_MAX_WORDS]) + '...'
|
||||
captions[vid] = transcript
|
||||
except Exception as e:
|
||||
_log(f"Transcript fetch failed for {vid}: {e}")
|
||||
|
||||
@@ -620,30 +574,17 @@ def _fetch_post_comments(
|
||||
List of comment dicts with author, text, digg_count (likes), date.
|
||||
Empty list on any error — comment failures never crash the pipeline.
|
||||
"""
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"url": post_url, "trim": "true"})
|
||||
url = f"{SCRAPECREATORS_BASE}/video/comments?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as exc:
|
||||
_log(f"Comment fetch error (urllib) for {post_url}: {exc}")
|
||||
return []
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_BASE}/video/comments",
|
||||
params={"url": post_url, "trim": "true"},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as exc:
|
||||
_log(f"Comment fetch error for {post_url}: {exc}")
|
||||
return []
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_BASE}/video/comments",
|
||||
params={"url": post_url, "trim": "true"},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as exc:
|
||||
_log(f"Comment fetch error for {post_url}: {exc}")
|
||||
return []
|
||||
|
||||
raw_comments = data.get("comments") or data.get("data") or []
|
||||
# Sort by digg_count desc so normalize sees the highest-signal first.
|
||||
|
||||
@@ -6,6 +6,8 @@ import threading
|
||||
import random
|
||||
from typing import Optional
|
||||
|
||||
from .render import _skill_version
|
||||
|
||||
# Check if we're in a real terminal (not captured by Claude Code)
|
||||
IS_TTY = sys.stderr.isatty()
|
||||
|
||||
@@ -198,7 +200,7 @@ Just start with "last30" and talk to me like normal.
|
||||
|
||||
# Shorter promo for single missing key
|
||||
PROMO_SINGLE_KEY = {
|
||||
"reddit": "\n💡 Unlock TikTok and Instagram with SCRAPECREATORS_API_KEY - 10,000 free calls, no CC - scrapecreators.com\n",
|
||||
"reddit": "\n💡 Unlock TikTok and Instagram with SCRAPECREATORS_API_KEY - 100 free credits, no CC - scrapecreators.com\n",
|
||||
"x": "\n💡 Unlock X: log into x.com in Firefox or Safari, then re-run. Or add AUTH_TOKEN/CT0 or XAI_API_KEY.\n",
|
||||
"web": "\n💡 You can unlock native grounded web search with BRAVE_API_KEY or SERPER_API_KEY.\n",
|
||||
}
|
||||
@@ -509,7 +511,8 @@ def show_diagnostic_banner(diag: dict):
|
||||
|
||||
if IS_TTY:
|
||||
lines.append(f"{Colors.DIM}┌─────────────────────────────────────────────────────┐{Colors.RESET}")
|
||||
lines.append(f"{Colors.DIM}│{Colors.RESET} {Colors.BOLD}/last30days v3.0.0 - Source Status{Colors.RESET} {Colors.DIM}│{Colors.RESET}")
|
||||
_header = f"/last30days v{_skill_version()} - Source Status"
|
||||
lines.append(f"{Colors.DIM}│{Colors.RESET} {Colors.BOLD}{_header}{Colors.RESET}{' ' * (52 - len(_header))}{Colors.DIM}│{Colors.RESET}")
|
||||
lines.append(f"{Colors.DIM}│{Colors.RESET} {Colors.DIM}│{Colors.RESET}")
|
||||
|
||||
# Reddit
|
||||
@@ -556,7 +559,8 @@ def show_diagnostic_banner(diag: dict):
|
||||
else:
|
||||
# Plain text for non-TTY (Claude Code / Codex)
|
||||
lines.append("┌─────────────────────────────────────────────────────┐")
|
||||
lines.append("│ /last30days v3.0.0 - Source Status │")
|
||||
_header_plain = f"/last30days v{_skill_version()} - Source Status"
|
||||
lines.append(f"│ {_header_plain}{' ' * (52 - len(_header_plain))}│")
|
||||
lines.append("│ │")
|
||||
|
||||
if has_reddit and has_scrapecreators:
|
||||
|
||||
@@ -8,7 +8,9 @@ Inspired by Peter Steinberger's toolchain approach (yt-dlp + summarize CLI).
|
||||
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
import shlex
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
@@ -96,10 +98,76 @@ def _log(msg: str):
|
||||
|
||||
|
||||
def is_ytdlp_installed() -> bool:
|
||||
"""Check if yt-dlp is available in PATH."""
|
||||
"""Check if yt-dlp is available locally, or if SSH routing is configured.
|
||||
|
||||
When LAST30DAYS_YOUTUBE_SSH_HOST is set, returns True without a local check —
|
||||
yt-dlp lives on the remote host. Failures surface naturally on first use.
|
||||
"""
|
||||
if _ytdlp_ssh_host():
|
||||
return True
|
||||
return shutil.which("yt-dlp") is not None
|
||||
|
||||
|
||||
# Host aliases must be plain hostnames / SSH config aliases — no flags, no
|
||||
# shell metacharacters. Rejects any value that could be reinterpreted by ssh
|
||||
# (or the surrounding shell) as something other than a destination.
|
||||
_SSH_HOST_ALIAS_RE = re.compile(r"^[a-zA-Z0-9._-]+$")
|
||||
|
||||
|
||||
def _ytdlp_ssh_host() -> Optional[str]:
|
||||
"""Return SSH host alias if yt-dlp should be routed via SSH, else None.
|
||||
|
||||
Set LAST30DAYS_YOUTUBE_SSH_HOST=<ssh-alias> (e.g. 'macmini') in the environment
|
||||
to route yt-dlp through SSH for residential IP egress. This bypasses
|
||||
YouTube's bot-wall on datacenter IPs (Hetzner, DigitalOcean, AWS, etc.)
|
||||
where ytsearch returns 0 results regardless of cookies.
|
||||
|
||||
The remote host must have yt-dlp installed and reachable via the named
|
||||
SSH alias (configured in ~/.ssh/config). On macOS hosts with Homebrew,
|
||||
add brew shellenv to ~/.zshenv (not just ~/.zprofile) so non-login SSH
|
||||
shells find yt-dlp on PATH.
|
||||
|
||||
Validation: host value must match ``[A-Za-z0-9._-]+``. Anything starting
|
||||
with ``-`` or containing shell/SSH metacharacters is rejected with a
|
||||
stderr warning and treated as unset, so a misconfigured or attacker-
|
||||
controlled value can't slip through as an SSH option flag or proxy command.
|
||||
The ``--`` option terminator in ``_wrap_ytdlp_cmd`` is a second line of
|
||||
defense; this regex closes the door on the env var ever reaching ssh
|
||||
in the first place.
|
||||
|
||||
To use a value from ~/.config/last30days/.env, export it into the
|
||||
environment before invoking the engine, e.g. in a wrapper:
|
||||
set -a; source ~/.config/last30days/.env; set +a
|
||||
python3 last30days.py "..."
|
||||
"""
|
||||
host = os.environ.get("LAST30DAYS_YOUTUBE_SSH_HOST", "").strip()
|
||||
if not host:
|
||||
return None
|
||||
if not _SSH_HOST_ALIAS_RE.match(host):
|
||||
sys.stderr.write(
|
||||
f"[youtube_yt] WARNING: LAST30DAYS_YOUTUBE_SSH_HOST={host!r} "
|
||||
"does not look like a plain hostname/alias; ignoring. "
|
||||
"Expected pattern: letters, digits, dot, underscore, hyphen.\n"
|
||||
)
|
||||
return None
|
||||
return host
|
||||
|
||||
|
||||
def _wrap_ytdlp_cmd(cmd: List[str]) -> List[str]:
|
||||
"""Wrap a yt-dlp command list with `ssh <host>` when SSH routing is set.
|
||||
|
||||
Args are shell-quoted to survive the remote shell. Uses BatchMode=yes so
|
||||
a misconfigured key fails fast instead of hanging on a password prompt.
|
||||
The `--` option terminator prevents an SSH option-injection if
|
||||
LAST30DAYS_YOUTUBE_SSH_HOST were ever set to a value starting with `-`.
|
||||
"""
|
||||
host = _ytdlp_ssh_host()
|
||||
if not host:
|
||||
return cmd
|
||||
remote_cmd = " ".join(shlex.quote(a) for a in cmd)
|
||||
return ["ssh", "-o", "BatchMode=yes", "--", host, remote_cmd]
|
||||
|
||||
|
||||
def _extract_core_subject(topic: str) -> str:
|
||||
"""Extract core subject from verbose query for YouTube search.
|
||||
|
||||
@@ -223,6 +291,7 @@ def search_youtube(
|
||||
"--no-warnings",
|
||||
"--no-download",
|
||||
]
|
||||
cmd = _wrap_ytdlp_cmd(cmd)
|
||||
|
||||
try:
|
||||
result = subproc.run_with_timeout(cmd, timeout=120)
|
||||
@@ -472,13 +541,22 @@ def fetch_transcript(video_id: str, temp_dir: str) -> Optional[str]:
|
||||
Plaintext transcript string, or None if no captions available.
|
||||
"""
|
||||
raw_vtt = None
|
||||
if is_ytdlp_installed():
|
||||
# When SSH-routing is on, the yt-dlp transcript path would write a VTT
|
||||
# file on the remote host that we can't easily read back. Skip it and
|
||||
# use the HTTP transcript fallback (different YouTube endpoint, less
|
||||
# bot-walled, works fine from datacenter IPs).
|
||||
ssh_host = _ytdlp_ssh_host()
|
||||
use_ytdlp = is_ytdlp_installed() and not ssh_host
|
||||
if use_ytdlp:
|
||||
raw_vtt = _fetch_transcript_ytdlp(video_id, temp_dir)
|
||||
if not raw_vtt:
|
||||
_log(f"yt-dlp transcript failed for {video_id}, trying direct HTTP fallback")
|
||||
raw_vtt = _fetch_transcript_direct(video_id)
|
||||
else:
|
||||
_log("yt-dlp not installed, using direct HTTP transcript fetch")
|
||||
if ssh_host:
|
||||
_log("SSH-routing active, using direct HTTP transcript fetch")
|
||||
else:
|
||||
_log("yt-dlp not installed, using direct HTTP transcript fetch")
|
||||
raw_vtt = _fetch_transcript_direct(video_id)
|
||||
|
||||
if not raw_vtt:
|
||||
@@ -617,11 +695,6 @@ def parse_youtube_response(response: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||
|
||||
SCRAPECREATORS_YT_BASE = "https://api.scrapecreators.com/v1/youtube"
|
||||
|
||||
try:
|
||||
import requests as _requests
|
||||
except ImportError:
|
||||
_requests = None
|
||||
|
||||
|
||||
def _total_engagement(item: Dict[str, Any]) -> int:
|
||||
"""Combined engagement score for ranking which videos to enrich."""
|
||||
@@ -701,30 +774,17 @@ def _fetch_video_comments(
|
||||
List of comment dicts with author, text, likes, date.
|
||||
"""
|
||||
video_url = f"https://www.youtube.com/watch?v={video_id}"
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"url": video_url})
|
||||
url = f"{SCRAPECREATORS_YT_BASE}/video/comments?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as exc:
|
||||
_log(f"Comment fetch error (urllib) for {video_id}: {exc}")
|
||||
return []
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_YT_BASE}/video/comments",
|
||||
params={"url": video_url},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except Exception as exc:
|
||||
_log(f"Comment fetch error for {video_id}: {exc}")
|
||||
return []
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_YT_BASE}/video/comments",
|
||||
params={"url": video_url},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
except Exception as exc:
|
||||
_log(f"Comment fetch error for {video_id}: {exc}")
|
||||
return []
|
||||
|
||||
raw_comments = data.get("comments", data.get("data", []))
|
||||
comments = []
|
||||
@@ -883,28 +943,17 @@ def _sc_youtube_search(keyword: str, token: str) -> List[Dict[str, Any]]:
|
||||
Returns:
|
||||
List of raw video dicts from the API.
|
||||
"""
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"keyword": keyword})
|
||||
url = f"{SCRAPECREATORS_YT_BASE}/search?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
return data.get("videos", data.get("data", data.get("items", [])))
|
||||
except Exception as exc:
|
||||
_log(f"SC YouTube search error (urllib): {exc}")
|
||||
return []
|
||||
|
||||
try:
|
||||
resp = _requests.get(
|
||||
# SC's /v1/youtube/search rejects ?keyword= with HTTP 400; the canonical
|
||||
# parameter for that endpoint is `query`. Other SC endpoints use their
|
||||
# own per-endpoint param names so this was the lone outlier.
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_YT_BASE}/search",
|
||||
params={"keyword": keyword},
|
||||
params={"query": keyword},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=2,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return data.get("videos", data.get("data", data.get("items", [])))
|
||||
except Exception as exc:
|
||||
_log(f"SC YouTube search error: {exc}")
|
||||
@@ -922,32 +971,17 @@ def _sc_fetch_transcript(video_id: str, token: str) -> Optional[str]:
|
||||
Plaintext transcript string, or None if unavailable.
|
||||
"""
|
||||
video_url = f"https://www.youtube.com/watch?v={video_id}"
|
||||
if not _requests:
|
||||
try:
|
||||
from urllib.parse import urlencode
|
||||
params = urlencode({"url": video_url})
|
||||
url = f"{SCRAPECREATORS_YT_BASE}/video/transcript?{params}"
|
||||
headers = http.scrapecreators_headers(token)
|
||||
headers["User-Agent"] = http.USER_AGENT
|
||||
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||
except Exception as exc:
|
||||
_log(f"SC transcript error (urllib) for {video_id}: {exc}")
|
||||
return None
|
||||
else:
|
||||
try:
|
||||
resp = _requests.get(
|
||||
f"{SCRAPECREATORS_YT_BASE}/video/transcript",
|
||||
params={"url": video_url},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
)
|
||||
if resp.status_code != 200:
|
||||
_log(f"SC transcript returned {resp.status_code} for {video_id}")
|
||||
return None
|
||||
data = resp.json()
|
||||
except Exception as exc:
|
||||
_log(f"SC transcript error for {video_id}: {exc}")
|
||||
return None
|
||||
try:
|
||||
data = http.get(
|
||||
f"{SCRAPECREATORS_YT_BASE}/video/transcript",
|
||||
params={"url": video_url},
|
||||
headers=http.scrapecreators_headers(token),
|
||||
timeout=30,
|
||||
retries=1,
|
||||
)
|
||||
except Exception as exc:
|
||||
_log(f"SC transcript error for {video_id}: {exc}")
|
||||
return None
|
||||
|
||||
transcript = data.get("transcript")
|
||||
if not transcript:
|
||||
|
||||
Executable
+122
@@ -0,0 +1,122 @@
|
||||
#!/bin/bash
|
||||
# Store last30days API keys in the macOS Keychain.
|
||||
#
|
||||
# Keys are stored as generic passwords with service name `last30days-<KEY>`
|
||||
# for the current user. The lib/env.py loader picks them up automatically as
|
||||
# the lowest-priority credential source on Darwin.
|
||||
#
|
||||
# Usage:
|
||||
# ./setup-keychain.sh # interactive: prompts for each key
|
||||
# ./setup-keychain.sh KEY [KEY..] # prompt only for the listed keys
|
||||
# ./setup-keychain.sh --list # list which last30days-* items exist
|
||||
# ./setup-keychain.sh --delete KEY # remove a stored key
|
||||
#
|
||||
# Existing values are shown as "(set)" and skipped unless --replace is passed.
|
||||
# Skip any prompt with empty input.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
PREFIX="last30days-"
|
||||
# Mirrors lib/env.py::KEYCHAIN_KEYS — kept in sync via
|
||||
# tests/test_env_keychain.py::test_keychain_keys_match_setup_script.
|
||||
ALL_KEYS=(
|
||||
OPENAI_API_KEY
|
||||
XAI_API_KEY
|
||||
GOOGLE_API_KEY
|
||||
GEMINI_API_KEY
|
||||
GOOGLE_GENAI_API_KEY
|
||||
SCRAPECREATORS_API_KEY
|
||||
APIFY_API_TOKEN
|
||||
AUTH_TOKEN
|
||||
CT0
|
||||
BSKY_HANDLE
|
||||
BSKY_APP_PASSWORD
|
||||
TRUTHSOCIAL_TOKEN
|
||||
BRAVE_API_KEY
|
||||
EXA_API_KEY
|
||||
SERPER_API_KEY
|
||||
OPENROUTER_API_KEY
|
||||
PARALLEL_API_KEY
|
||||
XQUIK_API_KEY
|
||||
XIAOHONGSHU_API_BASE
|
||||
)
|
||||
|
||||
if [[ "${OSTYPE:-}" != darwin* ]]; then
|
||||
echo "setup-keychain.sh requires macOS (security command). Got: $OSTYPE" >&2
|
||||
exit 1
|
||||
fi
|
||||
if ! command -v security >/dev/null 2>&1; then
|
||||
echo "security command not found on PATH" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
REPLACE=0
|
||||
ACTION="prompt"
|
||||
TARGETS=()
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--list) ACTION="list"; shift ;;
|
||||
--delete) ACTION="delete"; shift ;;
|
||||
--replace) REPLACE=1; shift ;;
|
||||
--help|-h) sed -n '2,/^$/p' "$0" | sed 's/^# //; s/^#//'; exit 0 ;;
|
||||
-*) echo "unknown flag: $1" >&2; exit 2 ;;
|
||||
*) TARGETS+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
case "$ACTION" in
|
||||
list)
|
||||
echo "Stored last30days-* keychain items:"
|
||||
for key in "${ALL_KEYS[@]}"; do
|
||||
if security find-generic-password -a "$USER" -s "${PREFIX}${key}" -w >/dev/null 2>&1; then
|
||||
echo " $key"
|
||||
fi
|
||||
done
|
||||
exit 0
|
||||
;;
|
||||
delete)
|
||||
if [[ ${#TARGETS[@]} -eq 0 ]]; then
|
||||
echo "--delete needs at least one KEY name" >&2; exit 2
|
||||
fi
|
||||
for key in "${TARGETS[@]}"; do
|
||||
if security delete-generic-password -a "$USER" -s "${PREFIX}${key}" >/dev/null 2>&1; then
|
||||
echo "deleted: $key"
|
||||
else
|
||||
echo "not found: $key"
|
||||
fi
|
||||
done
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
|
||||
if [[ ${#TARGETS[@]} -eq 0 ]]; then
|
||||
TARGETS=("${ALL_KEYS[@]}")
|
||||
fi
|
||||
|
||||
added=0; skipped=0; replaced=0
|
||||
for key in "${TARGETS[@]}"; do
|
||||
existing="$(security find-generic-password -a "$USER" -s "${PREFIX}${key}" -w 2>/dev/null || true)"
|
||||
if [[ -n "$existing" && "$REPLACE" -eq 0 ]]; then
|
||||
printf " %-28s (set, skipping — use --replace to overwrite)\n" "$key"
|
||||
skipped=$((skipped + 1))
|
||||
continue
|
||||
fi
|
||||
printf " %-28s " "$key"
|
||||
IFS= read -rs value
|
||||
echo
|
||||
if [[ -z "$value" ]]; then
|
||||
skipped=$((skipped + 1))
|
||||
continue
|
||||
fi
|
||||
security add-generic-password -U -a "$USER" -s "${PREFIX}${key}" -w "$value"
|
||||
if [[ -n "$existing" ]]; then
|
||||
replaced=$((replaced + 1))
|
||||
else
|
||||
added=$((added + 1))
|
||||
fi
|
||||
done
|
||||
|
||||
echo
|
||||
echo "Done. added=$added replaced=$replaced skipped=$skipped"
|
||||
echo "Verify with: $0 --list"
|
||||
@@ -14,7 +14,7 @@ import argparse
|
||||
import json
|
||||
import sqlite3
|
||||
import sys
|
||||
from datetime import datetime, timedelta
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
@@ -159,7 +159,30 @@ _UPDATABLE_FINDING_COLUMNS = frozenset({
|
||||
})
|
||||
|
||||
# Future migrations keyed by version number
|
||||
MIGRATIONS: Dict[int, str] = {}
|
||||
MIGRATIONS: Dict[int, str] = {
|
||||
2: """
|
||||
CREATE TABLE IF NOT EXISTS finding_sightings (
|
||||
id INTEGER PRIMARY KEY,
|
||||
finding_id INTEGER NOT NULL REFERENCES findings(id) ON DELETE CASCADE,
|
||||
run_id INTEGER REFERENCES research_runs(id) ON DELETE CASCADE,
|
||||
topic_id INTEGER REFERENCES topics(id) ON DELETE CASCADE,
|
||||
source TEXT NOT NULL,
|
||||
source_url TEXT NOT NULL,
|
||||
source_title TEXT,
|
||||
engagement_score REAL,
|
||||
relevance_score REAL,
|
||||
seen_at TEXT DEFAULT (datetime('now')),
|
||||
UNIQUE(run_id, finding_id)
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS idx_finding_sightings_run
|
||||
ON finding_sightings(run_id, topic_id);
|
||||
CREATE INDEX IF NOT EXISTS idx_finding_sightings_topic_seen
|
||||
ON finding_sightings(topic_id, seen_at);
|
||||
CREATE INDEX IF NOT EXISTS idx_finding_sightings_url
|
||||
ON finding_sightings(source_url);
|
||||
""",
|
||||
}
|
||||
|
||||
|
||||
def _connect(db_path: Optional[Path] = None) -> sqlite3.Connection:
|
||||
@@ -423,6 +446,7 @@ def store_findings(
|
||||
|
||||
new_count = len(insert_rows)
|
||||
updated_count = len(update_rows)
|
||||
_record_sightings(conn, run_id, topic_id, with_urls, existing_by_url)
|
||||
conn.execute(
|
||||
"UPDATE research_runs SET findings_new = ?, findings_updated = ? WHERE id = ?",
|
||||
(new_count, updated_count, run_id),
|
||||
@@ -434,6 +458,84 @@ def store_findings(
|
||||
return {"new": new_count, "updated": updated_count}
|
||||
|
||||
|
||||
def _record_sightings(
|
||||
conn: sqlite3.Connection,
|
||||
run_id: int,
|
||||
topic_id: int,
|
||||
findings_with_urls: List[tuple[str, Dict[str, Any]]],
|
||||
existing_by_url: Optional[Dict[str, sqlite3.Row]] = None,
|
||||
) -> None:
|
||||
"""Record the findings observed during this run.
|
||||
|
||||
The aggregate findings table keeps one row per URL and updates that row on
|
||||
re-sighting. This ledger preserves the run/topic membership needed for
|
||||
watchlist deltas and dossiers.
|
||||
"""
|
||||
if not findings_with_urls:
|
||||
return
|
||||
|
||||
by_url = {url: finding for url, finding in findings_with_urls}
|
||||
rows_by_url = dict(existing_by_url or {})
|
||||
|
||||
missing_urls = [url for url in by_url if url not in rows_by_url]
|
||||
if missing_urls:
|
||||
placeholders = ",".join("?" for _ in missing_urls)
|
||||
rows = conn.execute(
|
||||
f"SELECT id, source_url FROM findings WHERE source_url IN ({placeholders})",
|
||||
missing_urls,
|
||||
).fetchall()
|
||||
rows_by_url.update({row["source_url"]: row for row in rows})
|
||||
|
||||
sighting_rows = []
|
||||
for url, finding in by_url.items():
|
||||
row = rows_by_url.get(url)
|
||||
if row is None:
|
||||
continue
|
||||
sighting_rows.append((
|
||||
row["id"],
|
||||
run_id,
|
||||
topic_id,
|
||||
finding.get("source", "unknown"),
|
||||
url,
|
||||
finding.get("source_title") or finding.get("title", ""),
|
||||
finding.get("engagement_score", 0),
|
||||
finding.get("relevance_score", 0),
|
||||
))
|
||||
|
||||
if not sighting_rows:
|
||||
return
|
||||
|
||||
conn.executemany(
|
||||
"""INSERT INTO finding_sightings
|
||||
(finding_id, run_id, topic_id, source, source_url, source_title,
|
||||
engagement_score, relevance_score)
|
||||
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT(run_id, finding_id) DO UPDATE SET
|
||||
topic_id = excluded.topic_id,
|
||||
source = excluded.source,
|
||||
source_url = excluded.source_url,
|
||||
source_title = excluded.source_title,
|
||||
engagement_score = excluded.engagement_score,
|
||||
relevance_score = excluded.relevance_score""",
|
||||
sighting_rows,
|
||||
)
|
||||
|
||||
|
||||
def get_sightings_for_run(topic_id: int, run_id: int) -> List[Dict[str, Any]]:
|
||||
"""Return findings observed for a topic during a specific run."""
|
||||
conn = _connect()
|
||||
try:
|
||||
rows = conn.execute(
|
||||
"""SELECT * FROM finding_sightings
|
||||
WHERE topic_id = ? AND run_id = ?
|
||||
ORDER BY id""",
|
||||
(topic_id, run_id),
|
||||
).fetchall()
|
||||
return [dict(r) for r in rows]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def get_new_findings(
|
||||
topic_id: int,
|
||||
since: Optional[str] = None,
|
||||
@@ -519,7 +621,7 @@ def get_daily_cost(date: Optional[str] = None) -> float:
|
||||
conn = _connect()
|
||||
try:
|
||||
if not date:
|
||||
date = datetime.now().strftime("%Y-%m-%d")
|
||||
date = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
row = conn.execute(
|
||||
"""SELECT COALESCE(SUM(token_cost), 0) as total
|
||||
FROM research_runs
|
||||
@@ -575,7 +677,7 @@ def get_stats() -> Dict[str, Any]:
|
||||
topic_count = conn.execute("SELECT COUNT(*) FROM topics WHERE enabled = 1").fetchone()[0]
|
||||
finding_count = conn.execute("SELECT COUNT(*) FROM findings").fetchone()[0]
|
||||
|
||||
week_ago = (datetime.now() - timedelta(days=7)).strftime("%Y-%m-%d")
|
||||
week_ago = (datetime.now(timezone.utc) - timedelta(days=7)).strftime("%Y-%m-%d")
|
||||
runs_7d = conn.execute(
|
||||
"SELECT COUNT(*) FROM research_runs WHERE run_date >= ?", (week_ago,)
|
||||
).fetchone()[0]
|
||||
@@ -621,7 +723,7 @@ def get_trending(days: int = 7) -> List[Dict[str, Any]]:
|
||||
"""Get topics ranked by recent finding activity."""
|
||||
conn = _connect()
|
||||
try:
|
||||
since = (datetime.now() - timedelta(days=days)).strftime("%Y-%m-%d")
|
||||
since = (datetime.now(timezone.utc) - timedelta(days=days)).strftime("%Y-%m-%d")
|
||||
rows = conn.execute(
|
||||
"""SELECT t.name, t.id,
|
||||
COUNT(f.id) as new_findings,
|
||||
@@ -673,27 +775,31 @@ def findings_from_report(
|
||||
limit: Optional[int] = None,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Convert report into persisted findings.
|
||||
|
||||
|
||||
Uses ranked candidates (post-rerank) when available for quality scores and explanations.
|
||||
Supplements with raw items from items_by_source for HN/PM that didn't rank highly
|
||||
but are valuable for watchlist persistence.
|
||||
but are valuable for watchlist persistence. When ranked_candidates is empty
|
||||
(degraded path — rerank failed or was skipped), falls back to supplementing
|
||||
all sources from items_by_source so findings aren't silently dropped.
|
||||
"""
|
||||
findings = []
|
||||
seen_urls = set()
|
||||
|
||||
# Phase 1: Process ranked candidates (high-quality data with explanations and corroboration)
|
||||
|
||||
for candidate in report.ranked_candidates:
|
||||
finding = finding_from_candidate(candidate)
|
||||
findings.append(finding)
|
||||
findings.append(finding_from_candidate(candidate))
|
||||
seen_urls.add(candidate.url)
|
||||
|
||||
# Phase 2: Add HN/PM items not already captured in ranked candidates
|
||||
for source_name in ["hackernews", "polymarket"]:
|
||||
|
||||
supplement_sources = (
|
||||
list(report.items_by_source)
|
||||
if not report.ranked_candidates
|
||||
else ["hackernews", "polymarket"]
|
||||
)
|
||||
for source_name in supplement_sources:
|
||||
if source_name not in report.items_by_source:
|
||||
continue
|
||||
for item in report.items_by_source[source_name]:
|
||||
if item.url in seen_urls:
|
||||
continue # Already captured with rich data
|
||||
continue
|
||||
findings.append({
|
||||
"source": source_name,
|
||||
"source_url": item.url,
|
||||
@@ -705,8 +811,7 @@ def findings_from_report(
|
||||
"relevance_score": item.local_relevance or 0.5,
|
||||
})
|
||||
seen_urls.add(item.url)
|
||||
|
||||
# Apply global limit after collecting all findings (fix: was per-source, now global)
|
||||
|
||||
return findings[:limit] if limit is not None else findings
|
||||
|
||||
|
||||
@@ -722,9 +827,10 @@ def _cli_query(args):
|
||||
|
||||
since = None
|
||||
if args.since:
|
||||
# Parse duration like "7d", "30d"
|
||||
# Parse duration like "7d", "30d". Use UTC to match SQLite's
|
||||
# datetime('now') which writes first_seen in UTC.
|
||||
days = int(args.since.rstrip("d"))
|
||||
since = (datetime.now() - timedelta(days=days)).strftime("%Y-%m-%d")
|
||||
since = (datetime.now(timezone.utc) - timedelta(days=days)).strftime("%Y-%m-%d")
|
||||
|
||||
findings = get_new_findings(topic["id"], since)
|
||||
print(json.dumps({"topic": topic["name"], "findings": findings, "count": len(findings)}, default=str))
|
||||
|
||||
@@ -1,123 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# sync.sh - Deploy last30days skill to all host locations
|
||||
# Usage: bash skills/last30days/scripts/sync.sh (run from repo root)
|
||||
set -euo pipefail
|
||||
|
||||
SRC="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
echo "Source: $SRC"
|
||||
|
||||
COMMON_TARGETS=(
|
||||
# Claude Code plugin cache: marketplace installs overwrite on update,
|
||||
# but local development needs the cache kept in sync with the repo.
|
||||
# Do NOT add ~/.claude/skills/last30days - it creates a duplicate
|
||||
# /last30days-3 in the slash command menu alongside the plugin version.
|
||||
"$HOME/.claude/plugins/cache/last30days-skill-private/last30days-3/3.2.0"
|
||||
"$HOME/.claude/plugins/cache/last30days-skill-private/last30days-3-nogem/3.0.0-nogem"
|
||||
"$HOME/.agents/skills/last30days"
|
||||
"$HOME/.codex/skills/last30days"
|
||||
)
|
||||
OPENCLAW_TARGET="$HOME/.openclaw/skills/last30days"
|
||||
|
||||
sync_target() {
|
||||
local target="$1"
|
||||
local skill_md="$2"
|
||||
|
||||
echo ""
|
||||
echo "--- Syncing to $target ---"
|
||||
mkdir -p "$target/scripts/lib"
|
||||
|
||||
cp "$skill_md" "$target/SKILL.md"
|
||||
|
||||
rsync -a \
|
||||
"$SRC/scripts/last30days.py" \
|
||||
"$SRC/scripts/watchlist.py" \
|
||||
"$SRC/scripts/briefing.py" \
|
||||
"$SRC/scripts/store.py" \
|
||||
"$target/scripts/"
|
||||
rsync -a "$SRC/scripts/lib/"*.py "$target/scripts/lib/"
|
||||
|
||||
# The OpenClaw variant lives in the private repo only. Skip cleanly when
|
||||
# running this script from the public repo where variants/open does not exist.
|
||||
if [ -d "$SRC/variants/open" ]; then
|
||||
mkdir -p "$target/variants/open/references"
|
||||
rsync -a "$SRC/variants/open/" "$target/variants/open/"
|
||||
fi
|
||||
|
||||
if [ -d "$SRC/scripts/lib/vendor" ]; then
|
||||
rsync -a "$SRC/scripts/lib/vendor" "$target/scripts/lib/"
|
||||
fi
|
||||
|
||||
if [ -d "$SRC/fixtures" ]; then
|
||||
mkdir -p "$target/fixtures"
|
||||
rsync -a "$SRC/fixtures/" "$target/fixtures/"
|
||||
fi
|
||||
|
||||
mod_count=$(ls "$target/scripts/lib/"*.py 2>/dev/null | wc -l | tr -d ' ')
|
||||
echo " Copied $mod_count modules"
|
||||
|
||||
if (
|
||||
cd "$target/scripts" &&
|
||||
python3 -c "import briefing, store, watchlist; from lib import youtube_yt, bird_x, render, ui; print(' Import check: OK')"
|
||||
); then
|
||||
true
|
||||
else
|
||||
echo " Import check FAILED"
|
||||
fi
|
||||
}
|
||||
|
||||
for t in "${COMMON_TARGETS[@]}"; do
|
||||
sync_target "$t" "$SRC/SKILL.md"
|
||||
done
|
||||
|
||||
# Hermes sync: deploy to Hermes skills directory if it exists
|
||||
HERMES_TARGET="$HOME/.hermes/skills/research/last30days"
|
||||
if [ -d "$HOME/.hermes/skills/research" ]; then
|
||||
echo ""
|
||||
echo "--- Syncing to Hermes ---"
|
||||
mkdir -p "$HERMES_TARGET/scripts/lib"
|
||||
|
||||
cp "$SRC/SKILL.md" "$HERMES_TARGET/SKILL.md"
|
||||
|
||||
rsync -a \
|
||||
"$SRC/scripts/last30days.py" \
|
||||
"$SRC/scripts/watchlist.py" \
|
||||
"$SRC/scripts/briefing.py" \
|
||||
"$SRC/scripts/store.py" \
|
||||
"$HERMES_TARGET/scripts/"
|
||||
rsync -a "$SRC/scripts/lib/"*.py "$HERMES_TARGET/scripts/lib/"
|
||||
|
||||
if [ -d "$SRC/scripts/lib/vendor" ]; then
|
||||
rsync -a "$SRC/scripts/lib/vendor" "$HERMES_TARGET/scripts/lib/"
|
||||
fi
|
||||
|
||||
if [ -d "$SRC/fixtures" ]; then
|
||||
mkdir -p "$HERMES_TARGET/fixtures"
|
||||
rsync -a "$SRC/fixtures/" "$HERMES_TARGET/fixtures/"
|
||||
fi
|
||||
|
||||
mod_count=$(ls "$HERMES_TARGET/scripts/lib/"*.py 2>/dev/null | wc -l | tr -d ' ')
|
||||
echo " Copied $mod_count modules to Hermes"
|
||||
|
||||
if (
|
||||
cd "$HERMES_TARGET/scripts" &&
|
||||
python3 -c "import briefing, store, watchlist; from lib import youtube_yt, bird_x, render, ui; print(' Import check: OK')"
|
||||
); then
|
||||
true
|
||||
else
|
||||
echo " Import check FAILED"
|
||||
fi
|
||||
fi
|
||||
|
||||
# OpenClaw sync only runs when the private-repo OpenClaw variant is present
|
||||
# in the source tree. The public repo does not ship variants/open (the variant
|
||||
# is sanitized via strip_for_openclaw.py and published separately from
|
||||
# last30days-skill-private).
|
||||
if [ -d "$SRC/variants/open" ]; then
|
||||
sync_target "$OPENCLAW_TARGET" "$SRC/variants/open/SKILL.md"
|
||||
else
|
||||
echo ""
|
||||
echo "Skipping OpenClaw target (no variants/open in this repo)"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Sync complete."
|
||||
@@ -10,16 +10,11 @@ import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
import requests
|
||||
except ImportError:
|
||||
requests = None
|
||||
|
||||
SCRIPT_DIR = Path(__file__).parent.resolve()
|
||||
sys.path.insert(0, str(SCRIPT_DIR))
|
||||
|
||||
import store
|
||||
from lib import schema
|
||||
from lib import http, schema
|
||||
|
||||
|
||||
# --- Webhook Delivery Functions ---
|
||||
@@ -58,34 +53,21 @@ def _format_delivery_message(topic: str, counts: dict, mode: str) -> str:
|
||||
|
||||
def _send_slack_webhook(url: str, text: str) -> None:
|
||||
"""POST to Slack incoming webhook."""
|
||||
if not requests:
|
||||
raise RuntimeError("requests library not available for webhook delivery")
|
||||
|
||||
response = requests.post(
|
||||
url,
|
||||
json={"text": text},
|
||||
headers={"Content-Type": "application/json"},
|
||||
timeout=10,
|
||||
)
|
||||
response.raise_for_status()
|
||||
http.post(url, json_data={"text": text}, timeout=10, retries=1)
|
||||
|
||||
|
||||
def _send_generic_webhook(url: str, text: str) -> None:
|
||||
"""POST JSON payload to generic webhook."""
|
||||
if not requests:
|
||||
raise RuntimeError("requests library not available for webhook delivery")
|
||||
|
||||
response = requests.post(
|
||||
http.post(
|
||||
url,
|
||||
json={
|
||||
json_data={
|
||||
"message": text,
|
||||
"source": "last30days",
|
||||
"timestamp": time.time(),
|
||||
},
|
||||
headers={"Content-Type": "application/json"},
|
||||
timeout=10,
|
||||
retries=1,
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
|
||||
# --- Command Handlers ---
|
||||
|
||||
@@ -232,5 +232,79 @@ class TestVendoredBirdRuntime(unittest.TestCase):
|
||||
self.assertEqual(5, items[0]["engagement"]["likes"])
|
||||
|
||||
|
||||
class TestRunBirdSearchJsonDecodeRetry(unittest.TestCase):
|
||||
"""When bird-search returns non-JSON stdout, retry the subprocess.
|
||||
|
||||
Twitter's edge sometimes serves an HTML anti-bot interstitial in place of
|
||||
JSON. Before this fix, that response made json.loads raise JSONDecodeError
|
||||
and the function returned {"items": []} with no diagnostic — silent-empty
|
||||
against an orchestrator that can't distinguish "Twitter blocked us" from
|
||||
"no tweets matched the query."
|
||||
"""
|
||||
|
||||
def _make_result(self, stdout: str, stderr: str = "", returncode: int = 0):
|
||||
from lib.subproc import SubprocResult
|
||||
return SubprocResult(returncode=returncode, stdout=stdout, stderr=stderr)
|
||||
|
||||
def test_retries_subprocess_on_html_interstitial_then_succeeds(self):
|
||||
"""First subprocess attempt returns HTML; second returns JSON → success."""
|
||||
from unittest import mock
|
||||
from lib import bird_x
|
||||
|
||||
html_interstitial = "<!DOCTYPE html><html><body>Rate limited</body></html>"
|
||||
json_success = '[{"id": "1", "text": "tweet"}]'
|
||||
|
||||
results = [
|
||||
(self._make_result(stdout=html_interstitial), None),
|
||||
(self._make_result(stdout=json_success), None),
|
||||
]
|
||||
|
||||
with mock.patch.object(bird_x, "_invoke_bird_subprocess", side_effect=results), \
|
||||
mock.patch.object(bird_x.time, "sleep") as mock_sleep:
|
||||
response = bird_x._run_bird_search("test", count=10, timeout=30)
|
||||
|
||||
self.assertNotIn("error", response)
|
||||
self.assertEqual(response["items"], [{"id": "1", "text": "tweet"}])
|
||||
# Should have slept between the failed first attempt and the retry.
|
||||
mock_sleep.assert_called_once_with(bird_x.JSON_DECODE_RETRY_DELAY)
|
||||
|
||||
def test_returns_error_after_all_retries_exhausted(self):
|
||||
"""All attempts return HTML → error dict with diagnostic + items=[]."""
|
||||
from unittest import mock
|
||||
from lib import bird_x
|
||||
|
||||
html_interstitial = "<!DOCTYPE html><html>blocked</html>"
|
||||
results = [
|
||||
(self._make_result(stdout=html_interstitial), None),
|
||||
(self._make_result(stdout=html_interstitial), None),
|
||||
]
|
||||
|
||||
with mock.patch.object(bird_x, "_invoke_bird_subprocess", side_effect=results), \
|
||||
mock.patch.object(bird_x.time, "sleep"):
|
||||
response = bird_x._run_bird_search("test", count=10, timeout=30)
|
||||
|
||||
self.assertIn("error", response)
|
||||
self.assertIn("Invalid JSON response", response["error"])
|
||||
# Diagnostic message names the anti-bot interstitial so it's
|
||||
# distinguishable from a genuine no-results case in logs.
|
||||
self.assertIn("anti-bot interstitial", response["error"].lower())
|
||||
self.assertEqual(response["items"], [])
|
||||
|
||||
def test_terminal_subprocess_error_is_not_retried(self):
|
||||
"""Subprocess timeout / spawn failure → terminal error, no retry."""
|
||||
from unittest import mock
|
||||
from lib import bird_x
|
||||
|
||||
timeout_error = {"error": "Search timed out after 30s", "items": []}
|
||||
results = [(None, timeout_error)]
|
||||
|
||||
with mock.patch.object(bird_x, "_invoke_bird_subprocess", side_effect=results), \
|
||||
mock.patch.object(bird_x.time, "sleep") as mock_sleep:
|
||||
response = bird_x._run_bird_search("test", count=10, timeout=30)
|
||||
|
||||
self.assertEqual(response, timeout_error)
|
||||
mock_sleep.assert_not_called()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -27,8 +27,8 @@ class CliV3Tests(unittest.TestCase):
|
||||
generated_at="2026-03-16T00:00:00+00:00",
|
||||
provider_runtime=schema.ProviderRuntime(
|
||||
reasoning_provider="gemini",
|
||||
planner_model="gemini-3.1-flash-lite-preview",
|
||||
rerank_model="gemini-3.1-flash-lite-preview",
|
||||
planner_model="gemini-3.1-flash-lite",
|
||||
rerank_model="gemini-3.1-flash-lite",
|
||||
),
|
||||
query_plan=schema.QueryPlan(
|
||||
intent="comparison",
|
||||
@@ -114,13 +114,13 @@ class CliV3Tests(unittest.TestCase):
|
||||
def test_slugify_and_emit_output_cover_supported_modes(self):
|
||||
report = self.make_report()
|
||||
self.assertEqual("openclaw-vs-nanoclaw", cli.slugify(report.topic))
|
||||
self.assertEqual("last30days v3.0.0 CLI.", cli.__doc__)
|
||||
self.assertEqual("last30days CLI.", cli.__doc__)
|
||||
|
||||
compact = cli.emit_output(report, "compact")
|
||||
json_output = cli.emit_output(report, "json")
|
||||
context = cli.emit_output(report, "context")
|
||||
|
||||
self.assertIn("# last30days v3.0.0", compact)
|
||||
self.assertIn("# last30days v", compact)
|
||||
self.assertIn('"topic": "OpenClaw vs NanoClaw"', json_output)
|
||||
self.assertIsInstance(context, str)
|
||||
|
||||
|
||||
@@ -114,9 +114,10 @@ class TestGetConfigCookieIntegration:
|
||||
@patch("lib.cookie_extract.extract_cookies")
|
||||
@patch("lib.env._find_project_env", return_value=None)
|
||||
@patch("lib.env.load_env_file", return_value={})
|
||||
@patch("lib.env._load_keychain", return_value={})
|
||||
@patch("lib.env.get_openai_auth")
|
||||
def test_get_config_injects_cookies(
|
||||
self, mock_openai, mock_load, mock_proj, mock_extract
|
||||
self, mock_openai, mock_keychain, mock_load, mock_proj, mock_extract
|
||||
):
|
||||
from lib.env import get_config, OpenAIAuth
|
||||
mock_openai.return_value = OpenAIAuth(
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
"""Tests for macOS Keychain credential source in lib/env.py.
|
||||
|
||||
Covers:
|
||||
- non-Darwin returns {}
|
||||
- missing `security` binary returns {}
|
||||
- successful lookups return parsed key/value pairs
|
||||
- subprocess timeout / OSError are swallowed
|
||||
- get_config merges keychain at lowest priority and labels _CONFIG_SOURCE
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "skills" / "last30days" / "scripts"))
|
||||
|
||||
from lib import env # noqa: E402
|
||||
|
||||
SETUP_KEYCHAIN_SH = Path(__file__).resolve().parents[1] / "skills" / "last30days" / "scripts" / "setup-keychain.sh"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _load_keychain unit tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_load_keychain_returns_empty_on_non_darwin():
|
||||
with mock.patch("platform.system", return_value="Linux"):
|
||||
assert env._load_keychain(["XAI_API_KEY"]) == {}
|
||||
|
||||
|
||||
def test_load_keychain_returns_empty_when_security_missing():
|
||||
with mock.patch("platform.system", return_value="Darwin"), \
|
||||
mock.patch("shutil.which", return_value=None):
|
||||
assert env._load_keychain(["XAI_API_KEY"]) == {}
|
||||
|
||||
|
||||
def _run_result(returncode: int, stdout: str = "") -> subprocess.CompletedProcess:
|
||||
return subprocess.CompletedProcess(args=[], returncode=returncode, stdout=stdout, stderr="")
|
||||
|
||||
|
||||
def test_load_keychain_loads_present_keys_skips_missing():
|
||||
def fake_run(cmd, **kwargs):
|
||||
service = cmd[cmd.index("-s") + 1]
|
||||
if service == "last30days-XAI_API_KEY":
|
||||
return _run_result(0, "xai-abc\n")
|
||||
if service == "last30days-BRAVE_API_KEY":
|
||||
return _run_result(0, "brv-xyz\n")
|
||||
return _run_result(44) # security's "not found" exit code
|
||||
|
||||
with mock.patch("platform.system", return_value="Darwin"), \
|
||||
mock.patch("shutil.which", return_value="/usr/bin/security"), \
|
||||
mock.patch("subprocess.run", side_effect=fake_run):
|
||||
result = env._load_keychain(["XAI_API_KEY", "BRAVE_API_KEY", "OPENAI_API_KEY"])
|
||||
|
||||
assert result == {"XAI_API_KEY": "xai-abc", "BRAVE_API_KEY": "brv-xyz"}
|
||||
|
||||
|
||||
def test_load_keychain_strips_whitespace_and_newlines():
|
||||
with mock.patch("platform.system", return_value="Darwin"), \
|
||||
mock.patch("shutil.which", return_value="/usr/bin/security"), \
|
||||
mock.patch("subprocess.run", return_value=_run_result(0, " hello-key \n")):
|
||||
result = env._load_keychain(["FOO"])
|
||||
assert result == {"FOO": "hello-key"}
|
||||
|
||||
|
||||
def test_load_keychain_swallows_subprocess_errors():
|
||||
def fake_run(cmd, **kwargs):
|
||||
raise subprocess.TimeoutExpired(cmd=cmd, timeout=5)
|
||||
|
||||
with mock.patch("platform.system", return_value="Darwin"), \
|
||||
mock.patch("shutil.which", return_value="/usr/bin/security"), \
|
||||
mock.patch("subprocess.run", side_effect=fake_run):
|
||||
assert env._load_keychain(["XAI_API_KEY"]) == {}
|
||||
|
||||
|
||||
def test_load_keychain_swallows_oserror():
|
||||
with mock.patch("platform.system", return_value="Darwin"), \
|
||||
mock.patch("shutil.which", return_value="/usr/bin/security"), \
|
||||
mock.patch("subprocess.run", side_effect=OSError("boom")):
|
||||
assert env._load_keychain(["XAI_API_KEY"]) == {}
|
||||
|
||||
|
||||
def test_load_keychain_skips_empty_stdout():
|
||||
with mock.patch("platform.system", return_value="Darwin"), \
|
||||
mock.patch("shutil.which", return_value="/usr/bin/security"), \
|
||||
mock.patch("subprocess.run", return_value=_run_result(0, "")):
|
||||
assert env._load_keychain(["XAI_API_KEY"]) == {}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# get_config integration tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def clean_env(monkeypatch, tmp_path):
|
||||
"""Hide every key get_config might touch and point CONFIG_FILE at a
|
||||
non-existent path so no real user config bleeds in."""
|
||||
for var in [
|
||||
"OPENAI_API_KEY", "XAI_API_KEY", "BRAVE_API_KEY", "AUTH_TOKEN", "CT0",
|
||||
"SCRAPECREATORS_API_KEY", "APIFY_API_TOKEN", "BSKY_HANDLE",
|
||||
"BSKY_APP_PASSWORD", "TRUTHSOCIAL_TOKEN", "EXA_API_KEY",
|
||||
"SERPER_API_KEY", "OPENROUTER_API_KEY", "PARALLEL_API_KEY",
|
||||
"XQUIK_API_KEY", "GOOGLE_API_KEY", "GEMINI_API_KEY",
|
||||
"GOOGLE_GENAI_API_KEY", "INCLUDE_SOURCES", "FROM_BROWSER",
|
||||
]:
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
monkeypatch.setattr(env, "CONFIG_FILE", tmp_path / "does-not-exist.env")
|
||||
monkeypatch.chdir(tmp_path) # no project .env in this tree either
|
||||
|
||||
|
||||
def test_get_config_reports_keychain_source(clean_env):
|
||||
with mock.patch.object(env, "_load_keychain", return_value={"XAI_API_KEY": "xai-from-kc"}):
|
||||
cfg = env.get_config()
|
||||
assert cfg["_CONFIG_SOURCE"] == "keychain"
|
||||
assert cfg["XAI_API_KEY"] == "xai-from-kc"
|
||||
|
||||
|
||||
def test_get_config_env_var_overrides_keychain(clean_env, monkeypatch):
|
||||
monkeypatch.setenv("XAI_API_KEY", "xai-from-env")
|
||||
with mock.patch.object(env, "_load_keychain", return_value={"XAI_API_KEY": "xai-from-kc"}):
|
||||
cfg = env.get_config()
|
||||
assert cfg["XAI_API_KEY"] == "xai-from-env"
|
||||
|
||||
|
||||
def test_get_config_reports_env_only_when_keychain_empty(clean_env):
|
||||
with mock.patch.object(env, "_load_keychain", return_value={}):
|
||||
cfg = env.get_config()
|
||||
assert cfg["_CONFIG_SOURCE"] == "env_only"
|
||||
|
||||
|
||||
def test_get_config_global_file_outranks_keychain(clean_env, tmp_path, monkeypatch):
|
||||
cfg_file = tmp_path / "global.env"
|
||||
cfg_file.write_text("XAI_API_KEY=xai-from-file\n")
|
||||
monkeypatch.setattr(env, "CONFIG_FILE", cfg_file)
|
||||
with mock.patch.object(env, "_load_keychain", return_value={"XAI_API_KEY": "xai-from-kc"}):
|
||||
cfg = env.get_config()
|
||||
assert cfg["XAI_API_KEY"] == "xai-from-file"
|
||||
assert cfg["_CONFIG_SOURCE"].startswith("global:")
|
||||
|
||||
|
||||
def test_get_config_openai_key_can_come_from_keychain(clean_env):
|
||||
"""OPENAI_API_KEY must be visible to get_openai_auth via the keychain
|
||||
merge — wiring regression test."""
|
||||
with mock.patch.object(env, "_load_keychain", return_value={"OPENAI_API_KEY": "sk-from-kc"}):
|
||||
cfg = env.get_config()
|
||||
assert cfg["OPENAI_API_KEY"] == "sk-from-kc"
|
||||
assert cfg["OPENAI_AUTH_SOURCE"] == "api_key"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Drift guard: lib/env.py KEYCHAIN_KEYS and setup-keychain.sh ALL_KEYS must
|
||||
# stay in lockstep. A mismatch means users storing a key via the helper script
|
||||
# wouldn't see it picked up by the loader, or vice versa.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _parse_all_keys_from_shell(script: Path) -> list[str]:
|
||||
text = script.read_text(encoding="utf-8")
|
||||
match = re.search(r"ALL_KEYS=\(\s*(.*?)\s*\)", text, re.DOTALL)
|
||||
if not match:
|
||||
raise AssertionError(f"ALL_KEYS=( ... ) array not found in {script}")
|
||||
body = match.group(1)
|
||||
# Strip shell comments and split on whitespace
|
||||
body = re.sub(r"#[^\n]*", "", body)
|
||||
return [tok for tok in body.split() if tok]
|
||||
|
||||
|
||||
def test_keychain_keys_match_setup_script():
|
||||
shell_keys = _parse_all_keys_from_shell(SETUP_KEYCHAIN_SH)
|
||||
python_keys = list(env.KEYCHAIN_KEYS)
|
||||
assert shell_keys == python_keys, (
|
||||
"lib/env.py::KEYCHAIN_KEYS and scripts/setup-keychain.sh::ALL_KEYS "
|
||||
f"have drifted.\n python: {python_keys}\n shell: {shell_keys}"
|
||||
)
|
||||
@@ -41,6 +41,34 @@ class EnvV3Tests(unittest.TestCase):
|
||||
with mock.patch.dict(os.environ, {}, clear=False):
|
||||
self.assertIsNone(bird_x.is_bird_authenticated())
|
||||
|
||||
def test_file_permission_check_skips_windows_posix_mode_bits(self):
|
||||
path = mock.Mock(spec=Path)
|
||||
with mock.patch.object(env.os, "name", "nt"), mock.patch.object(env.sys.stderr, "write") as write:
|
||||
env._check_file_permissions(path)
|
||||
|
||||
path.stat.assert_not_called()
|
||||
write.assert_not_called()
|
||||
|
||||
|
||||
class ThreadsAvailabilityTests(unittest.TestCase):
|
||||
"""Threads is in the SC default-on family: same key, same per-call cost
|
||||
shape as TikTok / Instagram, so the same default-on rule applies.
|
||||
Suppression goes through EXCLUDE_SOURCES, not gated opt-in."""
|
||||
|
||||
def test_threads_available_with_sc_key_only(self):
|
||||
self.assertTrue(env.is_threads_available({"SCRAPECREATORS_API_KEY": "k"}))
|
||||
|
||||
def test_threads_unavailable_without_sc_key(self):
|
||||
self.assertFalse(env.is_threads_available({}))
|
||||
self.assertFalse(env.is_threads_available({"INCLUDE_SOURCES": "threads"}))
|
||||
|
||||
def test_threads_does_not_require_include_sources(self):
|
||||
"""Regression guard: INCLUDE_SOURCES should not be needed."""
|
||||
self.assertTrue(env.is_threads_available({
|
||||
"SCRAPECREATORS_API_KEY": "k",
|
||||
"INCLUDE_SOURCES": "",
|
||||
}))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -125,7 +125,7 @@ class EvaluatorV3Tests(unittest.TestCase):
|
||||
topic="test topic",
|
||||
query_type="general",
|
||||
items=[{"key": "a"}],
|
||||
judge_model="gemini-3.1-flash-lite-preview",
|
||||
judge_model="gemini-3.1-flash-lite",
|
||||
gemini_api_key="key",
|
||||
)
|
||||
self.assertEqual({"a": 3}, cached)
|
||||
@@ -136,7 +136,7 @@ class EvaluatorV3Tests(unittest.TestCase):
|
||||
topic="test topic",
|
||||
query_type="general",
|
||||
items=[],
|
||||
judge_model="gemini-3.1-flash-lite-preview",
|
||||
judge_model="gemini-3.1-flash-lite",
|
||||
gemini_api_key=None,
|
||||
)
|
||||
self.assertEqual({}, skipped)
|
||||
|
||||
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
@@ -27,13 +28,31 @@ class FooterNudgeSuppressionTests(unittest.TestCase):
|
||||
"--emit=md",
|
||||
*argv,
|
||||
]
|
||||
env = {**os.environ, "LAST30DAYS_SKIP_PREFLIGHT": "1"}
|
||||
env = {
|
||||
**os.environ,
|
||||
"LAST30DAYS_SKIP_PREFLIGHT": "1",
|
||||
# Skip ~/.config/last30days/.env so a contributor's saved
|
||||
# BRAVE/EXA/SERPER/PARALLEL key doesn't make grounding "available"
|
||||
# and suppress the promo we're checking for.
|
||||
"LAST30DAYS_CONFIG_DIR": "",
|
||||
# Pin X as available so _missing_sources_for_promo selects "web"
|
||||
# (otherwise the "x" promo wins and the BRAVE_API_KEY string never
|
||||
# appears).
|
||||
"XAI_API_KEY": "test-stub",
|
||||
}
|
||||
# Strip any grounded-web keys the host might have so the promo path
|
||||
# triggers deterministically in mock + no-backend.
|
||||
# triggers deterministically in mock + no-backend. Also strip X cookie
|
||||
# credentials so XAI_API_KEY is the unambiguous X backend.
|
||||
for key in ("BRAVE_API_KEY", "EXA_API_KEY", "SERPER_API_KEY",
|
||||
"PARALLEL_API_KEY", "OPENROUTER_API_KEY"):
|
||||
"PARALLEL_API_KEY", "OPENROUTER_API_KEY",
|
||||
"AUTH_TOKEN", "CT0", "LAST30DAYS_X_BACKEND"):
|
||||
env.pop(key, None)
|
||||
return subprocess.run(cmd, capture_output=True, text=True, env=env)
|
||||
# Run from a tmpdir so _find_project_env() can't walk up into any
|
||||
# .claude/last30days.env above the repo on the contributor's machine.
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
return subprocess.run(
|
||||
cmd, capture_output=True, text=True, env=env, cwd=tmp,
|
||||
)
|
||||
|
||||
def test_bare_run_emits_web_promo(self):
|
||||
result = self._run(topic="OpenAI")
|
||||
|
||||
@@ -190,5 +190,88 @@ class WebSearchDispatchTests(unittest.TestCase):
|
||||
grounding.web_search("test", ("2026-02-25", "2026-03-27"), {}, backend="google")
|
||||
|
||||
|
||||
class RedditEnrichmentGateTests(unittest.TestCase):
|
||||
"""EXCLUDE_SOURCES=reddit must suppress the web-search Reddit enrichment.
|
||||
|
||||
Otherwise a user who explicitly excluded Reddit would still get Reddit
|
||||
content smuggled back in via web-search URLs that happen to point at
|
||||
reddit.com threads.
|
||||
"""
|
||||
|
||||
def test_reddit_excluded_via_exclude_sources_skips_enrichment(self):
|
||||
config = {"BRAVE_API_KEY": "k", "EXCLUDE_SOURCES": "reddit"}
|
||||
items = [{"url": "https://www.reddit.com/r/python/comments/abc/title/", "snippet": "original"}]
|
||||
with patch("lib.grounding.brave_search", return_value=(items, {})), \
|
||||
patch("lib.grounding._enrich_reddit_items") as enrich_mock:
|
||||
grounding.web_search("test", ("2026-02-25", "2026-03-27"), config, backend="auto")
|
||||
enrich_mock.assert_not_called()
|
||||
|
||||
def test_reddit_excluded_case_insensitive(self):
|
||||
for value in ("REDDIT", "Reddit", " reddit ", "x,reddit,y"):
|
||||
config = {"BRAVE_API_KEY": "k", "EXCLUDE_SOURCES": value}
|
||||
self.assertTrue(
|
||||
grounding._reddit_excluded(config),
|
||||
msg=f"_reddit_excluded should be True for EXCLUDE_SOURCES={value!r}",
|
||||
)
|
||||
|
||||
def test_reddit_not_excluded_when_other_sources_listed(self):
|
||||
config = {"EXCLUDE_SOURCES": "tiktok,instagram"}
|
||||
self.assertFalse(grounding._reddit_excluded(config))
|
||||
|
||||
def test_enrichment_runs_when_reddit_not_excluded(self):
|
||||
config = {"BRAVE_API_KEY": "k"}
|
||||
items = [{"url": "https://www.reddit.com/r/python/comments/abc/title/", "snippet": "original"}]
|
||||
with patch("lib.grounding.brave_search", return_value=(items, {})), \
|
||||
patch("lib.grounding._enrich_reddit_items", return_value=items) as enrich_mock:
|
||||
grounding.web_search("test", ("2026-02-25", "2026-03-27"), config, backend="auto")
|
||||
enrich_mock.assert_called_once()
|
||||
|
||||
|
||||
class RedditEnrichItemsTests(unittest.TestCase):
|
||||
"""Direct tests for `_enrich_reddit_items` covering the selftext key path
|
||||
and the RedditRateLimitError early-exit behavior.
|
||||
"""
|
||||
|
||||
def test_selftext_under_submission_populates_snippet(self):
|
||||
from lib import reddit_enrich
|
||||
|
||||
item = {
|
||||
"url": "https://www.reddit.com/r/python/comments/abc/title/",
|
||||
"snippet": "original",
|
||||
}
|
||||
parsed = {
|
||||
"submission": {"selftext": "thread body content"},
|
||||
"comments": [],
|
||||
}
|
||||
with patch.object(reddit_enrich, "fetch_thread_data", return_value={"raw": True}), \
|
||||
patch.object(reddit_enrich, "parse_thread_data", return_value=parsed):
|
||||
result = grounding._enrich_reddit_items([item])
|
||||
self.assertEqual("thread body content", result[0]["snippet"])
|
||||
self.assertEqual("reddit_json_api", result[0]["enriched_via"])
|
||||
|
||||
def test_rate_limit_error_halts_iteration(self):
|
||||
from lib import reddit_enrich
|
||||
|
||||
item1 = {"url": "https://www.reddit.com/r/python/comments/aaa/x/"}
|
||||
item2 = {"url": "https://www.reddit.com/r/python/comments/bbb/y/"}
|
||||
|
||||
def fake_fetch(url, *args, **kwargs):
|
||||
raise reddit_enrich.RedditRateLimitError(f"429 for {url}")
|
||||
|
||||
captured_stderr: list[str] = []
|
||||
|
||||
with patch.object(reddit_enrich, "fetch_thread_data", side_effect=fake_fetch) as fetch_mock, \
|
||||
patch("lib.grounding.sys.stderr.write", side_effect=lambda s: captured_stderr.append(s)):
|
||||
grounding._enrich_reddit_items([item1, item2])
|
||||
|
||||
# Only the first item should have triggered a fetch attempt
|
||||
self.assertEqual(1, fetch_mock.call_count)
|
||||
# A stderr message about the rate-limit halt should have been emitted
|
||||
self.assertTrue(
|
||||
any("rate-limited" in msg.lower() or "rate limited" in msg.lower() for msg in captured_stderr),
|
||||
msg=f"Expected a rate-limit stderr message, got: {captured_stderr!r}",
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -160,12 +160,44 @@ def test_title_matches_query_empty_query():
|
||||
|
||||
|
||||
def test_title_matches_query_partial_match():
|
||||
"""Test that all query words must match."""
|
||||
"""Any-word matching: at least one query token in title is enough.
|
||||
|
||||
Previously required *all* tokens, which killed every hit on multi-keyword
|
||||
theme queries like 'claude, personal agents, agentic infra' since no real
|
||||
HN title contains all 5 tokens verbatim. Token-overlap relevance at parse
|
||||
time still demotes weak matches, so the loosened gate is safe.
|
||||
"""
|
||||
title = "New AI framework"
|
||||
query = "AI blockchain"
|
||||
|
||||
# "blockchain" is not in title, so should fail
|
||||
assert hackernews._title_matches_query(title, query) is False
|
||||
|
||||
# "AI" matches as a whole word, even though "blockchain" doesn't appear
|
||||
assert hackernews._title_matches_query(title, query) is True
|
||||
|
||||
|
||||
def test_title_matches_query_no_token_in_title():
|
||||
"""If no query token appears in the title at all, reject."""
|
||||
assert hackernews._title_matches_query("New rust compiler", "AI blockchain") is False
|
||||
|
||||
|
||||
def test_title_matches_query_word_boundary_not_substring():
|
||||
"""Short tokens must match on word boundaries, not as substrings.
|
||||
|
||||
Without word-boundary matching, 'ai' would falsely match 'email',
|
||||
'rail', 'artists', etc.
|
||||
"""
|
||||
# 'ai' as a substring of 'email' must not match
|
||||
assert hackernews._title_matches_query("New email service", "ai blockchain") is False
|
||||
# 'ai' as a whole word does match
|
||||
assert hackernews._title_matches_query("Cool AI tool launched", "ai blockchain") is True
|
||||
|
||||
|
||||
def test_title_matches_query_flattens_hyphens_and_commas():
|
||||
"""Query tokens split on hyphens/commas the same way search_hackernews
|
||||
flattens them, so the post-filter stays aligned with what Algolia saw."""
|
||||
# query 'ts-bun-node' flattens to ['ts', 'bun', 'node']; title contains 'bun'
|
||||
assert hackernews._title_matches_query("Bun 1.2 released", "ts-bun-node") is True
|
||||
# query 'rust, go, zig' flattens; title contains 'go'
|
||||
assert hackernews._title_matches_query("Go 1.24 generics update", "rust, go, zig") is True
|
||||
|
||||
|
||||
# === Tests for search_hackernews() ===
|
||||
|
||||
@@ -277,6 +277,26 @@ class HtmlCliIntegrationTests(unittest.TestCase):
|
||||
path = cli.compute_save_path_display("/tmp", report.topic, "v3", "html")
|
||||
self.assertTrue(path.endswith("/ai-agent-frameworks-raw-html-v3.html"))
|
||||
|
||||
def test_save_output_can_persist_comparison_html(self):
|
||||
reports = [
|
||||
("OpenClaw", _report("OpenClaw", ["Containers"])),
|
||||
("Hermes", _report("Hermes", ["Memory"])),
|
||||
]
|
||||
rendered = cli.emit_comparison_output(reports, "html")
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
path = cli.save_output(
|
||||
reports[0][1],
|
||||
"html",
|
||||
tmpdir,
|
||||
topic_override=cli.comparison_topic(reports),
|
||||
rendered_content=rendered,
|
||||
)
|
||||
self.assertEqual("openclaw-vs-hermes-raw-html.html", path.name)
|
||||
saved = path.read_text(encoding="utf-8")
|
||||
self.assertIn("last30days · OpenClaw vs Hermes", saved)
|
||||
self.assertIn("comparing 2: OpenClaw, Hermes", saved)
|
||||
self.assertNotIn("last30days · OpenClaw</title>", saved)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -104,3 +104,117 @@ class TestParamsEncoding(unittest.TestCase):
|
||||
sent_url = self._sent_url(mock_urlopen)
|
||||
self.assertIn("count=25", sent_url)
|
||||
self.assertIn("raw=True", sent_url)
|
||||
|
||||
|
||||
class TestDNSResolutionRetry(unittest.TestCase):
|
||||
"""DNS resolution failures (gaierror) must retry with exponential backoff.
|
||||
|
||||
Caller-passed `retries` values smaller than MIN_DNS_RETRIES are expanded
|
||||
on the first gaierror so a transient resolution failure doesn't wipe a
|
||||
request just because the caller passed retries=2.
|
||||
"""
|
||||
|
||||
@patch("lib.http.urllib.request.urlopen")
|
||||
@patch("lib.http.time.sleep")
|
||||
def test_gaierror_retries_up_to_min_dns_retries_even_when_caller_passes_fewer(
|
||||
self, mock_sleep, mock_urlopen
|
||||
):
|
||||
"""Caller passed retries=2; gaierror should still get MIN_DNS_RETRIES attempts."""
|
||||
import socket
|
||||
err = urllib.error.URLError(socket.gaierror(-2, "Name or service not known"))
|
||||
mock_urlopen.side_effect = err
|
||||
|
||||
with self.assertRaises(http.HTTPError):
|
||||
http.request("GET", "http://nonexistent.example", retries=2)
|
||||
|
||||
# Caller passed retries=2, but the budget expanded to MIN_DNS_RETRIES=3.
|
||||
self.assertEqual(mock_urlopen.call_count, http.MIN_DNS_RETRIES)
|
||||
|
||||
@patch("lib.http.urllib.request.urlopen")
|
||||
@patch("lib.http.time.sleep")
|
||||
def test_gaierror_succeeds_after_transient_failure(self, mock_sleep, mock_urlopen):
|
||||
"""gaierror on attempt 1, then success — should NOT raise."""
|
||||
import socket
|
||||
success_response = MagicMock()
|
||||
success_response.read.return_value = b'{"ok": true}'
|
||||
success_response.status = 200
|
||||
success_response.__enter__ = lambda self: self
|
||||
success_response.__exit__ = lambda *args: None
|
||||
|
||||
err = urllib.error.URLError(socket.gaierror(-2, "Name or service not known"))
|
||||
mock_urlopen.side_effect = [err, success_response]
|
||||
|
||||
result = http.request("GET", "http://flaky.example", retries=2)
|
||||
|
||||
self.assertEqual(result, {"ok": True})
|
||||
self.assertEqual(mock_urlopen.call_count, 2)
|
||||
|
||||
@patch("lib.http.urllib.request.urlopen")
|
||||
@patch("lib.http.time.sleep")
|
||||
def test_gaierror_uses_exponential_backoff(self, mock_sleep, mock_urlopen):
|
||||
"""Backoff delays for gaierror should be 1s, 2s, 4s — not the linear default."""
|
||||
import socket
|
||||
err = urllib.error.URLError(socket.gaierror(-2, "Name or service not known"))
|
||||
mock_urlopen.side_effect = err
|
||||
|
||||
with self.assertRaises(http.HTTPError):
|
||||
http.request("GET", "http://nonexistent.example", retries=3)
|
||||
|
||||
# Expected sleep calls: 1s (after attempt 1), 2s (after attempt 2).
|
||||
# No sleep after the final attempt (the loop exits to raise).
|
||||
sleep_delays = [call.args[0] for call in mock_sleep.call_args_list]
|
||||
self.assertEqual(sleep_delays, [1, 2])
|
||||
|
||||
@patch("lib.http.urllib.request.urlopen")
|
||||
@patch("lib.http.time.sleep")
|
||||
def test_non_dns_urlerror_uses_linear_backoff_not_dns_branch(
|
||||
self, mock_sleep, mock_urlopen
|
||||
):
|
||||
"""A URLError that's NOT a gaierror must NOT expand the retry budget."""
|
||||
# ConnectionRefusedError-style URLError reason (not gaierror)
|
||||
err = urllib.error.URLError(ConnectionRefusedError(111, "Connection refused"))
|
||||
mock_urlopen.side_effect = err
|
||||
|
||||
with self.assertRaises(http.HTTPError):
|
||||
http.request("GET", "http://refused.example", retries=2)
|
||||
|
||||
# Caller passed retries=2, and non-DNS URLError doesn't expand it.
|
||||
self.assertEqual(mock_urlopen.call_count, 2)
|
||||
|
||||
@patch("lib.http.urllib.request.urlopen")
|
||||
@patch("lib.http.time.sleep")
|
||||
def test_dns_widening_does_not_leak_into_subsequent_non_dns_urlerror(
|
||||
self, mock_sleep, mock_urlopen
|
||||
):
|
||||
"""Mixed sequence: DNS-then-non-DNS must respect caller's original retries.
|
||||
|
||||
Without the fix, the first gaierror widens effective_retries from 2 to
|
||||
MIN_DNS_RETRIES=3, and a subsequent ConnectionRefused on attempt 1
|
||||
slips into a third overall attempt — exceeding what the caller asked
|
||||
for. Each non-DNS error path must gate on the original `retries`.
|
||||
"""
|
||||
import socket
|
||||
dns_err = urllib.error.URLError(socket.gaierror(-2, "Name or service not known"))
|
||||
conn_err = urllib.error.URLError(ConnectionRefusedError(111, "Connection refused"))
|
||||
mock_urlopen.side_effect = [dns_err, conn_err, conn_err] # 3rd would only fire if budget leaked
|
||||
|
||||
with self.assertRaises(http.HTTPError):
|
||||
http.request("GET", "http://flaky.example", retries=2)
|
||||
|
||||
# Caller asked for at most 2 attempts. DNS widening must not give us a 3rd.
|
||||
self.assertEqual(mock_urlopen.call_count, 2)
|
||||
|
||||
@patch("lib.http.urllib.request.urlopen")
|
||||
@patch("lib.http.time.sleep")
|
||||
def test_dns_widening_does_not_leak_into_subsequent_oserror(
|
||||
self, mock_sleep, mock_urlopen
|
||||
):
|
||||
"""Mixed sequence: DNS-then-OSError must respect caller's original retries."""
|
||||
import socket
|
||||
dns_err = urllib.error.URLError(socket.gaierror(-2, "Name or service not known"))
|
||||
mock_urlopen.side_effect = [dns_err, TimeoutError("timed out"), TimeoutError("timed out")]
|
||||
|
||||
with self.assertRaises(http.HTTPError):
|
||||
http.request("GET", "http://flaky.example", retries=2)
|
||||
|
||||
self.assertEqual(mock_urlopen.call_count, 2)
|
||||
|
||||
@@ -904,5 +904,81 @@ class TestZeroKeyPipelineRun(unittest.TestCase):
|
||||
self.assertEqual("fallback-local-score", candidate.explanation)
|
||||
|
||||
|
||||
class TestExcludeSources(unittest.TestCase):
|
||||
"""EXCLUDE_SOURCES env var filters sources out of available_sources().
|
||||
|
||||
The existing INCLUDE_SOURCES allowlist (used by Perplexity opt-in) does
|
||||
not cover this case — tiktok and instagram are added unconditionally
|
||||
when SCRAPECREATORS_API_KEY is set, with no way to opt out short of
|
||||
unsetting the key. EXCLUDE_SOURCES gives runs a per-invocation denylist.
|
||||
"""
|
||||
|
||||
def test_excludes_tiktok_and_instagram(self):
|
||||
config = {
|
||||
"SCRAPECREATORS_API_KEY": "test-key",
|
||||
"EXCLUDE_SOURCES": "tiktok,instagram",
|
||||
}
|
||||
sources = pipeline.available_sources(config)
|
||||
self.assertNotIn("tiktok", sources)
|
||||
self.assertNotIn("instagram", sources)
|
||||
self.assertIn("reddit", sources)
|
||||
self.assertIn("hackernews", sources)
|
||||
|
||||
def test_no_exclusion_when_unset(self):
|
||||
config = {"SCRAPECREATORS_API_KEY": "test-key"}
|
||||
sources = pipeline.available_sources(config)
|
||||
self.assertIn("tiktok", sources)
|
||||
self.assertIn("instagram", sources)
|
||||
|
||||
def test_empty_exclude_sources_is_noop(self):
|
||||
config = {
|
||||
"SCRAPECREATORS_API_KEY": "test-key",
|
||||
"EXCLUDE_SOURCES": "",
|
||||
}
|
||||
sources = pipeline.available_sources(config)
|
||||
self.assertIn("tiktok", sources)
|
||||
self.assertIn("instagram", sources)
|
||||
|
||||
def test_whitespace_and_case_insensitive(self):
|
||||
config = {
|
||||
"SCRAPECREATORS_API_KEY": "test-key",
|
||||
"EXCLUDE_SOURCES": " TikTok , INSTAGRAM ",
|
||||
}
|
||||
sources = pipeline.available_sources(config)
|
||||
self.assertNotIn("tiktok", sources)
|
||||
self.assertNotIn("instagram", sources)
|
||||
|
||||
def test_excludes_non_scrapecreators_source(self):
|
||||
"""EXCLUDE_SOURCES applies to any source, not just SC-backed ones."""
|
||||
config = {"EXCLUDE_SOURCES": "hackernews"}
|
||||
sources = pipeline.available_sources(config)
|
||||
self.assertNotIn("hackernews", sources)
|
||||
self.assertIn("reddit", sources)
|
||||
|
||||
|
||||
class TestExcludeSourcesEndToEnd(unittest.TestCase):
|
||||
"""Wiring regression: EXCLUDE_SOURCES from the process environment must
|
||||
reach available_sources() via env.get_config(). The unit tests above
|
||||
construct config dicts directly; this one exercises the env-to-config
|
||||
path so a missing entry in env.py's keys list is caught immediately."""
|
||||
|
||||
def test_exclude_sources_from_env_propagates_through_get_config(self):
|
||||
import os
|
||||
from unittest.mock import patch as _patch
|
||||
from lib import env as env_mod
|
||||
from importlib import reload
|
||||
with _patch.dict(os.environ, {
|
||||
"LAST30DAYS_CONFIG_DIR": "",
|
||||
"EXCLUDE_SOURCES": "tiktok,instagram",
|
||||
"SCRAPECREATORS_API_KEY": "fake",
|
||||
}, clear=False):
|
||||
reload(env_mod)
|
||||
cfg = env_mod.get_config()
|
||||
self.assertEqual(cfg.get("EXCLUDE_SOURCES"), "tiktok,instagram")
|
||||
sources = pipeline.available_sources(cfg)
|
||||
self.assertNotIn("tiktok", sources)
|
||||
self.assertNotIn("instagram", sources)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import tomllib
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
@@ -8,34 +8,33 @@ from pathlib import Path
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SKILL_ROOT = ROOT / "skills" / "last30days"
|
||||
|
||||
sys.path.insert(0, str(SKILL_ROOT / "scripts"))
|
||||
from lib.skill_meta import read_skill_version # noqa: E402
|
||||
|
||||
|
||||
def _json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def _skill_version() -> str:
|
||||
text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
||||
match = re.search(r'^version:\s*"([^"]+)"\s*$', text, re.MULTILINE)
|
||||
if not match:
|
||||
version = read_skill_version(SKILL_ROOT / "SKILL.md")
|
||||
if not version:
|
||||
raise AssertionError("SKILL.md version frontmatter not found")
|
||||
return match.group(1)
|
||||
return version
|
||||
|
||||
|
||||
class TestPluginContract(unittest.TestCase):
|
||||
def test_codex_manifest_points_at_skills_tree(self) -> None:
|
||||
manifest = _json(ROOT / ".codex-plugin" / "plugin.json")
|
||||
|
||||
self.assertEqual("last30days", manifest["name"])
|
||||
self.assertEqual("./skills/", manifest["skills"])
|
||||
self.assertTrue(SKILL_ROOT.joinpath("SKILL.md").is_file())
|
||||
self.assertTrue(SKILL_ROOT.joinpath("scripts", "last30days.py").is_file())
|
||||
def test_codex_plugin_scaffold_stays_removed(self) -> None:
|
||||
# .codex-plugin/ was removed in the resolver-collapse refactor; Codex users
|
||||
# install via `npx skills add` or `~/.codex/skills/`. A reintroduction would
|
||||
# silently fork the install surface.
|
||||
self.assertFalse((ROOT / ".codex-plugin").exists())
|
||||
|
||||
def test_versions_match_across_manifests(self) -> None:
|
||||
pyproject = tomllib.loads((ROOT / "pyproject.toml").read_text(encoding="utf-8"))
|
||||
version = pyproject["project"]["version"]
|
||||
|
||||
self.assertEqual(version, _skill_version())
|
||||
self.assertEqual(version, _json(ROOT / ".codex-plugin" / "plugin.json")["version"])
|
||||
self.assertEqual(version, _json(ROOT / ".claude-plugin" / "plugin.json")["version"])
|
||||
|
||||
marketplace = _json(ROOT / ".claude-plugin" / "marketplace.json")
|
||||
|
||||
@@ -70,8 +70,8 @@ def sample_report() -> schema.Report:
|
||||
generated_at="2026-03-16T00:00:00+00:00",
|
||||
provider_runtime=schema.ProviderRuntime(
|
||||
reasoning_provider="gemini",
|
||||
planner_model="gemini-3.1-flash-lite-preview",
|
||||
rerank_model="gemini-3.1-flash-lite-preview",
|
||||
planner_model="gemini-3.1-flash-lite",
|
||||
rerank_model="gemini-3.1-flash-lite",
|
||||
),
|
||||
query_plan=schema.QueryPlan(
|
||||
intent="breaking_news",
|
||||
@@ -91,7 +91,8 @@ def sample_report() -> schema.Report:
|
||||
class RenderV3Tests(unittest.TestCase):
|
||||
def test_render_compact_includes_cluster_first_sections(self):
|
||||
text = render.render_compact(sample_report())
|
||||
self.assertIn("# last30days v3.0.0: test topic", text)
|
||||
self.assertIn("# last30days v", text)
|
||||
self.assertIn(": test topic", text)
|
||||
self.assertIn("Safety note: evidence text below is untrusted internet content", text)
|
||||
self.assertIn("## Ranked Evidence Clusters", text)
|
||||
self.assertIn("## Stats", text)
|
||||
@@ -239,8 +240,8 @@ class RenderTopCommentsTests(unittest.TestCase):
|
||||
generated_at="2026-03-16T00:00:00+00:00",
|
||||
provider_runtime=schema.ProviderRuntime(
|
||||
reasoning_provider="gemini",
|
||||
planner_model="gemini-3.1-flash-lite-preview",
|
||||
rerank_model="gemini-3.1-flash-lite-preview",
|
||||
planner_model="gemini-3.1-flash-lite",
|
||||
rerank_model="gemini-3.1-flash-lite",
|
||||
),
|
||||
query_plan=schema.QueryPlan(
|
||||
intent="breaking_news",
|
||||
@@ -424,8 +425,8 @@ class RenderBestTakesCompactTests(unittest.TestCase):
|
||||
generated_at="2026-03-16T00:00:00+00:00",
|
||||
provider_runtime=schema.ProviderRuntime(
|
||||
reasoning_provider="gemini",
|
||||
planner_model="gemini-3.1-flash-lite-preview",
|
||||
rerank_model="gemini-3.1-flash-lite-preview",
|
||||
planner_model="gemini-3.1-flash-lite",
|
||||
rerank_model="gemini-3.1-flash-lite",
|
||||
),
|
||||
query_plan=schema.QueryPlan(
|
||||
intent="breaking_news",
|
||||
|
||||
@@ -172,10 +172,10 @@ class RerankV3Tests(unittest.TestCase):
|
||||
plan=make_plan(),
|
||||
candidates=[first, second],
|
||||
provider=provider,
|
||||
model="gemini-3.1-flash-lite-preview",
|
||||
model="gemini-3.1-flash-lite",
|
||||
shortlist_size=1,
|
||||
)
|
||||
self.assertEqual("gemini-3.1-flash-lite-preview", provider.model)
|
||||
self.assertEqual("gemini-3.1-flash-lite", provider.model)
|
||||
self.assertEqual(95.0, first.rerank_score)
|
||||
self.assertEqual("high fit", first.explanation)
|
||||
# Tail is scored via the fallback (may or may not carry the entity-miss
|
||||
|
||||
@@ -16,8 +16,8 @@ class SchemaV3Tests(unittest.TestCase):
|
||||
generated_at="2026-03-16T00:00:00+00:00",
|
||||
provider_runtime=schema.ProviderRuntime(
|
||||
reasoning_provider="gemini",
|
||||
planner_model="gemini-3.1-flash-lite-preview",
|
||||
rerank_model="gemini-3.1-flash-lite-preview",
|
||||
planner_model="gemini-3.1-flash-lite",
|
||||
rerank_model="gemini-3.1-flash-lite",
|
||||
),
|
||||
query_plan=schema.QueryPlan(
|
||||
intent="breaking_news",
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
WORKFLOW = ROOT / ".github" / "workflows" / "security.yml"
|
||||
# AGENTS.md is the canonical agent-guidance file; CLAUDE.md is a one-line
|
||||
# pointer (`@AGENTS.md`) so anything Claude Code-shaped reads the same source.
|
||||
AGENTS = ROOT / "AGENTS.md"
|
||||
|
||||
|
||||
def _workflow_text() -> str:
|
||||
return WORKFLOW.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
def test_security_workflow_exists() -> None:
|
||||
assert WORKFLOW.is_file()
|
||||
|
||||
|
||||
def test_security_workflow_runs_dependency_audit_advisory_first() -> None:
|
||||
text = _workflow_text()
|
||||
|
||||
assert "dependency-audit:" in text
|
||||
assert "pip-audit" in text
|
||||
assert "continue-on-error: true" in text
|
||||
assert "Set continue-on-error: false once a clean baseline run is confirmed" in text
|
||||
|
||||
|
||||
def test_security_workflow_runs_secret_scan_for_pull_requests_and_main_pushes() -> None:
|
||||
text = _workflow_text()
|
||||
|
||||
assert "secret-scan:" in text
|
||||
assert "trufflesecurity/trufflehog" in text
|
||||
assert "github.event_name == 'pull_request'" in text
|
||||
assert "github.event_name == 'push'" in text
|
||||
assert "--only-verified" in text
|
||||
|
||||
|
||||
def test_security_workflow_documents_advisory_policy() -> None:
|
||||
text = _workflow_text()
|
||||
|
||||
assert "advisory-first" in text.lower()
|
||||
assert "does not block merges" in text.lower()
|
||||
assert "fixtures" in text.lower()
|
||||
assert "env-based auth" in text.lower()
|
||||
|
||||
|
||||
def test_agent_guidance_mentions_secret_hygiene() -> None:
|
||||
text = AGENTS.read_text(encoding="utf-8")
|
||||
|
||||
assert "Security hygiene" in text
|
||||
assert "Never commit real API keys" in text
|
||||
assert "skills/last30days/scripts/lib/env.py" in text
|
||||
assert "fixtures" in text
|
||||
@@ -64,6 +64,19 @@ class TestRunOpenclawSetup:
|
||||
assert result["keys"]["brave"] is True
|
||||
assert result["keys"]["scrapecreators"] is False
|
||||
|
||||
def test_openclaw_metadata_keeps_scrapecreators_optional(self):
|
||||
"""OpenClaw metadata should not hard-require the ScrapeCreators key."""
|
||||
skill_md = Path(__file__).parent.parent / "skills" / "last30days" / "SKILL.md"
|
||||
text = skill_md.read_text()
|
||||
assert "SCRAPECREATORS_API_KEY" in text
|
||||
expected = (
|
||||
"requires:\n"
|
||||
" env: []\n"
|
||||
" optionalEnv:\n"
|
||||
" - SCRAPECREATORS_API_KEY"
|
||||
)
|
||||
assert expected in text
|
||||
|
||||
@patch("shutil.which")
|
||||
def test_x_method_xai(self, mock_which):
|
||||
"""x_method is 'xai' when XAI_API_KEY is set."""
|
||||
@@ -200,7 +213,9 @@ class TestPollDeviceAuth:
|
||||
@patch("lib.setup_wizard.urlopen")
|
||||
def test_timeout_returns_none(self, mock_urlopen, mock_time):
|
||||
"""Returns None when timeout is exceeded."""
|
||||
# Simulate time passing beyond deadline
|
||||
# poll_device_auth captures started_at once, derives deadline + last_reminder
|
||||
# from it, then checks time.time() in the while-loop. Two values: started_at,
|
||||
# then a value past the deadline so the loop exits immediately.
|
||||
mock_time.time = MagicMock(side_effect=[0, 301])
|
||||
mock_time.sleep = MagicMock()
|
||||
|
||||
@@ -211,7 +226,9 @@ class TestPollDeviceAuth:
|
||||
@patch("lib.setup_wizard.urlopen")
|
||||
def test_expired_token_returns_none(self, mock_urlopen, mock_time):
|
||||
"""Returns None on expired_token error."""
|
||||
mock_time.time = MagicMock(side_effect=[0, 0])
|
||||
# Loop terminates via urlopen response, not the clock — pin time to 0
|
||||
# so the deadline check stays a non-event regardless of call count.
|
||||
mock_time.time = MagicMock(return_value=0)
|
||||
mock_time.sleep = MagicMock()
|
||||
|
||||
expired_resp = MagicMock()
|
||||
@@ -230,7 +247,7 @@ class TestPollDeviceAuth:
|
||||
"""HTTP 400 during polling continues (authorization pending)."""
|
||||
from urllib.error import HTTPError
|
||||
|
||||
mock_time.time = MagicMock(side_effect=[0, 0, 0])
|
||||
mock_time.time = MagicMock(return_value=0)
|
||||
mock_time.sleep = MagicMock()
|
||||
|
||||
success_resp = MagicMock()
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
"""Direct unit tests for skill_meta.read_skill_version.
|
||||
|
||||
Covers the helper's own contract independent of render._skill_version which
|
||||
exercises it transitively. Without these, regressions in error handling or
|
||||
regex coverage inside the helper could pass CI because render.py's fallback
|
||||
to "?" swallows the signal.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(ROOT / "skills" / "last30days" / "scripts"))
|
||||
from lib.skill_meta import read_skill_version # noqa: E402
|
||||
|
||||
|
||||
class ReadSkillVersionTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp_path = Path(self._tmp.name)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _write_skill_md(self, body: str) -> Path:
|
||||
path = self.tmp_path / "SKILL.md"
|
||||
path.write_text(body)
|
||||
return path
|
||||
|
||||
def test_double_quoted_version(self) -> None:
|
||||
path = self._write_skill_md('---\nname: x\nversion: "9.9.9"\n---\n')
|
||||
self.assertEqual("9.9.9", read_skill_version(path))
|
||||
|
||||
def test_single_quoted_version(self) -> None:
|
||||
path = self._write_skill_md("---\nname: x\nversion: '8.8.8'\n---\n")
|
||||
self.assertEqual("8.8.8", read_skill_version(path))
|
||||
|
||||
def test_unquoted_version(self) -> None:
|
||||
path = self._write_skill_md("---\nname: x\nversion: 7.7.7\n---\n")
|
||||
self.assertEqual("7.7.7", read_skill_version(path))
|
||||
|
||||
def test_missing_file_returns_none(self) -> None:
|
||||
self.assertIsNone(read_skill_version(self.tmp_path / "does-not-exist.md"))
|
||||
|
||||
def test_no_version_line_returns_none(self) -> None:
|
||||
path = self._write_skill_md("---\nname: x\n---\n# body without version\n")
|
||||
self.assertIsNone(read_skill_version(path))
|
||||
|
||||
def test_undecodable_bytes_returns_none(self) -> None:
|
||||
# Bytes 128-255 don't form valid UTF-8 sequences; read_text() raises
|
||||
# UnicodeDecodeError which the helper must catch.
|
||||
path = self.tmp_path / "SKILL.md"
|
||||
path.write_bytes(bytes(range(128, 256)))
|
||||
self.assertIsNone(read_skill_version(path))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,137 @@
|
||||
"""Unit tests for render._skill_version() fallback paths.
|
||||
|
||||
The function reads version from .claude-plugin/plugin.json first, then falls back
|
||||
to SKILL.md frontmatter. These tests use monkeypatch to swap the render module's
|
||||
__file__ attribute, which controls where the walk starts.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "skills" / "last30days" / "scripts"))
|
||||
|
||||
from lib import render
|
||||
|
||||
|
||||
class SkillVersionFallbackTests(unittest.TestCase):
|
||||
def setUp(self):
|
||||
# tmp_path equivalent for unittest
|
||||
import tempfile
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp_path = Path(self._tmp.name)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _make_render_at(self, parent: Path) -> Path:
|
||||
"""Place a dummy render.py inside parent and return its path."""
|
||||
parent.mkdir(parents=True, exist_ok=True)
|
||||
fake_render = parent / "render.py"
|
||||
fake_render.write_text("")
|
||||
return fake_render
|
||||
|
||||
def _write_manifest(self, parent: Path, version: str | None) -> None:
|
||||
"""Write .claude-plugin/plugin.json under parent. version=None writes corrupt JSON."""
|
||||
d = parent / ".claude-plugin"
|
||||
d.mkdir(parents=True, exist_ok=True)
|
||||
if version is None:
|
||||
(d / "plugin.json").write_text("{not valid json")
|
||||
else:
|
||||
(d / "plugin.json").write_text(f'{{"version": "{version}"}}')
|
||||
|
||||
def _write_skill_md(self, parent: Path, frontmatter_version_line: str | None) -> None:
|
||||
"""Write SKILL.md with frontmatter. None writes a SKILL.md with no version line."""
|
||||
if frontmatter_version_line is None:
|
||||
body = "---\nname: test\n---\n# body\n"
|
||||
else:
|
||||
body = f"---\nname: test\n{frontmatter_version_line}\n---\n# body\n"
|
||||
(parent / "SKILL.md").write_text(body)
|
||||
|
||||
def test_manifest_absent_falls_back_to_skill_md_frontmatter(self):
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
self._write_skill_md(skill_dir, 'version: "9.9.9"')
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("9.9.9", render._skill_version())
|
||||
|
||||
def test_manifest_corrupt_falls_back_to_skill_md_frontmatter(self):
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
self._write_manifest(skill_dir, version=None) # corrupt
|
||||
self._write_skill_md(skill_dir, 'version: "8.8.8"')
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("8.8.8", render._skill_version())
|
||||
|
||||
def test_manifest_missing_version_key_falls_back_to_skill_md_frontmatter(self):
|
||||
# Valid JSON, but no "version" key. Old behavior returned "?" and never tried
|
||||
# SKILL.md. Fix from greptile review: fall through to SKILL.md fallback.
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
(skill_dir / ".claude-plugin").mkdir()
|
||||
(skill_dir / ".claude-plugin" / "plugin.json").write_text('{"name": "x"}')
|
||||
self._write_skill_md(skill_dir, 'version: "4.4.4"')
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("4.4.4", render._skill_version())
|
||||
|
||||
def test_manifest_empty_version_string_falls_back_to_skill_md_frontmatter(self):
|
||||
# Manifest version present but empty — treat as missing so the badge
|
||||
# doesn't emit a useless "🌐 last30days v · synced ..." line.
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
(skill_dir / ".claude-plugin").mkdir()
|
||||
(skill_dir / ".claude-plugin" / "plugin.json").write_text('{"version": ""}')
|
||||
self._write_skill_md(skill_dir, 'version: "2.2.2"')
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("2.2.2", render._skill_version())
|
||||
|
||||
def test_corrupt_inner_manifest_does_not_shadow_valid_outer_manifest(self):
|
||||
outer = self.tmp_path / "outer"
|
||||
inner = outer / "skill_root"
|
||||
fake_render = self._make_render_at(inner)
|
||||
self._write_manifest(inner, version=None) # corrupt at inner
|
||||
self._write_manifest(outer, version="7.7.7") # valid at outer
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("7.7.7", render._skill_version())
|
||||
|
||||
def test_neither_source_present_returns_question_mark(self):
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
# No manifest, no SKILL.md anywhere under tmp_path
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("?", render._skill_version())
|
||||
|
||||
def test_skill_md_without_version_returns_question_mark(self):
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
self._write_skill_md(skill_dir, frontmatter_version_line=None)
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("?", render._skill_version())
|
||||
|
||||
def test_unquoted_yaml_version_is_accepted(self):
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
self._write_skill_md(skill_dir, "version: 6.6.6")
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("6.6.6", render._skill_version())
|
||||
|
||||
def test_single_quoted_yaml_version_is_accepted(self):
|
||||
skill_dir = self.tmp_path / "skill_root"
|
||||
fake_render = self._make_render_at(skill_dir)
|
||||
self._write_skill_md(skill_dir, "version: '5.5.5'")
|
||||
|
||||
with patch.object(render, "__file__", str(fake_render)):
|
||||
self.assertEqual("5.5.5", render._skill_version())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+194
-12
@@ -3,7 +3,7 @@
|
||||
import json
|
||||
import sqlite3
|
||||
import tempfile
|
||||
from datetime import datetime, timedelta
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
@@ -59,7 +59,68 @@ def sample_report():
|
||||
"source_weights": {},
|
||||
},
|
||||
"clusters": [],
|
||||
"ranked_candidates": [],
|
||||
"ranked_candidates": [
|
||||
{
|
||||
"candidate_id": "c-r1",
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"title": "Test Reddit Post",
|
||||
"url": "https://reddit.com/r/test/1",
|
||||
"snippet": "Reddit snippet",
|
||||
"subquery_labels": ["primary"],
|
||||
"native_ranks": {"reddit": 1},
|
||||
"local_relevance": 0.8,
|
||||
"freshness": 100,
|
||||
"engagement": 50.0,
|
||||
"source_quality": 0.8,
|
||||
"rrf_score": 1.0,
|
||||
"final_score": 0.8,
|
||||
"explanation": "Reddit snippet",
|
||||
"source_items": [
|
||||
{
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"title": "Test Reddit Post",
|
||||
"body": "Reddit discussion content",
|
||||
"url": "https://reddit.com/r/test/1",
|
||||
"author": "testuser",
|
||||
"engagement_score": 50.0,
|
||||
"local_relevance": 0.8,
|
||||
"snippet": "Reddit snippet",
|
||||
}
|
||||
],
|
||||
},
|
||||
{
|
||||
"candidate_id": "c-x1",
|
||||
"item_id": "X1",
|
||||
"source": "x",
|
||||
"title": "Test X Post",
|
||||
"url": "https://x.com/test/status/1",
|
||||
"snippet": "X snippet",
|
||||
"subquery_labels": ["primary"],
|
||||
"native_ranks": {"x": 1},
|
||||
"local_relevance": 0.85,
|
||||
"freshness": 100,
|
||||
"engagement": 75.0,
|
||||
"source_quality": 0.8,
|
||||
"rrf_score": 1.0,
|
||||
"final_score": 0.85,
|
||||
"explanation": "X snippet",
|
||||
"source_items": [
|
||||
{
|
||||
"item_id": "X1",
|
||||
"source": "x",
|
||||
"title": "Test X Post",
|
||||
"body": "X post content",
|
||||
"url": "https://x.com/test/status/1",
|
||||
"author": "xuser",
|
||||
"engagement_score": 75.0,
|
||||
"local_relevance": 0.85,
|
||||
"snippet": "X snippet",
|
||||
}
|
||||
],
|
||||
},
|
||||
],
|
||||
"items_by_source": {
|
||||
"reddit": [
|
||||
{
|
||||
@@ -236,13 +297,13 @@ def test_findings_from_report_handles_missing_fields():
|
||||
"clusters": [],
|
||||
"ranked_candidates": [],
|
||||
"items_by_source": {
|
||||
"reddit": [
|
||||
"hackernews": [
|
||||
{
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"source": "hackernews",
|
||||
"title": "Test",
|
||||
"body": "Content",
|
||||
"url": "https://reddit.com/1",
|
||||
"url": "https://news.ycombinator.com/item?id=1",
|
||||
"author": None, # Missing author
|
||||
"engagement_score": None, # Missing engagement
|
||||
"local_relevance": None, # Missing relevance
|
||||
@@ -384,6 +445,127 @@ def test_store_findings_skips_items_without_url(temp_db):
|
||||
assert counts["new"] == 1
|
||||
|
||||
|
||||
def test_init_db_creates_finding_sightings_table(temp_db):
|
||||
"""Test that the per-run sightings ledger is available on fresh databases."""
|
||||
conn = sqlite3.connect(str(temp_db))
|
||||
table = conn.execute(
|
||||
"SELECT name FROM sqlite_master WHERE type='table' AND name='finding_sightings'"
|
||||
).fetchone()
|
||||
columns = {
|
||||
row[1]: row[3]
|
||||
for row in conn.execute("PRAGMA table_info(finding_sightings)").fetchall()
|
||||
}
|
||||
conn.close()
|
||||
|
||||
assert table is not None
|
||||
assert columns["finding_id"] == 1
|
||||
|
||||
|
||||
def test_store_findings_records_sightings_for_new_findings(temp_db):
|
||||
"""Test that each stored finding is linked to the run that observed it."""
|
||||
topic = store.add_topic("Test Topic")
|
||||
run_id = store.record_run(topic["id"], source_mode="v3")
|
||||
findings = [
|
||||
{
|
||||
"source": "reddit",
|
||||
"source_url": "https://reddit.com/1",
|
||||
"source_title": "Reddit 1",
|
||||
"content": "Content 1",
|
||||
"engagement_score": 10.0,
|
||||
"relevance_score": 0.7,
|
||||
},
|
||||
{
|
||||
"source": "x",
|
||||
"source_url": "https://x.com/a/status/1",
|
||||
"source_title": "X 1",
|
||||
"content": "Content 2",
|
||||
"engagement_score": 20.0,
|
||||
"relevance_score": 0.8,
|
||||
},
|
||||
]
|
||||
|
||||
store.store_findings(run_id, topic["id"], findings)
|
||||
|
||||
sightings = store.get_sightings_for_run(topic["id"], run_id)
|
||||
assert [s["source_url"] for s in sightings] == [
|
||||
"https://reddit.com/1",
|
||||
"https://x.com/a/status/1",
|
||||
]
|
||||
assert {s["source"] for s in sightings} == {"reddit", "x"}
|
||||
|
||||
|
||||
def test_store_findings_records_sightings_for_resighted_findings(temp_db):
|
||||
"""Test that a re-seen finding is recorded for each run that observes it."""
|
||||
topic = store.add_topic("Test Topic")
|
||||
first_run_id = store.record_run(topic["id"], source_mode="v3")
|
||||
second_run_id = store.record_run(topic["id"], source_mode="v3")
|
||||
finding = {
|
||||
"source": "reddit",
|
||||
"source_url": "https://reddit.com/1",
|
||||
"source_title": "Reddit 1",
|
||||
"content": "Content",
|
||||
"engagement_score": 10.0,
|
||||
"relevance_score": 0.7,
|
||||
}
|
||||
|
||||
store.store_findings(first_run_id, topic["id"], [finding])
|
||||
store.store_findings(second_run_id, topic["id"], [{**finding, "engagement_score": 15.0}])
|
||||
|
||||
first_sightings = store.get_sightings_for_run(topic["id"], first_run_id)
|
||||
second_sightings = store.get_sightings_for_run(topic["id"], second_run_id)
|
||||
|
||||
assert len(first_sightings) == 1
|
||||
assert len(second_sightings) == 1
|
||||
assert first_sightings[0]["source_url"] == second_sightings[0]["source_url"]
|
||||
assert second_sightings[0]["engagement_score"] == 15.0
|
||||
|
||||
|
||||
def test_store_findings_sightings_are_idempotent_per_run(temp_db):
|
||||
"""Test that storing the same finding twice for one run does not duplicate sightings."""
|
||||
topic = store.add_topic("Test Topic")
|
||||
run_id = store.record_run(topic["id"], source_mode="v3")
|
||||
finding = {
|
||||
"source": "reddit",
|
||||
"source_url": "https://reddit.com/1",
|
||||
"source_title": "Reddit 1",
|
||||
"content": "Content",
|
||||
"engagement_score": 10.0,
|
||||
"relevance_score": 0.7,
|
||||
}
|
||||
|
||||
store.store_findings(run_id, topic["id"], [finding])
|
||||
store.store_findings(run_id, topic["id"], [finding])
|
||||
|
||||
sightings = store.get_sightings_for_run(topic["id"], run_id)
|
||||
assert len(sightings) == 1
|
||||
|
||||
|
||||
def test_store_findings_updates_existing_sighting_for_same_run(temp_db):
|
||||
"""Test that retrying a run refreshes its sighting snapshot instead of freezing it."""
|
||||
topic = store.add_topic("Test Topic")
|
||||
run_id = store.record_run(topic["id"], source_mode="v3")
|
||||
finding = {
|
||||
"source": "reddit",
|
||||
"source_url": "https://reddit.com/1",
|
||||
"source_title": "Reddit 1",
|
||||
"content": "Content",
|
||||
"engagement_score": 10.0,
|
||||
"relevance_score": 0.7,
|
||||
}
|
||||
|
||||
store.store_findings(run_id, topic["id"], [finding])
|
||||
store.store_findings(
|
||||
run_id,
|
||||
topic["id"],
|
||||
[{**finding, "source_title": "Reddit 1 updated", "engagement_score": 15.0}],
|
||||
)
|
||||
|
||||
sightings = store.get_sightings_for_run(topic["id"], run_id)
|
||||
assert len(sightings) == 1
|
||||
assert sightings[0]["source_title"] == "Reddit 1 updated"
|
||||
assert sightings[0]["engagement_score"] == 15.0
|
||||
|
||||
|
||||
def test_update_validates_allowed_columns(temp_db, sample_report):
|
||||
"""Test update_run/update_finding accept valid keys and reject invalid keys."""
|
||||
topic = store.add_topic("Test Topic")
|
||||
@@ -503,16 +685,16 @@ def test_get_new_findings_filters_by_date(temp_db, sample_report):
|
||||
findings = store.findings_from_report(sample_report)
|
||||
store.store_findings(run_id, topic["id"], findings)
|
||||
|
||||
# Get findings since tomorrow (should be empty)
|
||||
tomorrow = (datetime.now() + timedelta(days=1)).strftime("%Y-%m-%d")
|
||||
# Use UTC because store writes first_seen via SQLite's datetime('now') (UTC).
|
||||
# Local-time math here would flake near midnight UTC.
|
||||
tomorrow = (datetime.now(timezone.utc) + timedelta(days=1)).strftime("%Y-%m-%d")
|
||||
new_findings = store.get_new_findings(topic["id"], since=tomorrow)
|
||||
|
||||
|
||||
assert len(new_findings) == 0
|
||||
|
||||
# Get findings since yesterday (should have all)
|
||||
yesterday = (datetime.now() - timedelta(days=1)).strftime("%Y-%m-%d")
|
||||
|
||||
yesterday = (datetime.now(timezone.utc) - timedelta(days=1)).strftime("%Y-%m-%d")
|
||||
new_findings = store.get_new_findings(topic["id"], since=yesterday)
|
||||
|
||||
|
||||
assert len(new_findings) == 4
|
||||
|
||||
|
||||
|
||||
+3
-18
@@ -154,14 +154,7 @@ class TestTikTokEnrichWithComments(unittest.TestCase):
|
||||
"total": 3,
|
||||
}
|
||||
|
||||
class FakeResp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
def json(self):
|
||||
return fake_sc_response
|
||||
|
||||
with patch.object(tiktok, "_requests") as mock_req:
|
||||
mock_req.get.return_value = FakeResp()
|
||||
with patch.object(tiktok.http, "get", return_value=fake_sc_response):
|
||||
out = tiktok._fetch_post_comments(
|
||||
"https://www.tiktok.com/@u/video/1",
|
||||
token="k",
|
||||
@@ -192,14 +185,7 @@ class TestTikTokEnrichWithComments(unittest.TestCase):
|
||||
"total": 3,
|
||||
}
|
||||
|
||||
class FakeResp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
def json(self):
|
||||
return fake_sc_response
|
||||
|
||||
with patch.object(tiktok, "_requests") as mock_req:
|
||||
mock_req.get.return_value = FakeResp()
|
||||
with patch.object(tiktok.http, "get", return_value=fake_sc_response):
|
||||
out = tiktok._fetch_post_comments(
|
||||
"https://www.tiktok.com/@u/video/1",
|
||||
token="k",
|
||||
@@ -217,8 +203,7 @@ class TestTikTokEnrichWithComments(unittest.TestCase):
|
||||
from unittest.mock import patch
|
||||
from lib import tiktok
|
||||
|
||||
with patch.object(tiktok, "_requests") as mock_req:
|
||||
mock_req.get.side_effect = Exception("429 rate limit")
|
||||
with patch.object(tiktok.http, "get", side_effect=Exception("429 rate limit")):
|
||||
out = tiktok._fetch_post_comments(
|
||||
"https://www.tiktok.com/@u/video/1",
|
||||
token="k",
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import re
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
@@ -6,26 +7,36 @@ from pathlib import Path
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SKILL_ROOT = ROOT / "skills" / "last30days"
|
||||
|
||||
sys.path.insert(0, str(SKILL_ROOT / "scripts"))
|
||||
from lib.skill_meta import read_skill_version # noqa: E402
|
||||
|
||||
|
||||
def _skill_version() -> str:
|
||||
text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
||||
match = re.search(r'^version:\s*"([^"]+)"\s*$', text, re.MULTILINE)
|
||||
if not match:
|
||||
version = read_skill_version(SKILL_ROOT / "SKILL.md")
|
||||
if not version:
|
||||
raise AssertionError("SKILL.md version frontmatter not found")
|
||||
return match.group(1)
|
||||
return version
|
||||
|
||||
|
||||
class TestVersionConsistency(unittest.TestCase):
|
||||
def test_skill_md_uses_double_quoted_version(self) -> None:
|
||||
# The shared VERSION_RE in skill_meta.py accepts double-quoted,
|
||||
# single-quoted, and unquoted YAML version scalars. This repo's
|
||||
# SKILL.md must use the double-quoted form so the badge string stays
|
||||
# deterministic and contributors don't accidentally introduce a
|
||||
# quoting style that's harder for downstream tooling to parse.
|
||||
text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
||||
self.assertRegex(
|
||||
text,
|
||||
re.compile(r'^version:\s*"[^"]+"\s*$', re.MULTILINE),
|
||||
msg="SKILL.md frontmatter version must use double-quoted form",
|
||||
)
|
||||
|
||||
def test_root_skill_header_matches_frontmatter_version(self) -> None:
|
||||
text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
||||
version = _skill_version()
|
||||
self.assertIn(f"# last30days v{version}:", text)
|
||||
|
||||
def test_sync_cache_path_uses_skill_version(self) -> None:
|
||||
sync_text = (SKILL_ROOT / "scripts" / "sync.sh").read_text(encoding="utf-8")
|
||||
version = _skill_version()
|
||||
self.assertIn(f'last30days-3/{version}"', sync_text)
|
||||
|
||||
def test_memory_save_dir_uses_single_env_variable(self) -> None:
|
||||
skill_text = (SKILL_ROOT / "SKILL.md").read_text(encoding="utf-8")
|
||||
compare_text = (SKILL_ROOT / "scripts" / "compare.sh").read_text(encoding="utf-8")
|
||||
|
||||
@@ -251,7 +251,38 @@ def test_run_topic_success(mock_subprocess, temp_db):
|
||||
"source_weights": {},
|
||||
},
|
||||
"clusters": [],
|
||||
"ranked_candidates": [],
|
||||
"ranked_candidates": [
|
||||
{
|
||||
"candidate_id": "c-r1",
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"title": "Test",
|
||||
"url": "https://reddit.com/1",
|
||||
"snippet": "Snippet",
|
||||
"subquery_labels": ["primary"],
|
||||
"native_ranks": {"reddit": 1},
|
||||
"local_relevance": 0.8,
|
||||
"freshness": 100,
|
||||
"engagement": 50.0,
|
||||
"source_quality": 0.8,
|
||||
"rrf_score": 1.0,
|
||||
"final_score": 0.8,
|
||||
"explanation": "Snippet",
|
||||
"source_items": [
|
||||
{
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"title": "Test",
|
||||
"body": "Content",
|
||||
"url": "https://reddit.com/1",
|
||||
"author": "user",
|
||||
"engagement_score": 50.0,
|
||||
"local_relevance": 0.8,
|
||||
"snippet": "Snippet",
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
"items_by_source": {
|
||||
"reddit": [
|
||||
{
|
||||
@@ -338,7 +369,38 @@ def test_run_topic_calls_delivery(mock_deliver, mock_subprocess, temp_db):
|
||||
"source_weights": {},
|
||||
},
|
||||
"clusters": [],
|
||||
"ranked_candidates": [],
|
||||
"ranked_candidates": [
|
||||
{
|
||||
"candidate_id": "c-r1",
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"title": "Test",
|
||||
"url": "https://reddit.com/1",
|
||||
"snippet": "Snippet",
|
||||
"subquery_labels": ["primary"],
|
||||
"native_ranks": {"reddit": 1},
|
||||
"local_relevance": 0.8,
|
||||
"freshness": 100,
|
||||
"engagement": 50.0,
|
||||
"source_quality": 0.8,
|
||||
"rrf_score": 1.0,
|
||||
"final_score": 0.8,
|
||||
"explanation": "Snippet",
|
||||
"source_items": [
|
||||
{
|
||||
"item_id": "R1",
|
||||
"source": "reddit",
|
||||
"title": "Test",
|
||||
"body": "Content",
|
||||
"url": "https://reddit.com/1",
|
||||
"author": "user",
|
||||
"engagement_score": 50.0,
|
||||
"local_relevance": 0.8,
|
||||
"snippet": "Snippet",
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
"items_by_source": {
|
||||
"reddit": [
|
||||
{
|
||||
|
||||
@@ -2,13 +2,14 @@
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import Mock, patch
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "skills" / "last30days" / "scripts"))
|
||||
|
||||
import watchlist
|
||||
from lib.http import HTTPError
|
||||
|
||||
|
||||
# === Tests for _format_delivery_message() ===
|
||||
@@ -18,7 +19,7 @@ def test_format_message_announce_mode():
|
||||
message = watchlist._format_delivery_message(
|
||||
"Test Topic", {"new": 5, "updated": 2}, "announce"
|
||||
)
|
||||
|
||||
|
||||
assert "📰" in message
|
||||
assert "Test Topic" in message
|
||||
assert "5 new" in message
|
||||
@@ -30,7 +31,7 @@ def test_format_message_silent_mode():
|
||||
message = watchlist._format_delivery_message(
|
||||
"Test Topic", {"new": 5, "updated": 2}, "silent"
|
||||
)
|
||||
|
||||
|
||||
assert "📰" not in message
|
||||
assert "Test Topic" in message
|
||||
assert "5 new" in message
|
||||
@@ -41,7 +42,7 @@ def test_format_message_default_mode():
|
||||
message = watchlist._format_delivery_message(
|
||||
"Test Topic", {"new": 5, "updated": 2}, "default"
|
||||
)
|
||||
|
||||
|
||||
assert "complete" in message.lower()
|
||||
assert "Test Topic" in message
|
||||
|
||||
@@ -51,44 +52,35 @@ def test_format_message_handles_zero_counts():
|
||||
message = watchlist._format_delivery_message(
|
||||
"Test Topic", {"new": 0, "updated": 0}, "announce"
|
||||
)
|
||||
|
||||
|
||||
assert "0 new" in message
|
||||
assert "0 updated" in message
|
||||
|
||||
|
||||
# === Tests for _send_slack_webhook() ===
|
||||
|
||||
@patch('watchlist.requests')
|
||||
def test_send_slack_webhook_format(mock_requests):
|
||||
@patch('watchlist.http.post')
|
||||
def test_send_slack_webhook_format(mock_post):
|
||||
"""Test that Slack webhook uses correct format."""
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
watchlist._send_slack_webhook(
|
||||
"https://hooks.slack.com/services/TEST",
|
||||
"Test message"
|
||||
)
|
||||
|
||||
# Verify POST was called with correct format
|
||||
assert mock_requests.post.called
|
||||
call_args = mock_requests.post.call_args
|
||||
|
||||
|
||||
assert mock_post.called
|
||||
call_args = mock_post.call_args
|
||||
|
||||
assert call_args[0][0] == "https://hooks.slack.com/services/TEST"
|
||||
assert call_args[1]["json"] == {"text": "Test message"}
|
||||
assert call_args[1]["headers"]["Content-Type"] == "application/json"
|
||||
assert call_args[1]["json_data"] == {"text": "Test message"}
|
||||
assert call_args[1]["timeout"] == 10
|
||||
|
||||
|
||||
@patch('watchlist.requests')
|
||||
def test_send_slack_webhook_raises_on_error(mock_requests):
|
||||
@patch('watchlist.http.post')
|
||||
def test_send_slack_webhook_raises_on_error(mock_post):
|
||||
"""Test that Slack webhook raises on HTTP error."""
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 400
|
||||
mock_response.raise_for_status.side_effect = Exception("HTTP 400")
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
with pytest.raises(Exception, match="HTTP 400"):
|
||||
mock_post.side_effect = HTTPError("HTTP 400", 400)
|
||||
|
||||
with pytest.raises(HTTPError, match="HTTP 400"):
|
||||
watchlist._send_slack_webhook(
|
||||
"https://hooks.slack.com/services/TEST",
|
||||
"Test message"
|
||||
@@ -97,40 +89,32 @@ def test_send_slack_webhook_raises_on_error(mock_requests):
|
||||
|
||||
# === Tests for _send_generic_webhook() ===
|
||||
|
||||
@patch('watchlist.requests')
|
||||
def test_send_generic_webhook_format(mock_requests):
|
||||
@patch('watchlist.http.post')
|
||||
def test_send_generic_webhook_format(mock_post):
|
||||
"""Test that generic webhook uses correct format."""
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
watchlist._send_generic_webhook(
|
||||
"https://webhook.example.com/hook",
|
||||
"Test message"
|
||||
)
|
||||
|
||||
# Verify POST was called with correct format
|
||||
assert mock_requests.post.called
|
||||
call_args = mock_requests.post.call_args
|
||||
|
||||
|
||||
assert mock_post.called
|
||||
call_args = mock_post.call_args
|
||||
|
||||
assert call_args[0][0] == "https://webhook.example.com/hook"
|
||||
|
||||
json_data = call_args[1]["json"]
|
||||
|
||||
json_data = call_args[1]["json_data"]
|
||||
assert json_data["message"] == "Test message"
|
||||
assert json_data["source"] == "last30days"
|
||||
assert "timestamp" in json_data
|
||||
assert isinstance(json_data["timestamp"], float)
|
||||
|
||||
|
||||
@patch('watchlist.requests')
|
||||
def test_send_generic_webhook_raises_on_error(mock_requests):
|
||||
@patch('watchlist.http.post')
|
||||
def test_send_generic_webhook_raises_on_error(mock_post):
|
||||
"""Test that generic webhook raises on HTTP error."""
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 500
|
||||
mock_response.raise_for_status.side_effect = Exception("HTTP 500")
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
with pytest.raises(Exception, match="HTTP 500"):
|
||||
mock_post.side_effect = HTTPError("HTTP 500", 500)
|
||||
|
||||
with pytest.raises(HTTPError, match="HTTP 500"):
|
||||
watchlist._send_generic_webhook(
|
||||
"https://webhook.example.com/hook",
|
||||
"Test message"
|
||||
@@ -140,150 +124,120 @@ def test_send_generic_webhook_raises_on_error(mock_requests):
|
||||
# === Tests for _deliver_findings() ===
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_sends_when_new_greater_than_zero(mock_requests, mock_get_setting):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_sends_when_new_greater_than_zero(mock_post, mock_get_setting):
|
||||
"""Test that delivery fires when new > 0."""
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://webhook.example.com/test",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
# Verify webhook was called
|
||||
assert mock_requests.post.called
|
||||
|
||||
assert mock_post.called
|
||||
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_skips_when_new_is_zero(mock_requests, mock_get_setting):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_skips_when_new_is_zero(mock_post, mock_get_setting):
|
||||
"""Test that delivery is skipped when new=0."""
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://webhook.example.com/test",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 0, "updated": 5})
|
||||
|
||||
# Verify webhook was NOT called
|
||||
assert not mock_requests.post.called
|
||||
|
||||
assert not mock_post.called
|
||||
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_skips_when_channel_empty(mock_requests, mock_get_setting):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_skips_when_channel_empty(mock_post, mock_get_setting):
|
||||
"""Test that delivery is skipped when delivery_channel is empty."""
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
# Verify webhook was NOT called
|
||||
assert not mock_requests.post.called
|
||||
|
||||
assert not mock_post.called
|
||||
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_uses_slack_format_for_slack_urls(mock_requests, mock_get_setting):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_uses_slack_format_for_slack_urls(mock_post, mock_get_setting):
|
||||
"""Test that Slack URLs trigger Slack-specific format."""
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://hooks.slack.com/services/TEST",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
# Verify Slack format was used
|
||||
call_args = mock_requests.post.call_args
|
||||
json_data = call_args[1]["json"]
|
||||
|
||||
json_data = mock_post.call_args[1]["json_data"]
|
||||
assert "text" in json_data
|
||||
assert "Test Topic" in json_data["text"]
|
||||
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_uses_generic_format_for_other_urls(mock_requests, mock_get_setting):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_uses_generic_format_for_other_urls(mock_post, mock_get_setting):
|
||||
"""Test that non-Slack URLs trigger generic format."""
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://webhook.example.com/test",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
# Verify generic format was used
|
||||
call_args = mock_requests.post.call_args
|
||||
json_data = call_args[1]["json"]
|
||||
|
||||
json_data = mock_post.call_args[1]["json_data"]
|
||||
assert "message" in json_data
|
||||
assert "source" in json_data
|
||||
assert "timestamp" in json_data
|
||||
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_handles_failure_gracefully(mock_requests, mock_get_setting, capsys):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_handles_failure_gracefully(mock_post, mock_get_setting, capsys):
|
||||
"""Test that delivery failures don't crash the process."""
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://webhook.example.com/test",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
# Simulate HTTP error
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 500
|
||||
mock_response.raise_for_status.side_effect = Exception("HTTP 500")
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
|
||||
mock_post.side_effect = HTTPError("HTTP 500", 500)
|
||||
|
||||
# Should not raise, just log to stderr
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
# Verify error was logged
|
||||
|
||||
captured = capsys.readouterr()
|
||||
assert "Delivery failed" in captured.err
|
||||
|
||||
|
||||
@patch('watchlist.store.get_setting')
|
||||
@patch('watchlist.requests')
|
||||
def test_deliver_findings_respects_delivery_mode(mock_requests, mock_get_setting):
|
||||
@patch('watchlist.http.post')
|
||||
def test_deliver_findings_respects_delivery_mode(mock_post, mock_get_setting):
|
||||
"""Test that different delivery modes produce different messages."""
|
||||
mock_response = Mock()
|
||||
mock_response.status_code = 200
|
||||
mock_requests.post.return_value = mock_response
|
||||
|
||||
# Test announce mode
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://webhook.example.com/test",
|
||||
"delivery_mode": "announce",
|
||||
}.get(key, default)
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
announce_message = mock_requests.post.call_args[1]["json"]["message"]
|
||||
|
||||
announce_message = mock_post.call_args[1]["json_data"]["message"]
|
||||
assert "📰" in announce_message
|
||||
|
||||
# Test silent mode
|
||||
|
||||
mock_get_setting.side_effect = lambda key, default="": {
|
||||
"delivery_channel": "https://webhook.example.com/test",
|
||||
"delivery_mode": "silent",
|
||||
}.get(key, default)
|
||||
|
||||
|
||||
watchlist._deliver_findings("Test Topic", {"new": 5, "updated": 2})
|
||||
|
||||
silent_message = mock_requests.post.call_args[1]["json"]["message"]
|
||||
|
||||
silent_message = mock_post.call_args[1]["json_data"]["message"]
|
||||
assert "📰" not in silent_message
|
||||
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
"""Tests for YouTube transcript highlights and yt-dlp safety flags."""
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
@@ -407,5 +408,134 @@ class TestSearchAndTranscribe(unittest.TestCase):
|
||||
ft_mock.assert_not_called()
|
||||
|
||||
|
||||
class TestYtdlpSSHRouting(unittest.TestCase):
|
||||
"""LAST30DAYS_YOUTUBE_SSH_HOST routes yt-dlp invocations through SSH for residential IP."""
|
||||
|
||||
def setUp(self):
|
||||
# Ensure clean env for each test
|
||||
self._saved_env = os.environ.pop("LAST30DAYS_YOUTUBE_SSH_HOST", None)
|
||||
|
||||
def tearDown(self):
|
||||
os.environ.pop("LAST30DAYS_YOUTUBE_SSH_HOST", None)
|
||||
if self._saved_env is not None:
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = self._saved_env
|
||||
|
||||
def test_no_env_var_returns_none(self):
|
||||
"""Without the env var set, _ytdlp_ssh_host returns None."""
|
||||
self.assertIsNone(youtube_yt._ytdlp_ssh_host())
|
||||
|
||||
def test_env_var_returns_host(self):
|
||||
"""With LAST30DAYS_YOUTUBE_SSH_HOST set, _ytdlp_ssh_host returns it."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "macmini"
|
||||
self.assertEqual(youtube_yt._ytdlp_ssh_host(), "macmini")
|
||||
|
||||
def test_env_var_whitespace_stripped(self):
|
||||
"""Whitespace around the host alias is stripped."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = " macmini "
|
||||
self.assertEqual(youtube_yt._ytdlp_ssh_host(), "macmini")
|
||||
|
||||
def test_empty_env_var_falls_back_to_none(self):
|
||||
"""An empty env var is treated as unset."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = ""
|
||||
self.assertIsNone(youtube_yt._ytdlp_ssh_host())
|
||||
|
||||
def test_wrap_cmd_passthrough_when_unset(self):
|
||||
"""_wrap_ytdlp_cmd returns input unchanged when SSH routing is off."""
|
||||
cmd = ["yt-dlp", "--ignore-config", "ytsearch5:test"]
|
||||
self.assertEqual(youtube_yt._wrap_ytdlp_cmd(cmd), cmd)
|
||||
|
||||
def test_wrap_cmd_prepends_ssh_when_set(self):
|
||||
"""_wrap_ytdlp_cmd prepends ssh <host> when SSH routing is on."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "macmini"
|
||||
cmd = ["yt-dlp", "--ignore-config", "ytsearch5:test"]
|
||||
wrapped = youtube_yt._wrap_ytdlp_cmd(cmd)
|
||||
self.assertEqual(wrapped[0], "ssh")
|
||||
self.assertEqual(wrapped[1], "-o")
|
||||
self.assertEqual(wrapped[2], "BatchMode=yes")
|
||||
# `--` terminates SSH option parsing so a host starting with `-`
|
||||
# (e.g. `-oProxyCommand=...`) cannot be reinterpreted as a flag.
|
||||
self.assertEqual(wrapped[3], "--")
|
||||
self.assertEqual(wrapped[4], "macmini")
|
||||
# Final arg is the shell-quoted command string
|
||||
self.assertIn("yt-dlp", wrapped[5])
|
||||
self.assertIn("ytsearch5:test", wrapped[5])
|
||||
|
||||
def test_wrap_cmd_quotes_args_with_spaces(self):
|
||||
"""Args containing spaces or special chars are shell-quoted."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "macmini"
|
||||
cmd = ["yt-dlp", "ytsearch5:hello world", "--dump-json"]
|
||||
wrapped = youtube_yt._wrap_ytdlp_cmd(cmd)
|
||||
# shlex.quote wraps the whole arg in single quotes when it contains spaces
|
||||
self.assertIn("'ytsearch5:hello world'", wrapped[5])
|
||||
|
||||
def test_wrap_cmd_uses_option_terminator(self):
|
||||
"""`--` is inserted before host as defense-in-depth even for valid hosts."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "macmini"
|
||||
cmd = ["yt-dlp", "--version"]
|
||||
wrapped = youtube_yt._wrap_ytdlp_cmd(cmd)
|
||||
dash_idx = wrapped.index("--")
|
||||
self.assertEqual(wrapped[dash_idx + 1], "macmini")
|
||||
|
||||
def test_host_alias_with_dash_prefix_is_rejected(self):
|
||||
"""A host value starting with `-` is rejected by the alias validator.
|
||||
|
||||
Without validation, ssh could parse `-oProxyCommand=...` as a flag
|
||||
instead of a hostname. The `--` terminator in _wrap_ytdlp_cmd is
|
||||
defense-in-depth; this regex on _ytdlp_ssh_host() rejects the value
|
||||
before it ever reaches the ssh command line.
|
||||
"""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "-oProxyCommand=evil"
|
||||
self.assertIsNone(youtube_yt._ytdlp_ssh_host())
|
||||
# And the wrap function falls back to the local-execution path.
|
||||
cmd = ["yt-dlp", "--version"]
|
||||
self.assertEqual(youtube_yt._wrap_ytdlp_cmd(cmd), cmd)
|
||||
|
||||
def test_host_alias_with_shell_metacharacters_is_rejected(self):
|
||||
"""Host values containing spaces, semicolons, $, etc. are rejected."""
|
||||
for bad in ("host;rm -rf /", "host name", "host$IFS", "host`whoami`", "host&cmd"):
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = bad
|
||||
self.assertIsNone(
|
||||
youtube_yt._ytdlp_ssh_host(),
|
||||
msg=f"validator should reject {bad!r}",
|
||||
)
|
||||
|
||||
def test_host_alias_validator_accepts_realistic_aliases(self):
|
||||
"""Valid SSH config aliases are accepted: bare names, FQDNs, IPs."""
|
||||
for good in ("macmini", "home-server", "pi5.local", "192.168.1.10", "homelab_box"):
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = good
|
||||
self.assertEqual(youtube_yt._ytdlp_ssh_host(), good)
|
||||
|
||||
def test_is_ytdlp_installed_short_circuits_with_ssh(self):
|
||||
"""is_ytdlp_installed returns True without local check when SSH routing is on."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "macmini"
|
||||
with mock.patch("lib.youtube_yt.shutil.which", return_value=None) as which_mock:
|
||||
self.assertTrue(youtube_yt.is_ytdlp_installed())
|
||||
which_mock.assert_not_called()
|
||||
|
||||
def test_is_ytdlp_installed_falls_through_without_ssh(self):
|
||||
"""is_ytdlp_installed checks PATH normally when SSH routing is off."""
|
||||
with mock.patch("lib.youtube_yt.shutil.which", return_value="/usr/bin/yt-dlp"):
|
||||
self.assertTrue(youtube_yt.is_ytdlp_installed())
|
||||
with mock.patch("lib.youtube_yt.shutil.which", return_value=None):
|
||||
self.assertFalse(youtube_yt.is_ytdlp_installed())
|
||||
|
||||
def test_search_call_routes_through_ssh(self):
|
||||
"""search_youtube wraps the yt-dlp invocation when SSH routing is on."""
|
||||
os.environ["LAST30DAYS_YOUTUBE_SSH_HOST"] = "macmini"
|
||||
from lib.subproc import SubprocResult
|
||||
fake_result = SubprocResult(returncode=0, stdout="", stderr="")
|
||||
with mock.patch.object(youtube_yt.subproc, "run_with_timeout",
|
||||
return_value=fake_result) as run_mock:
|
||||
youtube_yt.search_youtube("test", "2026-02-01", "2026-03-01")
|
||||
cmd = run_mock.call_args.args[0]
|
||||
self.assertEqual(cmd[0], "ssh")
|
||||
self.assertEqual(cmd[3], "--")
|
||||
self.assertEqual(cmd[4], "macmini")
|
||||
# The shell-quoted yt-dlp invocation lives at index 5
|
||||
self.assertIn("yt-dlp", cmd[5])
|
||||
self.assertIn("--ignore-config", cmd[5])
|
||||
self.assertIn("--no-cookies-from-browser", cmd[5])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -2,88 +2,6 @@ version = 1
|
||||
revision = 3
|
||||
requires-python = ">=3.12"
|
||||
|
||||
[[package]]
|
||||
name = "certifi"
|
||||
version = "2026.2.25"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/af/2d/7bf41579a8986e348fa033a31cdd0e4121114f6bce2457e8876010b092dd/certifi-2026.2.25.tar.gz", hash = "sha256:e887ab5cee78ea814d3472169153c2d12cd43b14bd03329a39a9c6e2e80bfba7", size = 155029, upload-time = "2026-02-25T02:54:17.342Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/9a/3c/c17fb3ca2d9c3acff52e30b309f538586f9f5b9c9cf454f3845fc9af4881/certifi-2026.2.25-py3-none-any.whl", hash = "sha256:027692e4402ad994f1c42e52a4997a9763c646b73e4096e4d5d6db8af1d6f0fa", size = 153684, upload-time = "2026-02-25T02:54:15.766Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "charset-normalizer"
|
||||
version = "3.4.6"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/7b/60/e3bec1881450851b087e301bedc3daa9377a4d45f1c26aa90b0b235e38aa/charset_normalizer-3.4.6.tar.gz", hash = "sha256:1ae6b62897110aa7c79ea2f5dd38d1abca6db663687c0b1ad9aed6f6bae3d9d6", size = 143363, upload-time = "2026-03-15T18:53:25.478Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/e5/62/c0815c992c9545347aeea7859b50dc9044d147e2e7278329c6e02ac9a616/charset_normalizer-3.4.6-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:2ef7fedc7a6ecbe99969cd09632516738a97eeb8bd7258bf8a0f23114c057dab", size = 295154, upload-time = "2026-03-15T18:50:50.88Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a8/37/bdca6613c2e3c58c7421891d80cc3efa1d32e882f7c4a7ee6039c3fc951a/charset_normalizer-3.4.6-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:a4ea868bc28109052790eb2b52a9ab33f3aa7adc02f96673526ff47419490e21", size = 199191, upload-time = "2026-03-15T18:50:52.658Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/6c/92/9934d1bbd69f7f398b38c5dae1cbf9cc672e7c34a4adf7b17c0a9c17d15d/charset_normalizer-3.4.6-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:836ab36280f21fc1a03c99cd05c6b7af70d2697e374c7af0b61ed271401a72a2", size = 218674, upload-time = "2026-03-15T18:50:54.102Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/af/90/25f6ab406659286be929fd89ab0e78e38aa183fc374e03aa3c12d730af8a/charset_normalizer-3.4.6-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:f1ce721c8a7dfec21fcbdfe04e8f68174183cf4e8188e0645e92aa23985c57ff", size = 215259, upload-time = "2026-03-15T18:50:55.616Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/4e/ef/79a463eb0fff7f96afa04c1d4c51f8fc85426f918db467854bfb6a569ce3/charset_normalizer-3.4.6-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0e28d62a8fc7a1fa411c43bd65e346f3bce9716dc51b897fbe930c5987b402d5", size = 207276, upload-time = "2026-03-15T18:50:57.054Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f7/72/d0426afec4b71dc159fa6b4e68f868cd5a3ecd918fec5813a15d292a7d10/charset_normalizer-3.4.6-cp312-cp312-manylinux_2_31_armv7l.whl", hash = "sha256:530d548084c4a9f7a16ed4a294d459b4f229db50df689bfe92027452452943a0", size = 195161, upload-time = "2026-03-15T18:50:58.686Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/bf/18/c82b06a68bfcb6ce55e508225d210c7e6a4ea122bfc0748892f3dc4e8e11/charset_normalizer-3.4.6-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:30f445ae60aad5e1f8bdbb3108e39f6fbc09f4ea16c815c66578878325f8f15a", size = 203452, upload-time = "2026-03-15T18:51:00.196Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/44/d6/0c25979b92f8adafdbb946160348d8d44aa60ce99afdc27df524379875cb/charset_normalizer-3.4.6-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:ac2393c73378fea4e52aa56285a3d64be50f1a12395afef9cce47772f60334c2", size = 202272, upload-time = "2026-03-15T18:51:01.703Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/2e/3d/7fea3e8fe84136bebbac715dd1221cc25c173c57a699c030ab9b8900cbb7/charset_normalizer-3.4.6-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:90ca27cd8da8118b18a52d5f547859cc1f8354a00cd1e8e5120df3e30d6279e5", size = 195622, upload-time = "2026-03-15T18:51:03.526Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/57/8a/d6f7fd5cb96c58ef2f681424fbca01264461336d2a7fc875e4446b1f1346/charset_normalizer-3.4.6-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:8e5a94886bedca0f9b78fecd6afb6629142fd2605aa70a125d49f4edc6037ee6", size = 220056, upload-time = "2026-03-15T18:51:05.269Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/16/50/478cdda782c8c9c3fb5da3cc72dd7f331f031e7f1363a893cdd6ca0f8de0/charset_normalizer-3.4.6-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:695f5c2823691a25f17bc5d5ffe79fa90972cc34b002ac6c843bb8a1720e950d", size = 203751, upload-time = "2026-03-15T18:51:06.858Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/75/fc/cc2fcac943939c8e4d8791abfa139f685e5150cae9f94b60f12520feaa9b/charset_normalizer-3.4.6-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:231d4da14bcd9301310faf492051bee27df11f2bc7549bc0bb41fef11b82daa2", size = 216563, upload-time = "2026-03-15T18:51:08.564Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a8/b7/a4add1d9a5f68f3d037261aecca83abdb0ab15960a3591d340e829b37298/charset_normalizer-3.4.6-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:a056d1ad2633548ca18ffa2f85c202cfb48b68615129143915b8dc72a806a923", size = 209265, upload-time = "2026-03-15T18:51:10.312Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/6c/18/c094561b5d64a24277707698e54b7f67bd17a4f857bbfbb1072bba07c8bf/charset_normalizer-3.4.6-cp312-cp312-win32.whl", hash = "sha256:c2274ca724536f173122f36c98ce188fd24ce3dad886ec2b7af859518ce008a4", size = 144229, upload-time = "2026-03-15T18:51:11.694Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/ab/20/0567efb3a8fd481b8f34f739ebddc098ed062a59fed41a8d193a61939e8f/charset_normalizer-3.4.6-cp312-cp312-win_amd64.whl", hash = "sha256:c8ae56368f8cc97c7e40a7ee18e1cedaf8e780cd8bc5ed5ac8b81f238614facb", size = 154277, upload-time = "2026-03-15T18:51:13.004Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/15/57/28d79b44b51933119e21f65479d0864a8d5893e494cf5daab15df0247c17/charset_normalizer-3.4.6-cp312-cp312-win_arm64.whl", hash = "sha256:899d28f422116b08be5118ef350c292b36fc15ec2daeb9ea987c89281c7bb5c4", size = 142817, upload-time = "2026-03-15T18:51:14.408Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/1e/1d/4fdabeef4e231153b6ed7567602f3b68265ec4e5b76d6024cf647d43d981/charset_normalizer-3.4.6-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:11afb56037cbc4b1555a34dd69151e8e069bee82e613a73bef6e714ce733585f", size = 294823, upload-time = "2026-03-15T18:51:15.755Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/47/7b/20e809b89c69d37be748d98e84dce6820bf663cf19cf6b942c951a3e8f41/charset_normalizer-3.4.6-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:423fb7e748a08f854a08a222b983f4df1912b1daedce51a72bd24fe8f26a1843", size = 198527, upload-time = "2026-03-15T18:51:17.177Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/37/a6/4f8d27527d59c039dce6f7622593cdcd3d70a8504d87d09eb11e9fdc6062/charset_normalizer-3.4.6-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:d73beaac5e90173ac3deb9928a74763a6d230f494e4bfb422c217a0ad8e629bf", size = 218388, upload-time = "2026-03-15T18:51:18.934Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f6/9b/4770ccb3e491a9bacf1c46cc8b812214fe367c86a96353ccc6daf87b01ec/charset_normalizer-3.4.6-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:d60377dce4511655582e300dc1e5a5f24ba0cb229005a1d5c8d0cb72bb758ab8", size = 214563, upload-time = "2026-03-15T18:51:20.374Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/2b/58/a199d245894b12db0b957d627516c78e055adc3a0d978bc7f65ddaf7c399/charset_normalizer-3.4.6-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:530e8cebeea0d76bdcf93357aa5e41336f48c3dc709ac52da2bb167c5b8271d9", size = 206587, upload-time = "2026-03-15T18:51:21.807Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/7e/70/3def227f1ec56f5c69dfc8392b8bd63b11a18ca8178d9211d7cc5e5e4f27/charset_normalizer-3.4.6-cp313-cp313-manylinux_2_31_armv7l.whl", hash = "sha256:a26611d9987b230566f24a0a125f17fe0de6a6aff9f25c9f564aaa2721a5fb88", size = 194724, upload-time = "2026-03-15T18:51:23.508Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/58/ab/9318352e220c05efd31c2779a23b50969dc94b985a2efa643ed9077bfca5/charset_normalizer-3.4.6-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:34315ff4fc374b285ad7f4a0bf7dcbfe769e1b104230d40f49f700d4ab6bbd84", size = 202956, upload-time = "2026-03-15T18:51:25.239Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/75/13/f3550a3ac25b70f87ac98c40d3199a8503676c2f1620efbf8d42095cfc40/charset_normalizer-3.4.6-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:5f8ddd609f9e1af8c7bd6e2aca279c931aefecd148a14402d4e368f3171769fd", size = 201923, upload-time = "2026-03-15T18:51:26.682Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/1b/db/c5c643b912740b45e8eec21de1bbab8e7fc085944d37e1e709d3dcd9d72f/charset_normalizer-3.4.6-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:80d0a5615143c0b3225e5e3ef22c8d5d51f3f72ce0ea6fb84c943546c7b25b6c", size = 195366, upload-time = "2026-03-15T18:51:28.129Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/5a/67/3b1c62744f9b2448443e0eb160d8b001c849ec3fef591e012eda6484787c/charset_normalizer-3.4.6-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:92734d4d8d187a354a556626c221cd1a892a4e0802ccb2af432a1d85ec012194", size = 219752, upload-time = "2026-03-15T18:51:29.556Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f6/98/32ffbaf7f0366ffb0445930b87d103f6b406bc2c271563644bde8a2b1093/charset_normalizer-3.4.6-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:613f19aa6e082cf96e17e3ffd89383343d0d589abda756b7764cf78361fd41dc", size = 203296, upload-time = "2026-03-15T18:51:30.921Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/41/12/5d308c1bbe60cabb0c5ef511574a647067e2a1f631bc8634fcafaccd8293/charset_normalizer-3.4.6-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:2b1a63e8224e401cafe7739f77efd3f9e7f5f2026bda4aead8e59afab537784f", size = 215956, upload-time = "2026-03-15T18:51:32.399Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/53/e9/5f85f6c5e20669dbe56b165c67b0260547dea97dba7e187938833d791687/charset_normalizer-3.4.6-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:6cceb5473417d28edd20c6c984ab6fee6c6267d38d906823ebfe20b03d607dc2", size = 208652, upload-time = "2026-03-15T18:51:34.214Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f1/11/897052ea6af56df3eef3ca94edafee410ca699ca0c7b87960ad19932c55e/charset_normalizer-3.4.6-cp313-cp313-win32.whl", hash = "sha256:d7de2637729c67d67cf87614b566626057e95c303bc0a55ffe391f5205e7003d", size = 143940, upload-time = "2026-03-15T18:51:36.15Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a1/5c/724b6b363603e419829f561c854b87ed7c7e31231a7908708ac086cdf3e2/charset_normalizer-3.4.6-cp313-cp313-win_amd64.whl", hash = "sha256:572d7c822caf521f0525ba1bce1a622a0b85cf47ffbdae6c9c19e3b5ac3c4389", size = 154101, upload-time = "2026-03-15T18:51:37.876Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/01/a5/7abf15b4c0968e47020f9ca0935fb3274deb87cb288cd187cad92e8cdffd/charset_normalizer-3.4.6-cp313-cp313-win_arm64.whl", hash = "sha256:a4474d924a47185a06411e0064b803c68be044be2d60e50e8bddcc2649957c1f", size = 143109, upload-time = "2026-03-15T18:51:39.565Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/25/6f/ffe1e1259f384594063ea1869bfb6be5cdb8bc81020fc36c3636bc8302a1/charset_normalizer-3.4.6-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:9cc6e6d9e571d2f863fa77700701dae73ed5f78881efc8b3f9a4398772ff53e8", size = 294458, upload-time = "2026-03-15T18:51:41.134Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/56/60/09bb6c13a8c1016c2ed5c6a6488e4ffef506461aa5161662bd7636936fb1/charset_normalizer-3.4.6-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ef5960d965e67165d75b7c7ffc60a83ec5abfc5c11b764ec13ea54fbef8b4421", size = 199277, upload-time = "2026-03-15T18:51:42.953Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/00/50/dcfbb72a5138bbefdc3332e8d81a23494bf67998b4b100703fd15fa52d81/charset_normalizer-3.4.6-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:b3694e3f87f8ac7ce279d4355645b3c878d24d1424581b46282f24b92f5a4ae2", size = 218758, upload-time = "2026-03-15T18:51:44.339Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/03/b3/d79a9a191bb75f5aa81f3aaaa387ef29ce7cb7a9e5074ba8ea095cc073c2/charset_normalizer-3.4.6-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:5d11595abf8dd942a77883a39d81433739b287b6aa71620f15164f8096221b30", size = 215299, upload-time = "2026-03-15T18:51:45.871Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/76/7e/bc8911719f7084f72fd545f647601ea3532363927f807d296a8c88a62c0d/charset_normalizer-3.4.6-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:7bda6eebafd42133efdca535b04ccb338ab29467b3f7bf79569883676fc628db", size = 206811, upload-time = "2026-03-15T18:51:47.308Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/e2/40/c430b969d41dda0c465aa36cc7c2c068afb67177bef50905ac371b28ccc7/charset_normalizer-3.4.6-cp314-cp314-manylinux_2_31_armv7l.whl", hash = "sha256:bbc8c8650c6e51041ad1be191742b8b421d05bbd3410f43fa2a00c8db87678e8", size = 193706, upload-time = "2026-03-15T18:51:48.849Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/48/15/e35e0590af254f7df984de1323640ef375df5761f615b6225ba8deb9799a/charset_normalizer-3.4.6-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:22c6f0c2fbc31e76c3b8a86fba1a56eda6166e238c29cdd3d14befdb4a4e4815", size = 202706, upload-time = "2026-03-15T18:51:50.257Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/5e/bd/f736f7b9cc5e93a18b794a50346bb16fbfd6b37f99e8f306f7951d27c17c/charset_normalizer-3.4.6-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:7edbed096e4a4798710ed6bc75dcaa2a21b68b6c356553ac4823c3658d53743a", size = 202497, upload-time = "2026-03-15T18:51:52.012Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/9d/ba/2cc9e3e7dfdf7760a6ed8da7446d22536f3d0ce114ac63dee2a5a3599e62/charset_normalizer-3.4.6-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:7f9019c9cb613f084481bd6a100b12e1547cf2efe362d873c2e31e4035a6fa43", size = 193511, upload-time = "2026-03-15T18:51:53.723Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/9e/cb/5be49b5f776e5613be07298c80e1b02a2d900f7a7de807230595c85a8b2e/charset_normalizer-3.4.6-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:58c948d0d086229efc484fe2f30c2d382c86720f55cd9bc33591774348ad44e0", size = 220133, upload-time = "2026-03-15T18:51:55.333Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/83/43/99f1b5dad345accb322c80c7821071554f791a95ee50c1c90041c157ae99/charset_normalizer-3.4.6-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:419a9d91bd238052642a51938af8ac05da5b3343becde08d5cdeab9046df9ee1", size = 203035, upload-time = "2026-03-15T18:51:56.736Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/87/9a/62c2cb6a531483b55dddff1a68b3d891a8b498f3ca555fbcf2978e804d9d/charset_normalizer-3.4.6-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:5273b9f0b5835ff0350c0828faea623c68bfa65b792720c453e22b25cc72930f", size = 216321, upload-time = "2026-03-15T18:51:58.17Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/6e/79/94a010ff81e3aec7c293eb82c28f930918e517bc144c9906a060844462eb/charset_normalizer-3.4.6-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:0e901eb1049fdb80f5bd11ed5ea1e498ec423102f7a9b9e4645d5b8204ff2815", size = 208973, upload-time = "2026-03-15T18:51:59.998Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/2a/57/4ecff6d4ec8585342f0c71bc03efaa99cb7468f7c91a57b105bcd561cea8/charset_normalizer-3.4.6-cp314-cp314-win32.whl", hash = "sha256:b4ff1d35e8c5bd078be89349b6f3a845128e685e751b6ea1169cf2160b344c4d", size = 144610, upload-time = "2026-03-15T18:52:02.213Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/80/94/8434a02d9d7f168c25767c64671fead8d599744a05d6a6c877144c754246/charset_normalizer-3.4.6-cp314-cp314-win_amd64.whl", hash = "sha256:74119174722c4349af9708993118581686f343adc1c8c9c007d59be90d077f3f", size = 154962, upload-time = "2026-03-15T18:52:03.658Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/46/4c/48f2cdbfd923026503dfd67ccea45c94fd8fe988d9056b468579c66ed62b/charset_normalizer-3.4.6-cp314-cp314-win_arm64.whl", hash = "sha256:e5bcc1a1ae744e0bb59641171ae53743760130600da8db48cbb6e4918e186e4e", size = 143595, upload-time = "2026-03-15T18:52:05.123Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/31/93/8878be7569f87b14f1d52032946131bcb6ebbd8af3e20446bc04053dc3f1/charset_normalizer-3.4.6-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:ad8faf8df23f0378c6d527d8b0b15ea4a2e23c89376877c598c4870d1b2c7866", size = 314828, upload-time = "2026-03-15T18:52:06.831Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/06/b6/fae511ca98aac69ecc35cde828b0a3d146325dd03d99655ad38fc2cc3293/charset_normalizer-3.4.6-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f5ea69428fa1b49573eef0cc44a1d43bebd45ad0c611eb7d7eac760c7ae771bc", size = 208138, upload-time = "2026-03-15T18:52:08.239Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/54/57/64caf6e1bf07274a1e0b7c160a55ee9e8c9ec32c46846ce59b9c333f7008/charset_normalizer-3.4.6-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:06a7e86163334edfc5d20fe104db92fcd666e5a5df0977cb5680a506fe26cc8e", size = 224679, upload-time = "2026-03-15T18:52:10.043Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/aa/cb/9ff5a25b9273ef160861b41f6937f86fae18b0792fe0a8e75e06acb08f1d/charset_normalizer-3.4.6-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:e1f6e2f00a6b8edb562826e4632e26d063ac10307e80f7461f7de3ad8ef3f077", size = 223475, upload-time = "2026-03-15T18:52:11.854Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/fc/97/440635fc093b8d7347502a377031f9605a1039c958f3cd18dcacffb37743/charset_normalizer-3.4.6-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:95b52c68d64c1878818687a473a10547b3292e82b6f6fe483808fb1468e2f52f", size = 215230, upload-time = "2026-03-15T18:52:13.325Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/cd/24/afff630feb571a13f07c8539fbb502d2ab494019492aaffc78ef41f1d1d0/charset_normalizer-3.4.6-cp314-cp314t-manylinux_2_31_armv7l.whl", hash = "sha256:7504e9b7dc05f99a9bbb4525c67a2c155073b44d720470a148b34166a69c054e", size = 199045, upload-time = "2026-03-15T18:52:14.752Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/e5/17/d1399ecdaf7e0498c327433e7eefdd862b41236a7e484355b8e0e5ebd64b/charset_normalizer-3.4.6-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:172985e4ff804a7ad08eebec0a1640ece87ba5041d565fff23c8f99c1f389484", size = 211658, upload-time = "2026-03-15T18:52:16.278Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/b5/38/16baa0affb957b3d880e5ac2144caf3f9d7de7bc4a91842e447fbb5e8b67/charset_normalizer-3.4.6-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:4be9f4830ba8741527693848403e2c457c16e499100963ec711b1c6f2049b7c7", size = 210769, upload-time = "2026-03-15T18:52:17.782Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/05/34/c531bc6ac4c21da9ddfddb3107be2287188b3ea4b53b70fc58f2a77ac8d8/charset_normalizer-3.4.6-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:79090741d842f564b1b2827c0b82d846405b744d31e84f18d7a7b41c20e473ff", size = 201328, upload-time = "2026-03-15T18:52:19.553Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/fa/73/a5a1e9ca5f234519c1953608a03fe109c306b97fdfb25f09182babad51a7/charset_normalizer-3.4.6-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:87725cfb1a4f1f8c2fc9890ae2f42094120f4b44db9360be5d99a4c6b0e03a9e", size = 225302, upload-time = "2026-03-15T18:52:21.043Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/ba/f6/cd782923d112d296294dea4bcc7af5a7ae0f86ab79f8fefbda5526b6cfc0/charset_normalizer-3.4.6-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:fcce033e4021347d80ed9c66dcf1e7b1546319834b74445f561d2e2221de5659", size = 211127, upload-time = "2026-03-15T18:52:22.491Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/0e/c5/0b6898950627af7d6103a449b22320372c24c6feda91aa24e201a478d161/charset_normalizer-3.4.6-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:ca0276464d148c72defa8bb4390cce01b4a0e425f3b50d1435aa6d7a18107602", size = 222840, upload-time = "2026-03-15T18:52:24.113Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/7d/25/c4bba773bef442cbdc06111d40daa3de5050a676fa26e85090fc54dd12f0/charset_normalizer-3.4.6-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:197c1a244a274bb016dd8b79204850144ef77fe81c5b797dc389327adb552407", size = 216890, upload-time = "2026-03-15T18:52:25.541Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/35/1a/05dacadb0978da72ee287b0143097db12f2e7e8d3ffc4647da07a383b0b7/charset_normalizer-3.4.6-cp314-cp314t-win32.whl", hash = "sha256:2a24157fa36980478dd1770b585c0f30d19e18f4fb0c47c13aa568f871718579", size = 155379, upload-time = "2026-03-15T18:52:27.05Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/5d/7a/d269d834cb3a76291651256f3b9a5945e81d0a49ab9f4a498964e83c0416/charset_normalizer-3.4.6-cp314-cp314t-win_amd64.whl", hash = "sha256:cd5e2801c89992ed8c0a3f0293ae83c159a60d9a5d685005383ef4caca77f2c4", size = 169043, upload-time = "2026-03-15T18:52:28.502Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/23/06/28b29fba521a37a8932c6a84192175c34d49f84a6d4773fa63d05f9aff22/charset_normalizer-3.4.6-cp314-cp314t-win_arm64.whl", hash = "sha256:47955475ac79cc504ef2704b192364e51d0d473ad452caedd0002605f780101c", size = 148523, upload-time = "2026-03-15T18:52:29.956Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/2a/68/687187c7e26cb24ccbd88e5069f5ef00eba804d36dde11d99aad0838ab45/charset_normalizer-3.4.6-py3-none-any.whl", hash = "sha256:947cf925bc916d90adba35a64c82aace04fa39b46b52d4630ece166655905a69", size = 61455, upload-time = "2026-03-15T18:53:23.833Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "colorama"
|
||||
version = "0.4.6"
|
||||
@@ -177,15 +95,6 @@ wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/9e/ee/a4cf96b8ce1e566ed238f0659ac2d3f007ed1d14b181bcb684e19561a69a/coverage-7.13.5-py3-none-any.whl", hash = "sha256:34b02417cf070e173989b3db962f7ed56d2f644307b2cf9d5a0f258e13084a61", size = 211346, upload-time = "2026-03-17T10:33:15.691Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "idna"
|
||||
version = "3.11"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/6f/6d/0703ccc57f3a7233505399edb88de3cbd678da106337b9fcde432b65ed60/idna-3.11.tar.gz", hash = "sha256:795dafcc9c04ed0c1fb032c2aa73654d8e8c5023a7df64a53f39190ada629902", size = 194582, upload-time = "2025-10-12T14:55:20.501Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/0e/61/66938bbb5fc52dbdf84594873d5b51fb1f7c7794e9c0f5bd885f30bc507b/idna-3.11-py3-none-any.whl", hash = "sha256:771a87f49d9defaf64091e6e6fe9c18d4833f140bd19464795bc32d966ca37ea", size = 71008, upload-time = "2025-10-12T14:55:18.883Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "iniconfig"
|
||||
version = "2.3.0"
|
||||
@@ -197,11 +106,8 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "last30days-skill"
|
||||
version = "3.1.1"
|
||||
version = "3.2.4"
|
||||
source = { virtual = "." }
|
||||
dependencies = [
|
||||
{ name = "requests" },
|
||||
]
|
||||
|
||||
[package.dev-dependencies]
|
||||
dev = [
|
||||
@@ -210,11 +116,10 @@ dev = [
|
||||
]
|
||||
|
||||
[package.metadata]
|
||||
requires-dist = [{ name = "requests", specifier = ">=2.32,<3" }]
|
||||
|
||||
[package.metadata.requires-dev]
|
||||
dev = [
|
||||
{ name = "pytest", specifier = ">=9,<10" },
|
||||
{ name = "pytest", specifier = ">=9.0.3,<10" },
|
||||
{ name = "pytest-cov", specifier = ">=7,<8" },
|
||||
]
|
||||
|
||||
@@ -247,7 +152,7 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "pytest"
|
||||
version = "9.0.2"
|
||||
version = "9.0.3"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
dependencies = [
|
||||
{ name = "colorama", marker = "sys_platform == 'win32'" },
|
||||
@@ -256,9 +161,9 @@ dependencies = [
|
||||
{ name = "pluggy" },
|
||||
{ name = "pygments" },
|
||||
]
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/d1/db/7ef3487e0fb0049ddb5ce41d3a49c235bf9ad299b6a25d5780a89f19230f/pytest-9.0.2.tar.gz", hash = "sha256:75186651a92bd89611d1d9fc20f0b4345fd827c41ccd5c299a868a05d70edf11", size = 1568901, upload-time = "2025-12-06T21:30:51.014Z" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/7d/0d/549bd94f1a0a402dc8cf64563a117c0f3765662e2e668477624baeec44d5/pytest-9.0.3.tar.gz", hash = "sha256:b86ada508af81d19edeb213c681b1d48246c1a91d304c6c81a427674c17eb91c", size = 1572165, upload-time = "2026-04-07T17:16:18.027Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/3b/ab/b3226f0bd7cdcf710fbede2b3548584366da3b19b5021e74f5bde2a8fa3f/pytest-9.0.2-py3-none-any.whl", hash = "sha256:711ffd45bf766d5264d487b917733b453d917afd2b0ad65223959f59089f875b", size = 374801, upload-time = "2025-12-06T21:30:49.154Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/d4/24/a372aaf5c9b7208e7112038812994107bc65a84cd00e0354a88c2c77a617/pytest-9.0.3-py3-none-any.whl", hash = "sha256:2c5efc453d45394fdd706ade797c0a81091eccd1d6e4bccfcd476e2b8e0ab5d9", size = 375249, upload-time = "2026-04-07T17:16:16.13Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -274,27 +179,3 @@ sdist = { url = "https://files.pythonhosted.org/packages/b1/51/a849f96e117386044
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/9d/7a/d968e294073affff457b041c2be9868a40c1c71f4a35fcc1e45e5493067b/pytest_cov-7.1.0-py3-none-any.whl", hash = "sha256:a0461110b7865f9a271aa1b51e516c9a95de9d696734a2f71e3e78f46e1d4678", size = 22876, upload-time = "2026-03-21T20:11:14.438Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "requests"
|
||||
version = "2.33.0"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
dependencies = [
|
||||
{ name = "certifi" },
|
||||
{ name = "charset-normalizer" },
|
||||
{ name = "idna" },
|
||||
{ name = "urllib3" },
|
||||
]
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/34/64/8860370b167a9721e8956ae116825caff829224fbca0ca6e7bf8ddef8430/requests-2.33.0.tar.gz", hash = "sha256:c7ebc5e8b0f21837386ad0e1c8fe8b829fa5f544d8df3b2253bff14ef29d7652", size = 134232, upload-time = "2026-03-25T15:10:41.586Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/56/5d/c814546c2333ceea4ba42262d8c4d55763003e767fa169adc693bd524478/requests-2.33.0-py3-none-any.whl", hash = "sha256:3324635456fa185245e24865e810cecec7b4caf933d7eb133dcde67d48cee69b", size = 65017, upload-time = "2026-03-25T15:10:40.382Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "urllib3"
|
||||
version = "2.6.3"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/c7/24/5f1b3bdffd70275f6661c76461e25f024d5a38a46f04aaca912426a2b1d3/urllib3-2.6.3.tar.gz", hash = "sha256:1b62b6884944a57dbe321509ab94fd4d3b307075e0c2eae991ac71ee15ad38ed", size = 435556, upload-time = "2026-01-07T16:24:43.925Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/39/08/aaaad47bc4e9dc8c725e68f9d04865dbcb2052843ff09c97b08904852d84/urllib3-2.6.3-py3-none-any.whl", hash = "sha256:bf272323e553dfb2e87d9bfd225ca7b0f467b919d7bbd355436d3fd37cb0acd4", size = 131584, upload-time = "2026-01-07T16:24:42.685Z" },
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user