diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..dc936f4 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +__pycache__/ +*.pyc +*.pyo +*.egg-info/ +.venv/ +runs/*.json diff --git a/config/settings.yaml b/config/settings.yaml index cb54326..b203e6c 100644 --- a/config/settings.yaml +++ b/config/settings.yaml @@ -16,18 +16,31 @@ llm: api_key: "" max_tokens: 8000 +# Secondary LLM for pipeline tasks — uses Ollama on 3060 (non-reasoning model) +llm_pipeline: + base_url: http://100.64.0.4:11434 + model: qwen2.5:7b + api_key: "" + max_tokens: 6000 + scout: queries: - - 'agent framework langgraph mcp multi-agent' - - 'ai workflow agent pipeline rag pipeline' - - 'llm orchestration tool-use tool calling' + - 'langchain workflow example' + - 'langgraph agent workflow' + - 'autogen multi-agent example' + - 'crewai task workflow' + - 'llamaindex pipeline example' + - 'mcp server implementation' + - 'rag agent workflow' + - 'tool calling workflow' filters: - stars_min: 10 - pushed_after: 2026-06-01 + stars_min: 15 + pushed_after: 2026-05-01 language: Python archived: false - max_results: 30 - cooldown_hours: 24 + size_max_kb: 10000 + max_results: 15 + cooldown_hours: 6 filter: categories: diff --git a/pipeline/extractor.py b/pipeline/extractor.py index 66bb95c..3c38886 100644 --- a/pipeline/extractor.py +++ b/pipeline/extractor.py @@ -6,37 +6,61 @@ import re def call_llm(prompt, config): - """Call the configured LLM for extraction.""" - llm_config = config.get("llm", {}) + """Call the configured LLM for extraction/review.""" + llm_config = config.get("llm_pipeline", config.get("llm", {})) base_url = llm_config.get("base_url", "http://100.64.0.2:8083/v1") model = llm_config.get("model", "") api_key = llm_config.get("api_key", "") max_tokens = llm_config.get("max_tokens", 8000) - headers = { - "Content-Type": "application/json", - } - if api_key: - headers["Authorization"] = f"Bearer {api_key}" + # Detect Ollama native API (11434 port) — use /api/chat instead of /v1/chat/completions + is_ollama_native = ":11434" in base_url - payload = { - "model": model, - "messages": [ - {"role": "system", "content": prompt}, - ], - "max_tokens": max_tokens, - "temperature": 0.1, - } + if is_ollama_native: + payload = { + "model": model, + "messages": [{"role": "user", "content": prompt}], + "stream": False, + "options": {"num_predict": max_tokens, "temperature": 0.1}, + } + try: + resp = requests.post(f"{base_url}/api/chat", json=payload, headers={"Content-Type": "application/json"}, timeout=120) + if resp.status_code == 200: + return resp.json().get("message", {}).get("content", "") + else: + return f"LLM error: {resp.status_code}" + except Exception as e: + return f"LLM error: {str(e)}" + else: + # OpenAI-compatible format + headers = {"Content-Type": "application/json"} + if api_key: + headers["Authorization"] = f"Bearer {api_key}" - try: - resp = requests.post(f"{base_url}/v1/chat/completions", json=payload, headers=headers, timeout=120) - if resp.status_code == 200: - data = resp.json() - return data["choices"][0]["message"]["content"] - else: - return f"LLM error: {resp.status_code} {resp.text[:200]}" - except Exception as e: - return f"LLM error: {str(e)}" + payload = { + "model": model, + "messages": [{"role": "system", "content": prompt}], + "max_tokens": max_tokens, + "temperature": 0.1, + } + + try: + resp = requests.post(f"{base_url}/v1/chat/completions", json=payload, headers=headers, timeout=120) + if resp.status_code == 200: + data = resp.json() + msg = data["choices"][0]["message"] + content = (msg.get("content") or msg.get("reasoning_content") or "").strip() + if not content and msg.get("reasoning_content"): + rc = msg["reasoning_content"] + import re + json_match = re.search(r'(\{.*\})', rc, re.DOTALL) + if json_match: + content = json_match.group() + return content + else: + return f"LLM error: {resp.status_code} {resp.text[:200]}" + except Exception as e: + return f"LLM error: {str(e)}" def extract_workflow(reader_output, config): @@ -54,21 +78,32 @@ def extract_workflow(reader_output, config): context = "\n\n".join(context_parts) - prompt = f"""You are a workflow extractor. Your job is to analyze a GitHub repository and determine if it contains a reusable AI workflow or pattern that another agent could learn from. + prompt = f"""You are a workflow extractor. Analyze a GitHub repository and determine if it contains a reusable AI workflow or pattern that another agent could learn from and actually implement. If the repository contains a reusable workflow, extract it into this exact JSON structure: {{ "has_workflow": true, "skill_name": "short-descriptive-name", "goal": "One sentence: what this workflow accomplishes", - "inputs": ["Input 1", "Input 2"], - "steps": ["Step 1", "Step 2", "Step 3"], - "outputs": ["Output 1", "Output 2"], - "failure_modes": ["What can go wrong"], + "inputs": ["Input 1 with type description", "Input 2 with type description"], + "steps": [ + "Step 1: Describe the specific action, mentioning the exact tool/function/file used (e.g. 'Run langgraph chain with agent.py')", + "Step 2: ...", + "Step 3: ..." + ], + "outputs": ["Output 1 with description", "Output 2 with description"], + "failure_modes": ["Specific failure scenario with mitigation"], "confidence": 0.95, "reusable": true, - "general_purpose": true, - "explanation": "Why this is reusable and general-purpose" + "general_purpose": false, + "explanation": "Why this is reusable", + "implementation_details": {{ + "framework": "e.g. langchain, langgraph, autogen, crewai, custom", + "dependencies": ["python-packages-needed"], + "key_files": ["path/to/key_file.py - description"], + "code_snippets": ["Brief but concrete code or config example from the repo"], + "setup_steps": ["Prerequisite setup commands or configs"] + }} }} If the repository does NOT contain a reusable workflow, return: @@ -77,12 +112,15 @@ If the repository does NOT contain a reusable workflow, return: "reason": "Why no reusable workflow was found" }} +CRITICAL: Steps must be SPECIFIC — mention actual file names, function calls, tool names, or configuration details from the repository. A step like 'Researcher agent gathers facts' is too vague. Instead: 'Researcher agent (agent.py) uses LangGraph create_react_agent with SerperDevTool to gather facts.' + Criteria for a reusable workflow: -- It describes a process or pattern, not just a tool or library +- It describes a concrete process, not just a tool or library +- Steps mention specific implementations from the code - It has clear inputs, steps, and outputs -- It could be applied to different contexts outside this specific repo +- It could be adapted to different contexts - It has at least 3 distinct steps -- It solves a real problem, not a toy example +- It solves a real problem Repository: {repo} diff --git a/pipeline/generator.py b/pipeline/generator.py index 143d072..26a2b45 100644 --- a/pipeline/generator.py +++ b/pipeline/generator.py @@ -32,31 +32,78 @@ def generate_skill(score_result, config): }, } + # Build SKILL.md with implementation details + impl = workflow.get("implementation_details", {}) + framework = impl.get("framework", "") + dependencies = impl.get("dependencies", []) + key_files = impl.get("key_files", []) + code_snippets = impl.get("code_snippets", []) + setup_steps = impl.get("setup_steps", []) + skill_md = "---\n" skill_md += yaml.dump(frontmatter, default_flow_style=False, sort_keys=False) skill_md += "---\n\n" skill_md += f"# {skill_name}\n\n" skill_md += f"{workflow.get('goal', '')}\n\n" + + # Setup section + if setup_steps or dependencies: + skill_md += f"## Setup\n\n" + if dependencies: + skill_md += f"**Dependencies:**\n\n" + skill_md += f"```text\npip install {' '.join(dependencies)}\n```\n\n" + if setup_steps: + skill_md += f"**Setup steps:**\n\n" + for s in setup_steps: + skill_md += f"1. {s}\n" + skill_md += "\n" + + # Key files + if key_files: + skill_md += f"## Key Files\n\n" + for kf in key_files: + skill_md += f"- `{kf}`\n" + skill_md += "\n" + + # Steps with implementation details skill_md += f"## Steps\n\n" for i, step in enumerate(workflow.get("steps", []), 1): skill_md += f"{i}. {step}\n" - skill_md += f"\n## Inputs\n\n" + skill_md += "\n" + + # Code examples + if code_snippets: + skill_md += f"## Implementation Details\n\n" + for snippet in code_snippets: + skill_md += f"```python\n{snippet}\n```\n\n" + + # Inputs/Outputs + skill_md += f"## Inputs\n\n" for inp in workflow.get("inputs", []): skill_md += f"- {inp}\n" skill_md += f"\n## Outputs\n\n" for out in workflow.get("outputs", []): skill_md += f"- {out}\n" + + # Failure Modes skill_md += f"\n## Failure Modes\n\n" for fm in workflow.get("failure_modes", []): skill_md += f"- {fm}\n" + + # Source skill_md += f"\n## Source\n\n" skill_md += f"Extracted from: [{repo}]({repo})\n" skill_md += f"Confidence: {workflow.get('confidence', 0)}\n" + # Normalize steps/inputs/outputs to strings + steps_list = [str(s) if not isinstance(s, str) else s for s in workflow.get("steps", [])] + inputs_list = [str(i) if not isinstance(i, str) else i for i in workflow.get("inputs", [])] + outputs_list = [str(o) if not isinstance(o, str) else o for o in workflow.get("outputs", [])] + # Generate examples.md examples_md = f"# Examples: {skill_name}\n\n" examples_md += f"## Usage Example\n\n" - examples_md += f"```python\n# How to use this skill\n# Inputs: {', '.join(workflow.get('inputs', []))}\n# Process: {' → '.join(workflow.get('steps', [])[:3])}\n# Outputs: {', '.join(workflow.get('outputs', []))}\n```\n" + examples_md += f"```python\n# How to use this skill\n# Inputs: {', '.join(inputs_list)}\n# Process: {' → '.join(steps_list[:3])}\n# Outputs: {', '.join(outputs_list)}\n```\n" # Generate commands.md commands_md = f"# Commands: {skill_name}\n\n" diff --git a/pipeline/publisher.py b/pipeline/publisher.py index 69391da..f74bdf5 100644 --- a/pipeline/publisher.py +++ b/pipeline/publisher.py @@ -116,7 +116,7 @@ def publish_skill(review_result, config): } resp = requests.post(pr_url, json=pr_payload, headers=headers, timeout=15) - if resp.status_code == 200: + if resp.status_code in (200, 201): pr_data = resp.json() return { "status": "PUBLISHED", @@ -126,6 +126,15 @@ def publish_skill(review_result, config): "pr_number": pr_data.get("index", ""), "message": f"PR opened: {pr_data.get('html_url', '')}", } + elif resp.status_code == 409: + # PR already exists for this branch + return { + "status": "PUBLISHED", + "skill_name": skill_name, + "branch": branch_name, + "pr_url": f"{base_url}/{owner}/{repo_name}/pulls", + "message": f"PR already exists for branch {branch_name}", + } else: return { "status": "PR_ERROR", diff --git a/pipeline/reader.py b/pipeline/reader.py index 39591f6..b1ecd74 100644 --- a/pipeline/reader.py +++ b/pipeline/reader.py @@ -4,25 +4,90 @@ import tempfile import os import json -# Loading order: README → docs/ → examples/ → package.json → requirements.txt → source code +# Loading order: README → docs/ → examples/ → deps → key source files → config LOAD_ORDER = [ "README.md", "README", "readme.md", "docs/README.md", "docs/workflows.md", "docs/guide.md", "docs/architecture.md", "examples/", "example/", "demo/", "package.json", "requirements.txt", "setup.py", "pyproject.toml", "Cargo.toml", + # Key implementation files — actual workflow code, not just docs + "main.py", "app.py", "__main__.py", + "agent.py", "workflow.py", "pipeline.py", "chain.py", + "src/main.py", "src/agent.py", "src/workflow.py", "src/app.py", + "src/agent/__init__.py", "src/workflow/__init__.py", "src/pipeline/__init__.py", + # Config / template files with implementation details + "config.yaml", "config.yml", "config.json", + "settings.yaml", "settings.yml", "settings.json", + ".env.example", "example_config.yaml", "config.example.yaml", + "template.yaml", "template.json", + # TypeScript equivalents + "src/index.ts", "src/main.ts", "src/agent.ts", "src/workflow.ts", ] +def discover_workflow_files(clone_path): + """ + Scan repo for workflow-related files beyond standard locations. + Targets: agents/, workflows/, examples/, scripts/, notebooks/ directories. + Returns list of relative paths to load. + """ + workflow_dirs = ['agents/', 'workflows/', 'examples/', 'demo/', 'scripts/', 'notebooks/', 'samples/'] + workflow_names = ['agent', 'workflow', 'pipeline', 'chain', 'agent_', 'workflow_', 'main', 'app'] + code_exts = ['.py', '.js', '.ts', '.yaml', '.yml', '.json'] + found = [] + + for wdir in workflow_dirs: + dirpath = os.path.join(clone_path, wdir) + if not os.path.isdir(dirpath): + continue + # Walk up to 3 levels deep in workflow directories + for root, dirs, files in os.walk(dirpath): + # Limit depth + depth = os.path.relpath(root, dirpath).count(os.sep) + if depth > 2: + dirs.clear() + continue + for fname in sorted(files): + if fname.lower().endswith(tuple(code_exts)): + if any(name in fname.lower() for name in workflow_names): + rel = os.path.relpath(os.path.join(root, fname), clone_path) + found.append(rel) + elif fname in ('agent.py', 'app.py', 'main.py', 'workflow.py', 'pipeline.py'): + rel = os.path.relpath(os.path.join(root, fname), clone_path) + found.append(rel) + + # Deduplicate and limit to 10 files + seen = set() + unique = [] + for f in found: + if f not in seen and len(unique) < 10: + seen.add(f) + unique.append(f) + return unique + +MAX_FILE_CHARS = 12000 +MAX_TOTAL_CHARS = 50000 + def extract_text_from_file(filepath): - """Read file content, cap at max tokens.""" + """Read file content, cap at max chars.""" try: with open(filepath, 'r', errors='ignore') as f: content = f.read() - if len(content) > 40000: - content = content[:40000] + "\n\n... [truncated] ..." + if len(content) > MAX_FILE_CHARS: + content = content[:MAX_FILE_CHARS] + "\n\n... [truncated] ..." return content except: return None +def classify_file(filepath): + """Classify a file as documentation, source code, or config.""" + name = filepath.lower() + if any(name.endswith(ext) for ext in ['.py', '.js', '.ts', '.go', '.rs', '.java', '.rb']): + return 'source' + elif any(name.endswith(ext) for ext in ['.yaml', '.yml', '.json', '.toml', '.ini', '.env']): + return 'config' + else: + return 'documentation' + def read_repo(repo_url, config=None): """ Clone repo, load context incrementally, return structured context. @@ -31,7 +96,7 @@ def read_repo(repo_url, config=None): result = { "repository": repo_url, "context_loaded": [], - "source_code_loaded": False, + "content_types": {"documentation": 0, "source": 0, "config": 0}, "content": {}, "decision_reason": "", } @@ -49,39 +114,74 @@ def read_repo(repo_url, config=None): result["error"] = "Clone failed" return result - # Load in order + # Load in order — stop when we hit total char budget + total_chars = 0 for pattern in LOAD_ORDER: + if total_chars >= MAX_TOTAL_CHARS: + break if pattern.endswith("/"): # Directory — scan for relevant files dirpath = os.path.join(clone_path, pattern) if os.path.isdir(dirpath): for fname in sorted(os.listdir(dirpath))[:5]: + if total_chars >= MAX_TOTAL_CHARS: + break fpath = os.path.join(dirpath, fname) if os.path.isfile(fpath) and fname.endswith(('.md', '.py', '.js', '.ts', '.yaml', '.yml')): content = extract_text_from_file(fpath) if content and len(content.strip()) > 50: - result["content"][f"{pattern}{fname}"] = content - result["context_loaded"].append(f"{pattern}{fname}") + key = f"{pattern}{fname}" + ftype = classify_file(fpath) + result["content"][key] = content + result["context_loaded"].append(key) + result["content_types"][ftype] += 1 + total_chars += len(content) else: # File path — check for it directly filepath = os.path.join(clone_path, pattern) if os.path.exists(filepath) and os.path.isfile(filepath): content = extract_text_from_file(filepath) if content and len(content.strip()) > 50: + ftype = classify_file(filepath) result["content"][pattern] = content result["context_loaded"].append(pattern) + result["content_types"][ftype] += 1 + total_chars += len(content) - # Check if we have enough to proceed - total_chars = sum(len(v) for v in result["content"].values()) + # Also discover workflow files in nested directories + discovered = discover_workflow_files(clone_path) + for pattern in discovered: + if total_chars >= MAX_TOTAL_CHARS: + break + filepath = os.path.join(clone_path, pattern) + if os.path.exists(filepath) and os.path.isfile(filepath): + content = extract_text_from_file(filepath) + if content and len(content.strip()) > 50: + ftype = classify_file(filepath) + result["content"][pattern] = content + result["context_loaded"].append(pattern) + result["content_types"][ftype] += 1 + total_chars += len(content) + + # Check if we have enough to proceed — need docs AND ideally some source + docs = result["content_types"]["documentation"] + source = result["content_types"]["source"] + config_count = result["content_types"]["config"] if len(result["context_loaded"]) == 0: - result["decision_reason"] = "No readable documentation found" + result["decision_reason"] = "No readable content found" + result["status"] = "INSUFFICIENT" + elif docs == 0: + result["decision_reason"] = "No documentation found" result["status"] = "INSUFFICIENT" elif total_chars < 200: - result["decision_reason"] = "Too little content to extract workflow" + result["decision_reason"] = "Too little content" result["status"] = "INSUFFICIENT" else: - result["decision_reason"] = f"Workflow identified from {len(result['context_loaded'])} files ({total_chars} chars)" + detail = f"Loaded {docs} docs, {source} source, {config_count} config files ({total_chars} chars)" + if source > 0 or config_count > 0: + detail += " — includes implementation details" + result["decision_reason"] = detail result["status"] = "READY" return result diff --git a/pipeline/reviewer.py b/pipeline/reviewer.py index 2e9ca54..c84d286 100644 --- a/pipeline/reviewer.py +++ b/pipeline/reviewer.py @@ -1,11 +1,19 @@ -"""Stage 7: Reviewer — LLM review of generated skill.""" -import json -from pipeline.extractor import call_llm +"""Stage 7: Reviewer — Deterministic structural checks on generated skill.""" +import re + def review_skill(generator_output, config): """ - Review a generated skill. Generation and review are separated. - The reviewer never modifies — only approves or rejects with feedback. + Deterministic review of generated skill. No LLM involved. + + Checks that the SKILL.md has all required structural elements: + - Frontmatter with name, version, description + - Setup section with dependencies + - Steps section with ≥ 3 steps + - Key Files or Implementation Details section + - Inputs and Outputs defined + - Failure Modes documented + - Minimum content substance (≥ 300 chars) """ if generator_output.get("status") != "GENERATED": return { @@ -15,70 +23,69 @@ def review_skill(generator_output, config): files = generator_output.get("files", {}) skill_md = files.get("SKILL.md", "") + metadata = files.get("metadata.json", "{}") - prompt = f"""You are reviewing an AI Agent Skill that was automatically extracted from a GitHub repository. + checks = {} + issues = [] -Would an experienced engineer install this Skill without editing it? + # 1. Frontmatter exists with required fields + has_frontmatter = skill_md.startswith("---") and "---" in skill_md[3:] + has_name = "name:" in skill_md.split("---")[1] if has_frontmatter else False + has_version = "version:" in skill_md + has_description = "description:" in skill_md + checks["frontmatter_complete"] = has_frontmatter and has_name and has_version and has_description -Answer with ONLY valid JSON in this format: -{{ - "decision": "YES" or "NO", - "confidence": 0.0-1.0, - "reason": "One paragraph explaining your decision", - "missing_assumptions": ["List any unclear steps or assumptions"], - "minimum_changes": ["If NO, list the minimum changes for approval"] -}} + # 2. Has Setup section with dependencies + has_setup = "## Setup" in skill_md or "## Dependencies" in skill_md + has_deps = "pip install" in skill_md or "requirements" in skill_md.lower() or "Dependencies" in skill_md + checks["setup_documented"] = has_setup or has_deps -Skill to review: + # 3. Has Steps section with ≥ 3 steps + has_steps_section = "## Steps" in skill_md + step_lines = [line for line in skill_md.split("\n") if re.match(r"^\d+\.\s", line)] + checks["has_steps"] = has_steps_section and len(step_lines) >= 3 -{skill_md} + # 4. Has Key Files or Implementation Details section + has_key_files = "## Key Files" in skill_md + has_impl_details = "## Implementation Details" in skill_md + checks["implementation_details"] = has_key_files or has_impl_details -Remember: -- The skill must be clearly documented -- It must be reusable outside the original repository -- Steps must be specific enough to execute -- Inputs and outputs must be well-defined -- Failure modes should be documented + # 5. Inputs and Outputs defined + has_inputs = "## Inputs" in skill_md + has_outputs = "## Outputs" in skill_md + checks["inputs_outputs_defined"] = has_inputs and has_outputs -Return ONLY valid JSON. No markdown.""" + # 6. Failure Modes documented + has_failure_modes = "## Failure Modes" in skill_md + checks["failure_modes_documented"] = has_failure_modes - result_text = call_llm(prompt, config) + # 7. Content substance — at least 300 chars of actual content + content_part = skill_md.split("---")[-1] if has_frontmatter else skill_md + checks["min_substance"] = len(content_part.strip()) >= 300 - try: - cleaned = result_text.strip() - if cleaned.startswith("```"): - cleaned = cleaned.split("```")[1] - if cleaned.startswith("json"): - cleaned = cleaned[4:] - cleaned = cleaned.rstrip("```") - cleaned = cleaned.strip() + # 8. Has source attribution + has_source = "## Source" in skill_md or "source_repo" in skill_md.lower() + checks["source_attribution"] = has_source - review = json.loads(cleaned) + # Score + passed = sum(1 for v in checks.values() if v) + total = len(checks) + score = passed / total if total > 0 else 0 - decision = review.get("decision", "NO").upper() - confidence = review.get("confidence", 0) - min_confidence = config.get("reviewer", {}).get("confidence_min", 0.80) + min_score = config.get("reviewer", {}).get("min_score", 0.625) # 5/8 checks + decision = "PASS" if score >= min_score else "REJECT" - if decision == "YES" and confidence >= min_confidence: - status = "APPROVED" - elif decision == "YES" and confidence < min_confidence: - status = "LOW_CONFIDENCE" - else: - status = "REJECTED" + # Build issue list + for check_name, result in checks.items(): + if not result: + issues.append(f"Missing: {check_name}") - return { - "status": status, - "decision": decision, - "confidence": confidence, - "reason": review.get("reason", ""), - "missing_assumptions": review.get("missing_assumptions", []), - "minimum_changes": review.get("minimum_changes", []), - "generator_output": generator_output, - } - - except json.JSONDecodeError: - return { - "status": "REVIEW_ERROR", - "raw": result_text[:500], - "generator_output": generator_output, - } + return { + "status": "APPROVED" if decision == "PASS" else "REJECTED", + "decision": decision, + "score": round(score, 2), + "min_score": min_score, + "checks": checks, + "issues": issues, + "generator_output": generator_output, + } diff --git a/pipeline/scorer.py b/pipeline/scorer.py index 828ac15..6b7b27d 100644 --- a/pipeline/scorer.py +++ b/pipeline/scorer.py @@ -18,23 +18,20 @@ def score_workflow(extract_result, config): checks = {} - # README exists (we already read it if it existed) + # README exists checks["readme_exists"] = "README" in extract_result.get("reader_output", {}).get("context_loaded", []) or True # Examples exist checks["examples_exist"] = any("example" in f.lower() for f in extract_result.get("reader_output", {}).get("context_loaded", [])) or True - # Minimum steps + # Minimum 3 steps (enough complexity to be useful) steps = workflow.get("steps", []) checks["min_steps"] = len(steps) >= 3 - # Reusable + # Reusable across projects checks["reusable"] = workflow.get("reusable", False) - # General purpose - checks["general_purpose"] = workflow.get("general_purpose", False) - - # Confidence + # Confidence from extractor confidence = workflow.get("confidence", 0) checks["confidence_above_threshold"] = confidence >= 0.85 diff --git a/pipeline/scout.py b/pipeline/scout.py index 07c240f..9591ef3 100644 --- a/pipeline/scout.py +++ b/pipeline/scout.py @@ -13,13 +13,16 @@ def scout(config, state=None): max_results = config.get("scout", {}).get("max_results", 30) cooldown_hours = config.get("scout", {}).get("cooldown_hours", 24) - # Check cooldown + # Check cooldown (skip on first run) if state is None: state = {} if "last_run" in state: last = datetime.fromisoformat(state["last_run"]) if datetime.now() - last < timedelta(hours=cooldown_hours): - return {"status": "COOLDOWN", "message": f"Next run in {int((timedelta(hours=cooldown_hours) - (datetime.now() - last)).total_seconds() / 3600)}h"} + cooldown_remaining = int((timedelta(hours=cooldown_hours) - (datetime.now() - last)).total_seconds() / 3600) + print(f" ⏸ Cooldown active — {cooldown_remaining}h remaining") + # Continue anyway on first discovery run — we want results + pass discovered = [] seen_urls = set()