Files

170 lines
7.1 KiB
Python
Raw Permalink Normal View History

"""DeepSeek AI extractor for structured tech profiling from scraped text."""
from __future__ import annotations
import json
import logging
import os
import re
from datetime import datetime, timezone
from typing import Optional
from openai import AsyncOpenAI
from models import CompanyTechProfile
logger = logging.getLogger(__name__)
DEEPSEEK_BASE_URL = "https://openrouter.ai/api/v1"
DEEPSEEK_MODEL = "qwen/qwen3-coder:free"
FALLBACK_MODELS = [
"meta-llama/llama-3.3-70b-instruct:free",
"google/gemma-4-26b-a4b-it:free",
"nvidia/nemotron-3-super-120b-a12b:free",
]
MAX_INPUT_CHARS = 50_000 # Truncate input to stay within context limits
EXTRACTION_SYSTEM_PROMPT = """You are a Principal AI Architect and technical due-diligence expert.
Analyze the company website text and extract structured technical intelligence.
CRITICAL RULES:
- detected_technologies MUST contain at least 3 items. Look for ANY technology mention: programming languages, databases, cloud providers, DevOps tools, frameworks, AI/ML tools, data platforms, monitoring, CI/CD, etc. If the page mentions nothing technical, list technologies commonly associated with their industry.
- current_pain_points MUST contain at least 2 items. Infer from their industry, product description, or company stage.
- hiring_roles and desired_skills: If there is a careers/jobs section in the text, list actual job titles and required skills. Look for words like "hiring", "we are looking for", "join our team", "job", "position", "engineer", "developer", "scientist", "architect".
- outreach_hook MUST be at least 30 characters. Never leave it empty.
Return ONLY valid JSON matching this exact schema:
{
"company_name": "string",
"detected_technologies": ["at least 3 items, be creative if needed based on industry"],
"current_pain_points": ["at least 2 items"],
"estimated_data_maturity": "Low | Medium | High",
"outreach_hook": "string (minimum 1 sentence, be specific)",
"industry": "string or null",
"company_size_hint": "Startup | Scale-up | Mid-market | Enterprise or null",
"tech_blog_posts": "url or null",
"hiring_roles": ["list of job titles from careers section, empty if no jobs section found"],
"desired_skills": ["list of skills from job requirements, empty if no jobs section found"],
"job_listings_url": "url to careers page or null"
}
No markdown fences, no trailing commas. Every array MUST have content unless truly no information exists."""
class DeepSeekExtractor:
"""Extracts structured CompanyTechProfile from text using DeepSeek API."""
def __init__(self, api_key: Optional[str] = None, base_url: Optional[str] = None):
self.api_key = api_key or os.getenv("DEEPSEEK_API_KEY", "")
self.base_url = base_url or os.getenv("DEEPSEEK_BASE_URL", DEEPSEEK_BASE_URL)
self._client: Optional[AsyncOpenAI] = None
@property
def client(self) -> AsyncOpenAI:
if self._client is None:
if not self.api_key:
raise ValueError(
"DeepSeek API key not configured. "
"Set DEEPSEEK_API_KEY environment variable or pass api_key parameter."
)
self._client = AsyncOpenAI(
api_key=self.api_key,
base_url=self.base_url,
max_retries=2,
timeout=60.0,
default_headers={
"HTTP-Referer": "http://mek-tech.nl",
"X-Title": "Mek-Tech LeadGen Engine",
},
)
return self._client
async def extract(self, text: str, url: str, model: str = DEEPSEEK_MODEL, fallback_models: list[str] | None = None) -> CompanyTechProfile:
"""Extract CompanyTechProfile from scraped page text."""
if not text.strip():
raise ValueError("Empty text content cannot extract profile")
# Truncate text to fit model context
truncated = text[:MAX_INPUT_CHARS]
if len(text) > MAX_INPUT_CHARS:
logger.info("Text truncated from %d to %d chars", len(text), MAX_INPUT_CHARS)
user_prompt = f"URL: {url}\n\nANALYZE THIS COMPANY WEBSITE TEXT:\n\n{truncated}"
models_to_try = [model] + (fallback_models or FALLBACK_MODELS)
last_error = None
for m in models_to_try:
try:
response = await self.client.chat.completions.create(
model=m,
messages=[
{"role": "system", "content": EXTRACTION_SYSTEM_PROMPT},
{"role": "user", "content": user_prompt},
],
temperature=0.1,
max_tokens=2048,
)
logger.info("Used model: %s", m)
break
except Exception as exc:
last_error = exc
logger.warning("Model %s failed: %s", m, str(exc)[:200])
if m != models_to_try[-1]:
import asyncio
await asyncio.sleep(1)
continue
else:
raise RuntimeError(f"All models failed. Last error: {str(last_error)}") from last_error
raw_content = response.choices[0].message.content or "{}"
parsed = self._parse_json(raw_content)
# Enrich with metadata
parsed["extracted_at"] = datetime.now(timezone.utc).isoformat()
parsed["source_url"] = url
# Normalize maturity
maturity = str(parsed.get("estimated_data_maturity", "")).strip().capitalize()
if maturity not in ("Low", "Medium", "High"):
maturity = "Medium"
parsed["estimated_data_maturity"] = maturity
# Ensure lists are lists, strings are strings
for field in ("detected_technologies", "current_pain_points", "hiring_roles", "desired_skills"):
if not isinstance(parsed.get(field), list):
parsed[field] = []
for field in ("outreach_hook", "industry", "company_size_hint", "tech_blog_posts", "job_listings_url"):
if parsed.get(field) is None:
parsed[field] = ""
if not parsed.get("company_name"):
parsed["company_name"] = url.split("//")[-1].split("/")[0]
try:
return CompanyTechProfile(**parsed)
except Exception as exc:
logger.error("Pydantic validation failed: %s\nRaw data: %s", exc, parsed)
raise ValueError(f"Failed to validate extracted profile: {str(exc)}") from exc
@staticmethod
def _parse_json(text: str) -> dict:
"""Robust JSON parsing from LLM output."""
# Strip markdown code fences if present
text = text.strip()
if text.startswith("```"):
text = re.sub(r"^```(?:json)?\s*", "", text)
text = re.sub(r"\s*```$", "", text)
try:
return json.loads(text)
except json.JSONDecodeError:
pass
# Attempt to extract JSON object from text
match = re.search(r"\{.*\}", text, re.DOTALL)
if match:
try:
return json.loads(match.group(0))
except json.JSONDecodeError:
pass
raise ValueError(f"Unable to parse valid JSON from LLM response: {text[:500]}")