FastAPI + Celery + Next.js + Postgres/Redis app with company monitoring, source collection, LLM-based change analysis, enrichment, and account security (Turnstile, escalating lockout, email verification).
191 lines
7.2 KiB
Python
191 lines
7.2 KiB
Python
"""Company-metadata discovery: given just a name (plus optional user
|
|
hints), proposes official_website/industry/country/region/headquarters/
|
|
aliases/competitors/public_identifiers and a preview of sources the
|
|
pipeline would start monitoring - all before anything is persisted. This
|
|
is the "System performs company discovery" step between "user enters a
|
|
name" and "user confirms/edits" in the onboarding flow.
|
|
|
|
Two independent, deliberately-separated concerns per the architecture
|
|
direction:
|
|
- SearchProvider answers "where should we look" (this module's job).
|
|
- The LLM only ever analyzes evidence this module already gathered - see
|
|
app/prompts/company_profile.py's docstring. It is never asked to recall
|
|
facts about the company from its own training data.
|
|
|
|
Runs exactly once, at onboarding time, driven by an explicit user action
|
|
(the wizard's "Discover" step) - never re-triggered by scheduled monitoring
|
|
runs. Source *persistence* still happens exactly as it already did before
|
|
this module existed: lazily, on the company's first monitoring run (see
|
|
tasks/collection.py). This module only ever previews what that step would
|
|
find, via the same collector.discover() calls, without writing anything.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from app.analysis.llm.base import LLMProvider
|
|
from app.collectors.base import CompanyContext
|
|
from app.collectors.extraction import extract_readable_text
|
|
from app.collectors.registry import get_collector
|
|
from app.collectors.robots import is_allowed
|
|
from app.core.config import Settings
|
|
from app.core.http import FetchError, SsrfBlockedError, fetch_with_retries
|
|
from app.core.logging import get_logger
|
|
from app.models.enums import SourceType
|
|
from app.prompts.company_profile import extract_company_profile
|
|
from app.schemas.discovery import DiscoveredCompanyProfile, PotentialSource
|
|
from app.search.base import SearchProvider
|
|
|
|
logger = get_logger(__name__)
|
|
|
|
# A "{name} official website" search frequently ranks a reference/social
|
|
# page above the company's own domain for well-known companies (observed
|
|
# live: Brave's top result for "Stripe official website" was Stripe's
|
|
# Wikipedia article, not stripe.com). Picking that as `official_website`
|
|
# then feeds a wrong base domain into every downstream source-preview
|
|
# collector. Skip these hosts when a better-ranked alternative exists in
|
|
# the same result set, rather than blindly taking the top hit.
|
|
_NON_CORPORATE_HOSTS = (
|
|
"wikipedia.org",
|
|
"linkedin.com",
|
|
"crunchbase.com",
|
|
"bloomberg.com",
|
|
"facebook.com",
|
|
"twitter.com",
|
|
"x.com",
|
|
"youtube.com",
|
|
"reddit.com",
|
|
"glassdoor.com",
|
|
)
|
|
|
|
|
|
def _is_non_corporate_host(url: str) -> bool:
|
|
host = url.split("//", 1)[-1].split("/", 1)[0].lower()
|
|
return any(host == d or host.endswith(f".{d}") for d in _NON_CORPORATE_HOSTS)
|
|
|
|
|
|
# Same set collection_service.py's discover_sources_for_company already
|
|
# discovers from for a real company - kept in sync deliberately, not
|
|
# imported, since this module previews without a persisted Company/Source
|
|
# and the coupling would only make both harder to read.
|
|
_PREVIEWABLE_TYPES = (
|
|
SourceType.WEBSITE,
|
|
SourceType.GITHUB,
|
|
SourceType.SEC_EDGAR,
|
|
SourceType.JOB_POSTING,
|
|
SourceType.RSS,
|
|
SourceType.GOV_CONTRACT,
|
|
SourceType.PATENT,
|
|
)
|
|
|
|
|
|
async def _resolve_official_website(
|
|
search: SearchProvider, name: str, hint: str | None
|
|
) -> tuple[str | None, list[str]]:
|
|
if hint:
|
|
return hint, []
|
|
results = await search.search(f"{name} official website", count=3)
|
|
if not results:
|
|
return None, []
|
|
best = next((r for r in results if not _is_non_corporate_host(r.url)), results[0])
|
|
return best.url, [best.url]
|
|
|
|
|
|
async def _fetch_homepage_text(
|
|
settings: Settings, official_website: str | None
|
|
) -> tuple[str | None, list[str]]:
|
|
if not official_website:
|
|
return None, []
|
|
try:
|
|
if not await is_allowed(official_website, settings=settings):
|
|
return None, []
|
|
result = await fetch_with_retries(official_website, settings=settings, max_attempts=1)
|
|
except (FetchError, SsrfBlockedError) as exc:
|
|
logger.info("discovery_homepage_fetch_failed", url=official_website, error=str(exc))
|
|
return None, []
|
|
|
|
if result.status_code >= 400:
|
|
return None, []
|
|
|
|
text, _method = extract_readable_text(result.text, official_website)
|
|
return (text or None), [official_website]
|
|
|
|
|
|
async def _preview_sources(
|
|
context: CompanyContext,
|
|
) -> list[PotentialSource]:
|
|
previews: list[PotentialSource] = []
|
|
for source_type in _PREVIEWABLE_TYPES:
|
|
collector = get_collector(source_type)
|
|
try:
|
|
discovered = await collector.discover(context)
|
|
except Exception as exc: # pragma: no cover - defensive, preview is best-effort
|
|
logger.warning(
|
|
"discovery_source_preview_failed", source_type=source_type, error=str(exc)
|
|
)
|
|
continue
|
|
previews.extend(
|
|
PotentialSource(source_type=d.source_type, name=d.name, base_url=d.base_url)
|
|
for d in discovered
|
|
)
|
|
return previews
|
|
|
|
|
|
async def discover_company_profile(
|
|
search: SearchProvider,
|
|
llm: LLMProvider,
|
|
settings: Settings,
|
|
*,
|
|
name: str,
|
|
official_website: str | None,
|
|
monitoring_focus: str | None,
|
|
competitor_names: list[str],
|
|
alias_names: list[str],
|
|
) -> DiscoveredCompanyProfile:
|
|
resolved_website, consulted_website = await _resolve_official_website(
|
|
search, name, official_website
|
|
)
|
|
homepage_text, consulted_homepage = await _fetch_homepage_text(settings, resolved_website)
|
|
|
|
search_results = []
|
|
consulted_queries: list[str] = []
|
|
for query in (f"{name} headquarters", f"{name} competitors", f"{name} formerly known as"):
|
|
results = await search.search(query, count=3)
|
|
search_results.extend({"query": query, **r.model_dump()} for r in results)
|
|
consulted_queries.append(query)
|
|
|
|
extraction = await extract_company_profile(
|
|
llm,
|
|
company_name=name,
|
|
homepage_url=resolved_website,
|
|
homepage_text=homepage_text,
|
|
search_results=search_results,
|
|
)
|
|
|
|
context = CompanyContext(
|
|
id="pending",
|
|
name=name,
|
|
official_website=resolved_website,
|
|
monitoring_focus=monitoring_focus,
|
|
uspto_api_key=settings.uspto_api_key,
|
|
)
|
|
# Individual collectors already handle a missing official_website
|
|
# gracefully (e.g. JobPostingCollector.discover returns [] rather than
|
|
# raising) - GitHub/SEC EDGAR search by name and don't need one at all.
|
|
potential_sources = await _preview_sources(context)
|
|
|
|
return DiscoveredCompanyProfile(
|
|
name=name,
|
|
official_website=resolved_website,
|
|
description=extraction.description,
|
|
monitoring_focus=monitoring_focus,
|
|
industry=extraction.industry,
|
|
country=extraction.country,
|
|
region=extraction.region,
|
|
headquarters=extraction.headquarters,
|
|
aliases=alias_names or extraction.aliases,
|
|
competitors=competitor_names or extraction.competitors,
|
|
public_identifiers={pi.key: pi.value for pi in extraction.public_identifiers},
|
|
potential_sources=potential_sources,
|
|
sources_consulted=[*consulted_website, *consulted_homepage, *consulted_queries],
|
|
)
|