Files
CIAgent/apps/api/tests/integration/test_discovery_service.py
saksham 1a4c80958f Initial commit: CI Agent competitive-intelligence monitoring app
FastAPI + Celery + Next.js + Postgres/Redis app with company monitoring,
source collection, LLM-based change analysis, enrichment, and account
security (Turnstile, escalating lockout, email verification).
2026-08-05 10:48:20 -04:00

190 lines
6.6 KiB
Python

"""Company discovery, end-to-end against the mock providers: name in,
DiscoveredCompanyProfile out, real evidence flow through search -> fetch ->
LLM extraction, with user hints always winning over discovered values. No
live network - respx mocks every HTTP call, DNS resolution is patched to a
fixed public IP (matching the pattern in test_collection_service.py)."""
from __future__ import annotations
import json
from unittest.mock import patch
import httpx
import pytest
import respx
from app.analysis.llm.mock import MockLLMProvider
from app.models.enums import SourceType
from app.search.base import SearchResult
from app.search.mock import MockSearchProvider
from app.services.discovery_service import _resolve_official_website, discover_company_profile
@pytest.fixture(autouse=True)
def _no_real_dns():
with patch("socket.getaddrinfo", return_value=[(2, 1, 6, "", ("93.184.216.34", 0))]):
yield
def _mock_empty_github_sec():
respx.get("https://api.github.com/search/users").mock(
return_value=httpx.Response(200, text=json.dumps({"items": []}))
)
respx.get("https://www.sec.gov/cgi-bin/browse-edgar").mock(
return_value=httpx.Response(
200, text='<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom"></feed>'
)
)
@pytest.mark.asyncio
async def test_discover_resolves_website_and_extracts_profile_from_real_evidence(settings):
search = MockSearchProvider()
llm = MockLLMProvider()
with respx.mock:
_mock_empty_github_sec()
respx.get("https://acmemobility.com/robots.txt").mock(return_value=httpx.Response(404))
respx.get("https://acmemobility.com").mock(
return_value=httpx.Response(
200,
html=(
"<html><head><title>Acme Mobility</title></head><body>"
"<article><h1>About</h1><p>Acme Mobility is headquartered in Austin, Texas. "
"Formerly known as Acme Scooters.</p></article></body></html>"
),
)
)
profile = await discover_company_profile(
search,
llm,
settings,
name="Acme Mobility",
official_website=None,
monitoring_focus="pricing changes",
competitor_names=[],
alias_names=[],
)
assert profile.official_website == "https://acmemobility.com"
assert profile.headquarters == "Austin, Texas"
assert profile.aliases == ["Acme Scooters"]
assert profile.monitoring_focus == "pricing changes"
assert "https://acmemobility.com" in profile.sources_consulted
assert any(p.source_type == SourceType.WEBSITE for p in profile.potential_sources)
@pytest.mark.asyncio
async def test_discover_prefers_user_hints_over_discovered_values(settings):
search = MockSearchProvider()
llm = MockLLMProvider()
with respx.mock:
_mock_empty_github_sec()
respx.get("https://acme.example/robots.txt").mock(return_value=httpx.Response(404))
respx.get("https://acme.example").mock(
return_value=httpx.Response(
200, html="<html><body><p>Acme, based in Denver, Colorado.</p></body></html>"
)
)
profile = await discover_company_profile(
search,
llm,
settings,
name="Acme",
official_website="https://acme.example",
monitoring_focus=None,
competitor_names=["Rival Corp"],
alias_names=["Acme Inc"],
)
# Hint website used as-is (no "official website" search performed for it)
assert profile.official_website == "https://acme.example"
assert profile.competitors == ["Rival Corp"]
assert profile.aliases == ["Acme Inc"]
# Still extracted from the real fetched page since that hint wasn't given
assert profile.headquarters == "Denver, Colorado"
@pytest.mark.asyncio
async def test_discover_handles_a_website_that_fails_to_resolve_gracefully(settings):
search = MockSearchProvider()
llm = MockLLMProvider()
with respx.mock:
_mock_empty_github_sec()
respx.get("https://nowhereco.com/robots.txt").mock(return_value=httpx.Response(404))
respx.get("https://nowhereco.com").mock(return_value=httpx.Response(500))
profile = await discover_company_profile(
search,
llm,
settings,
name="Nowhere Co",
official_website=None,
monitoring_focus=None,
competitor_names=[],
alias_names=[],
)
assert profile.official_website == "https://nowhereco.com"
assert profile.headquarters is None
assert profile.aliases == []
class _StubSearchProvider:
"""Returns a fixed result list regardless of query - lets a test control
exactly what "official website" search ranking looks like, independent
of MockSearchProvider's domain-guessing heuristic."""
provider_name = "stub"
def __init__(self, results: list[SearchResult]) -> None:
self._results = results
async def search(self, query: str, *, count: int = 5) -> list[SearchResult]:
return self._results[:count]
@pytest.mark.asyncio
async def test_resolve_official_website_skips_a_top_ranked_wikipedia_result():
# Observed live against the real Brave API: "Stripe official website"
# ranked Stripe's Wikipedia article above stripe.com itself.
search = _StubSearchProvider(
[
SearchResult(
title="Stripe, Inc. - Wikipedia",
url="https://en.wikipedia.org/wiki/Stripe,_Inc.",
snippet="Stripe, Inc. is an American financial services company.",
),
SearchResult(
title="Stripe | Financial Infrastructure",
url="https://stripe.com",
snippet="Stripe powers online and in-person payment processing.",
),
]
)
url, consulted = await _resolve_official_website(search, "Stripe", None)
assert url == "https://stripe.com"
assert consulted == ["https://stripe.com"]
@pytest.mark.asyncio
async def test_resolve_official_website_falls_back_to_top_result_when_all_are_reference_sites():
search = _StubSearchProvider(
[
SearchResult(
title="Acme - Wikipedia",
url="https://en.wikipedia.org/wiki/Acme",
snippet="An encyclopedia article.",
),
]
)
url, consulted = await _resolve_official_website(search, "Acme", None)
assert url == "https://en.wikipedia.org/wiki/Acme"
assert consulted == ["https://en.wikipedia.org/wiki/Acme"]