Files
CIAgent/apps/api/tests/unit/test_report_backfill.py
sakshamandClaude Sonnet 5 18305b545c Backfill sparse report sections, add enrichment section refresh, and bootstrap first-admin
Reports: the LLM reliably used company_enrichment for prose fields but
inconsistently populated the parallel Finding-list/string-list fields from
the same evidence, even with progressively more explicit prompting. Add a
code-level backfill (products, recent developments, financial signals,
strategic initiatives, regulatory signals, risks/opportunities mirrored
from SWOT, unknowns, monitoring recommendations) that only ever fills in
what the model left empty, never overwrites what it produced.

Enrichment tab: reorder sections (Products/Recent updates before
Customers/Competitors) and add a per-section "Refresh" button that
re-fetches just one of NinjaPear's six independent per-company endpoints
when it came back empty - confirmed live that a data-coverage gap (e.g.
Amazon returning no products) is real provider behavior, not a bug.

Auth: the first account registered on a deployment with zero existing
admins is now auto-promoted to admin, closing the chicken-and-egg gap
where the only path to admin access was direct DB access. Self-heals if
the last admin ever deletes their account.

Also bumps nginx's proxy_read_timeout for api.ciagent.org to cover the
enrichment refresh's synchronous funding-endpoint call (up to 5 minutes
per NinjaPear's docs).

Co-Authored-By: Claude Sonnet 5 <[email protected]>
2026-08-06 17:41:02 -04:00

284 lines
9.8 KiB
Python

"""_backfill_from_enrichment / _mirror_swot_into_flat_lists: company_enrichment
data (products/recent_updates/funding) and the model's own SWOT output are
direct, unambiguous sources for products_and_services/recent_developments/
financial_signals/risks/opportunities - no additional LLM judgment required.
Confirmed live that a real Anthropic call can still leave these fields empty
despite explicit prompt instructions to populate them, so this backfill
guarantees the data surfaces regardless of what the model chose to do."""
from __future__ import annotations
from app.prompts.report_generation import ReportContent, SwotAnalysis
from app.prompts.schemas import Finding
from app.services.report_service import (
_backfill_from_enrichment,
_backfill_reflective_sections,
_mirror_swot_into_flat_lists,
)
def _empty_report() -> ReportContent:
return ReportContent(
executive_summary="",
company_overview="",
market_positioning="",
customer_sentiment="",
competitor_comparison="",
swot=SwotAnalysis(),
methodology="",
limitations="",
)
def test_backfills_products_from_enrichment_when_model_left_it_empty():
report = _empty_report()
enrichment = {
"products": [
{"name": "Acme Pay", "category": "Payments", "description": "Accept cards online."},
{"name": "Acme Billing", "category": "Billing", "description": "Recurring invoices."},
]
}
_backfill_from_enrichment(report, enrichment)
assert [f.headline for f in report.products_and_services] == ["Acme Pay", "Acme Billing"]
assert report.products_and_services[0].summary == "Accept cards online."
assert report.products_and_services[0].confidence == "confirmed"
assert report.products_and_services[0].evidence == []
def test_backfills_recent_developments_from_enrichment_updates():
report = _empty_report()
enrichment = {
"recent_updates": [
{
"url": "https://acme.example/blog/launch",
"date": "2026-01-01",
"text": "Acme launches new dashboard",
"type": "blog",
}
]
}
_backfill_from_enrichment(report, enrichment)
assert len(report.recent_developments) == 1
finding = report.recent_developments[0]
assert finding.headline == "Acme launches new dashboard"
assert finding.date == "2026-01-01"
assert finding.evidence[0].url == "https://acme.example/blog/launch"
def test_does_not_overwrite_findings_the_model_already_produced():
report = _empty_report()
report.products_and_services = [
Finding(headline="Model-provided product", summary="From the model itself")
]
enrichment = {"products": [{"name": "Should not appear", "description": "..."}]}
_backfill_from_enrichment(report, enrichment)
assert len(report.products_and_services) == 1
assert report.products_and_services[0].headline == "Model-provided product"
def test_no_enrichment_data_leaves_lists_empty():
report = _empty_report()
_backfill_from_enrichment(report, None)
assert report.products_and_services == []
assert report.recent_developments == []
def test_skips_entries_missing_the_required_key():
report = _empty_report()
enrichment = {
"products": [{"category": "No name here"}],
"recent_updates": [{"url": "https://acme.example", "date": "2026-01-01"}],
}
_backfill_from_enrichment(report, enrichment)
assert report.products_and_services == []
assert report.recent_developments == []
def test_backfills_financial_signals_from_funding_rounds_and_total():
report = _empty_report()
enrichment = {
"funding": {
"total_raised": "9810000000",
"rounds": [
{
"date": "2026-02-01",
"amount": None,
"investors": ["Thrive Capital", "Coatue Management"],
"round_name": "SECONDARY_SALE",
},
{
"date": "2023-03-01",
"amount": "6870000000",
"investors": ["Thrive Capital", "Andreessen Horowitz"],
"round_name": "SERIES_I",
},
],
}
}
_backfill_from_enrichment(report, enrichment)
assert len(report.financial_signals) == 3
assert report.financial_signals[0].headline == "Total funding raised: $9,810,000,000"
assert report.financial_signals[1].headline == "Secondary Sale"
assert "Thrive Capital" in report.financial_signals[1].summary
assert report.financial_signals[2].headline == "Series I - $6,870,000,000"
assert report.financial_signals[2].date == "2023-03-01"
def test_no_funding_data_leaves_financial_signals_empty():
report = _empty_report()
_backfill_from_enrichment(report, {"products": []})
assert report.financial_signals == []
def test_does_not_overwrite_financial_signals_the_model_already_produced():
report = _empty_report()
report.financial_signals = [Finding(headline="Model-provided signal", summary="From the model")]
enrichment = {"funding": {"total_raised": "1000", "rounds": []}}
_backfill_from_enrichment(report, enrichment)
assert len(report.financial_signals) == 1
assert report.financial_signals[0].headline == "Model-provided signal"
def test_mirrors_swot_threats_into_risks_when_risks_empty():
report = _empty_report()
report.swot = SwotAnalysis(threats=["Regulatory scrutiny", "New entrants"])
_mirror_swot_into_flat_lists(report)
assert report.risks == ["Regulatory scrutiny", "New entrants"]
def test_mirrors_swot_opportunities_into_opportunities_when_empty():
report = _empty_report()
report.swot = SwotAnalysis(opportunities=["Expand into new markets"])
_mirror_swot_into_flat_lists(report)
assert report.opportunities == ["Expand into new markets"]
def test_does_not_overwrite_risks_or_opportunities_the_model_already_produced():
report = _empty_report()
report.risks = ["Model-provided risk"]
report.opportunities = ["Model-provided opportunity"]
report.swot = SwotAnalysis(threats=["Should not appear"], opportunities=["Should not appear"])
_mirror_swot_into_flat_lists(report)
assert report.risks == ["Model-provided risk"]
assert report.opportunities == ["Model-provided opportunity"]
def test_empty_swot_leaves_risks_and_opportunities_empty():
report = _empty_report()
_mirror_swot_into_flat_lists(report)
assert report.risks == []
assert report.opportunities == []
def test_backfills_strategic_initiatives_from_specialties():
report = _empty_report()
enrichment = {"specialties": ["Payment Processing", "Billing Models"]}
_backfill_reflective_sections(
report, enrichment=enrichment, documents=[], changes=[], company_name="Acme"
)
assert [f.headline for f in report.strategic_initiatives] == [
"Focus area: Payment Processing",
"Focus area: Billing Models",
]
assert report.strategic_initiatives[0].confidence == "possible"
def test_regulatory_signals_get_an_insufficient_evidence_placeholder_when_empty():
report = _empty_report()
_backfill_reflective_sections(
report, enrichment=None, documents=[], changes=[], company_name="Acme"
)
assert len(report.regulatory_and_legal_signals) == 1
assert report.regulatory_and_legal_signals[0].confidence == "insufficient_evidence"
def test_unknowns_reflect_actual_gaps_in_the_evidence_set():
report = _empty_report()
enrichment = {"funding": {}, "leadership_team": [], "customers": []}
_backfill_reflective_sections(
report, enrichment=enrichment, documents=[], changes=[], company_name="Acme"
)
assert any("financial" in u.lower() for u in report.unknowns_and_missing_data)
assert any("leadership" in u.lower() for u in report.unknowns_and_missing_data)
assert any("customers" in u.lower() for u in report.unknowns_and_missing_data)
assert any("source documents" in u.lower() for u in report.unknowns_and_missing_data)
assert any("changes" in u.lower() for u in report.unknowns_and_missing_data)
def test_unknowns_omit_gaps_that_are_actually_covered():
report = _empty_report()
enrichment = {
"funding": {"total_raised": "1000"},
"leadership_team": [{"name": "Jane"}],
"customers": [{"name": "Acme Corp"}],
}
_backfill_reflective_sections(
report,
enrichment=enrichment,
documents=[{"id": "d1"}],
changes=[{"id": "c1"}],
company_name="Acme",
)
assert report.unknowns_and_missing_data == []
def test_monitoring_recommendations_populated_when_model_left_empty():
report = _empty_report()
_backfill_reflective_sections(
report, enrichment=None, documents=[], changes=[], company_name="Acme"
)
assert len(report.monitoring_recommendations) >= 2
assert any("Acme" in r for r in report.monitoring_recommendations)
def test_reflective_sections_do_not_overwrite_model_output():
report = _empty_report()
report.strategic_initiatives = [Finding(headline="Model theme", summary="From the model")]
report.regulatory_and_legal_signals = [Finding(headline="Model signal", summary="From model")]
report.unknowns_and_missing_data = ["Model-noted gap"]
report.monitoring_recommendations = ["Model recommendation"]
enrichment = {"specialties": ["Should not appear"]}
_backfill_reflective_sections(
report, enrichment=enrichment, documents=[], changes=[], company_name="Acme"
)
assert report.strategic_initiatives[0].headline == "Model theme"
assert report.regulatory_and_legal_signals[0].headline == "Model signal"
assert report.unknowns_and_missing_data == ["Model-noted gap"]
assert report.monitoring_recommendations == ["Model recommendation"]