Backfill sparse report sections, add enrichment section refresh, and bootstrap first-admin

Reports: the LLM reliably used company_enrichment for prose fields but
inconsistently populated the parallel Finding-list/string-list fields from
the same evidence, even with progressively more explicit prompting. Add a
code-level backfill (products, recent developments, financial signals,
strategic initiatives, regulatory signals, risks/opportunities mirrored
from SWOT, unknowns, monitoring recommendations) that only ever fills in
what the model left empty, never overwrites what it produced.

Enrichment tab: reorder sections (Products/Recent updates before
Customers/Competitors) and add a per-section "Refresh" button that
re-fetches just one of NinjaPear's six independent per-company endpoints
when it came back empty - confirmed live that a data-coverage gap (e.g.
Amazon returning no products) is real provider behavior, not a bug.

Auth: the first account registered on a deployment with zero existing
admins is now auto-promoted to admin, closing the chicken-and-egg gap
where the only path to admin access was direct DB access. Self-heals if
the last admin ever deletes their account.

Also bumps nginx's proxy_read_timeout for api.ciagent.org to cover the
enrichment refresh's synchronous funding-endpoint call (up to 5 minutes
per NinjaPear's docs).

Co-Authored-By: Claude Sonnet 5 <[email protected]>
This commit is contained in:
2026-08-06 17:41:02 -04:00
co-authored by Claude Sonnet 5
parent 1be3e53584
commit 18305b545c
18 changed files with 1087 additions and 53 deletions
+14
View File
@@ -125,11 +125,25 @@ async def register(
if existing is not None:
raise ConflictError("An account with this email already exists")
# Bootstraps admin access on a fresh deployment - otherwise the only way
# to ever get an admin account is direct DB access, which is a real
# chicken-and-egg problem for anyone self-hosting from a clean clone.
# Checked by admin *count*, not total user count, so this also
# self-heals if the last admin ever deletes their own account (Settings
# -> Delete account) - the next registration becomes admin again rather
# than leaving the deployment permanently admin-less. A benign race is
# possible if two people register in the same instant on a brand-new,
# zero-admin deployment (both could become admin) - acceptable for a
# bootstrapping check that only ever matters once, before any real
# traffic exists.
is_first_admin = await repo.count_admins() == 0
user = await repo.create(
email=payload.email,
password_hash=hash_password(payload.password),
display_name=payload.display_name,
timezone=payload.timezone,
is_admin=is_first_admin,
# Test suite has no inbox to read a real code from - same
# app_env == "test" precedent already used to disable rate limiting
# (app/core/rate_limit.py). The code-generation/sending/throttle
+102
View File
@@ -20,6 +20,7 @@ from datetime import UTC, datetime
from sqlalchemy.ext.asyncio import AsyncSession
from app.core.config import Settings
from app.core.errors import NotFoundError
from app.core.logging import get_logger
from app.enrichment.base import EnrichmentProvider
from app.models.company import Company
@@ -174,3 +175,104 @@ async def enrich_company(
failed_sections=list(errors.keys()),
)
return enrichment
# Every entry here maps 1:1 onto one of NinjaPear's independent
# per-company endpoints (see app/enrichment/ninjapear.py) - refreshing one
# section never re-fetches or touches any of the others.
REFRESHABLE_SECTIONS = ("details", "funding", "updates", "competitors", "products", "customers")
async def refresh_section(
db: AsyncSession,
settings: Settings,
provider: EnrichmentProvider,
company: Company,
section: str,
) -> CompanyEnrichment:
"""Re-runs a single NinjaPear endpoint for a company that already has
an enrichment record but came back empty or failed for just this one
section (e.g. `products` returning `[]` while everything else
succeeded - a real, honest gap in NinjaPear's own coverage for that
company, not a bug in this app). Only ever replaces this section's own
slice of `data`/`errors` - every other section's stored data is left
exactly as it was."""
if section not in REFRESHABLE_SECTIONS:
raise ValueError(f"Unknown enrichment section: {section}")
repo = CompanyEnrichmentRepository(db)
existing = await repo.get_for_company(company.id)
if existing is None:
raise NotFoundError("Company has no enrichment record to refresh")
website = company.official_website
data = dict(existing.data)
errors = dict(existing.errors)
credits_spent = existing.credits_spent or 0
try:
if section == "details":
details = await provider.get_company_details(company.name, website)
data["employee_count"] = details.employee_count_range
data["description"] = details.description
data["industry"] = details.industry
data["founded_year"] = details.founded_year
data["specialties"] = details.specialties
new_leadership = [m.model_dump() for m in details.leadership_team]
# A details refresh re-fetches the leadership roster itself,
# but not the separate per-leader work_email/person_profile
# lookups - preserve those by matching on name so a refresh
# never regresses contact info that was already found.
old_by_name = {
m.get("name"): m for m in (data.get("leadership_team") or []) if m.get("name")
}
for member in new_leadership:
old = old_by_name.get(member.get("name"))
if old:
member["work_email"] = member.get("work_email") or old.get("work_email")
member["profile_url"] = member.get("profile_url") or old.get("profile_url")
member["bio"] = member.get("bio") or old.get("bio")
data["leadership_team"] = new_leadership
elif section == "funding":
funding = await provider.get_funding(company.name, website)
data["funding"] = funding.model_dump()
elif section == "updates":
updates = await provider.get_updates(company.name, website)
data["recent_updates"] = [u.model_dump() for u in updates]
elif section == "competitors":
competitors = await provider.get_competitors(company.name, website)
data["competitors"] = [c.model_dump() for c in competitors]
elif section == "products":
products = await provider.get_products(company.name, website)
data["products"] = [p.model_dump() for p in products]
elif section == "customers":
customers = await provider.get_customers(company.name, website)
data["customers"] = [c.model_dump() for c in customers]
errors.pop(section, None)
credits_spent += _CREDIT_COSTS.get(section, 0)
except Exception as exc: # noqa: BLE001 - surfaced via `errors`, same as the initial run
errors[section] = str(exc)
logger.warning(
"enrichment_section_refresh_failed",
section=section,
company_id=str(company.id),
error=str(exc),
)
new_status = EnrichmentStatus.PARTIAL if errors else EnrichmentStatus.COMPLETE
enrichment = await repo.upsert(
company.id,
status=new_status,
data=data,
errors=errors,
credits_spent=credits_spent,
fetched_at=datetime.now(UTC),
)
await db.commit()
logger.info(
"company_enrichment_section_refreshed",
company_id=str(company.id),
section=section,
succeeded=section not in errors,
)
return enrichment
+215 -1
View File
@@ -23,7 +23,8 @@ from app.models.enums import EnrichmentStatus, ReportType, SourceStatus
from app.models.report import Report
from app.models.source import Source
from app.models.source_document import SourceDocument
from app.prompts.report_generation import generate_report
from app.prompts.report_generation import ReportContent, generate_report
from app.prompts.schemas import ConfidenceLabel, EvidenceRef, Finding
from app.repositories.company_enrichment_repository import CompanyEnrichmentRepository
from app.repositories.report_repository import ReportRepository
from app.services import company_service
@@ -31,6 +32,210 @@ from app.services.report_markdown import render_report_markdown
_MAX_DOCUMENTS = 40
_MAX_CHANGES = 20
_MAX_BACKFILLED_PRODUCTS = 30
_MAX_BACKFILLED_UPDATES = 15
_MAX_BACKFILLED_FUNDING_ROUNDS = 15
def _format_amount(amount: object) -> str | None:
if amount is None:
return None
text = str(amount)
return f"${int(text):,}" if text.isdigit() else text
def _backfill_from_enrichment(content: ReportContent, enrichment: dict | None) -> None:
"""company_enrichment.products/recent_updates/funding map onto
products_and_services/recent_developments/financial_signals with no
inference required - a direct transcription, not a judgment call.
Confirmed live that the LLM is nonetheless inconsistent about
populating these Finding-list fields from enrichment alone (it
reliably uses the same data for prose fields like company_overview,
but repeated real API calls with increasingly explicit instructions
still came back with these lists empty). Rather than keep fighting
prompt compliance for a purely mechanical transform, backfill directly
from the source data whenever the model leaves a field empty despite
the data being available - this only ever *adds* real, non-fabricated
content the model chose not to surface, never overwrites what the
model did produce."""
if not enrichment:
return
if not content.products_and_services:
content.products_and_services = [
Finding(
headline=product["name"],
summary=product.get("description") or product.get("category") or product["name"],
category=product.get("category"),
confidence=ConfidenceLabel.CONFIRMED,
)
for product in enrichment.get("products", [])[:_MAX_BACKFILLED_PRODUCTS]
if product.get("name")
]
if not content.recent_developments:
content.recent_developments = [
Finding(
headline=update["text"],
summary=f"{update.get('type', 'update').capitalize()} published by the company.",
date=update.get("date"),
confidence=ConfidenceLabel.CONFIRMED,
evidence=(
[EvidenceRef(url=update["url"], description="Company-published update.")]
if update.get("url")
else []
),
)
for update in enrichment.get("recent_updates", [])[:_MAX_BACKFILLED_UPDATES]
if update.get("text")
]
if not content.financial_signals:
funding = enrichment.get("funding") or {}
findings = []
total_raised = _format_amount(funding.get("total_raised"))
if total_raised:
findings.append(
Finding(
headline=f"Total funding raised: {total_raised}",
summary="Cumulative funding raised across all rounds, per company_enrichment.",
confidence=ConfidenceLabel.CONFIRMED,
)
)
for round_ in funding.get("rounds", [])[:_MAX_BACKFILLED_FUNDING_ROUNDS]:
name = round_.get("round_name")
if not name:
continue
amount = _format_amount(round_.get("amount"))
investors = round_.get("investors") or []
headline = name.replace("_", " ").title() + (f" - {amount}" if amount else "")
findings.append(
Finding(
headline=headline,
summary=(
f"Investors: {', '.join(investors)}."
if investors
else "No investor detail provided."
),
date=round_.get("date"),
confidence=ConfidenceLabel.CONFIRMED,
)
)
content.financial_signals = findings
def _mirror_swot_into_flat_lists(content: ReportContent) -> None:
"""risks/opportunities are meant to be the same analysis as
swot.threats/swot.opportunities, just in a flat top-level list rather
than nested under swot - not a second, independent judgment call.
Confirmed live: the model reliably populates the SWOT fields but is
inconsistent about also populating these parallel flat fields with the
same content, even though nothing about them requires different
evidence. Mirror rather than re-derive, since the model already did
the real analytical work once."""
if not content.risks and content.swot.threats:
content.risks = list(content.swot.threats)
if not content.opportunities and content.swot.opportunities:
content.opportunities = list(content.swot.opportunities)
def _backfill_reflective_sections(
content: ReportContent,
*,
enrichment: dict | None,
documents: list[dict],
changes: list[dict],
company_name: str,
) -> None:
"""strategic_initiatives, regulatory_and_legal_signals,
unknowns_and_missing_data, and monitoring_recommendations are the
fields the LLM was most persistently reluctant to populate even after
two rounds of explicit prompt strengthening (confirmed live - see
SYSTEM_PROMPT in report_generation.py). Unlike products/recent_updates,
these don't have a single obvious mechanical source, but each still
has SOMETHING honest and non-fabricated to fall back on:
- strategic_initiatives: company_enrichment.specialties names the
company's own stated focus areas - a real, low-confidence signal,
not invented.
- regulatory_and_legal_signals: when truly nothing applies, a single
insufficient_evidence Finding is what the prompt already asks the
model to emit in this situation instead of leaving the list empty -
the model just isn't doing it reliably, so this fills the same gap.
- unknowns_and_missing_data / monitoring_recommendations: these are
meta-analysis of the evidence set itself, not claims about the
company, so they can be derived honestly from what evidence this
pipeline actually did or didn't collect for this company."""
enrichment = enrichment or {}
if not content.strategic_initiatives:
specialties = enrichment.get("specialties") or []
content.strategic_initiatives = [
Finding(
headline=f"Focus area: {specialty}",
summary=(
f"{company_name} lists \"{specialty}\" among its specialties, "
"indicating an area of strategic focus."
),
confidence=ConfidenceLabel.POSSIBLE,
)
for specialty in specialties
if specialty
]
if not content.regulatory_and_legal_signals:
content.regulatory_and_legal_signals = [
Finding(
headline="No regulatory or legal signals identified",
summary=(
"No licensing, compliance, jurisdictional, or legal-structure "
"information was present in the available evidence for "
f"{company_name}."
),
confidence=ConfidenceLabel.INSUFFICIENT_EVIDENCE,
)
]
if not content.unknowns_and_missing_data:
unknowns = []
if not enrichment:
unknowns.append(
"No third-party company enrichment data was available for this company."
)
else:
if not enrichment.get("funding", {}).get("total_raised") and not enrichment.get(
"funding", {}
).get("rounds"):
unknowns.append(
"No detailed financial statements, revenue, or funding figures were found."
)
if not enrichment.get("leadership_team"):
unknowns.append("No leadership or executive team data was found.")
if not enrichment.get("customers"):
unknowns.append("No named customers or case studies were found.")
if not documents:
unknowns.append(
"No source documents have been collected yet from ongoing monitoring, so "
"hiring, technology, patent, and manufacturing signals are not yet available."
)
if not changes:
unknowns.append(
"No changes have been detected yet between monitoring runs for this company."
)
content.unknowns_and_missing_data = unknowns
if not content.monitoring_recommendations:
recommendations = [
f"Monitor company_enrichment for {company_name} on its next refresh for changes "
"to products, leadership, or funding.",
"Track newly published company updates for announcements of new initiatives "
"or partnerships.",
]
if documents:
recommendations.append(
"Continue reviewing newly collected source documents for signals not yet "
"reflected in company_enrichment."
)
content.monitoring_recommendations = recommendations
async def _gather_documents(db: AsyncSession, company_id: uuid.UUID) -> list[dict]:
@@ -130,6 +335,15 @@ async def generate_and_persist_report(
public_identifiers=company.public_identifiers,
enrichment=enrichment,
)
_backfill_from_enrichment(content, enrichment)
_mirror_swot_into_flat_lists(content)
_backfill_reflective_sections(
content,
enrichment=enrichment,
documents=documents,
changes=changes,
company_name=company.name,
)
generated_at = datetime.now(UTC)
model_name = _model_name(settings, llm)