Backfill sparse report sections, add enrichment section refresh, and bootstrap first-admin
Reports: the LLM reliably used company_enrichment for prose fields but inconsistently populated the parallel Finding-list/string-list fields from the same evidence, even with progressively more explicit prompting. Add a code-level backfill (products, recent developments, financial signals, strategic initiatives, regulatory signals, risks/opportunities mirrored from SWOT, unknowns, monitoring recommendations) that only ever fills in what the model left empty, never overwrites what it produced. Enrichment tab: reorder sections (Products/Recent updates before Customers/Competitors) and add a per-section "Refresh" button that re-fetches just one of NinjaPear's six independent per-company endpoints when it came back empty - confirmed live that a data-coverage gap (e.g. Amazon returning no products) is real provider behavior, not a bug. Auth: the first account registered on a deployment with zero existing admins is now auto-promoted to admin, closing the chicken-and-egg gap where the only path to admin access was direct DB access. Self-heals if the last admin ever deletes their account. Also bumps nginx's proxy_read_timeout for api.ciagent.org to cover the enrichment refresh's synchronous funding-endpoint call (up to 5 minutes per NinjaPear's docs). Co-Authored-By: Claude Sonnet 5 <[email protected]>
This commit is contained in:
@@ -125,11 +125,25 @@ async def register(
|
||||
if existing is not None:
|
||||
raise ConflictError("An account with this email already exists")
|
||||
|
||||
# Bootstraps admin access on a fresh deployment - otherwise the only way
|
||||
# to ever get an admin account is direct DB access, which is a real
|
||||
# chicken-and-egg problem for anyone self-hosting from a clean clone.
|
||||
# Checked by admin *count*, not total user count, so this also
|
||||
# self-heals if the last admin ever deletes their own account (Settings
|
||||
# -> Delete account) - the next registration becomes admin again rather
|
||||
# than leaving the deployment permanently admin-less. A benign race is
|
||||
# possible if two people register in the same instant on a brand-new,
|
||||
# zero-admin deployment (both could become admin) - acceptable for a
|
||||
# bootstrapping check that only ever matters once, before any real
|
||||
# traffic exists.
|
||||
is_first_admin = await repo.count_admins() == 0
|
||||
|
||||
user = await repo.create(
|
||||
email=payload.email,
|
||||
password_hash=hash_password(payload.password),
|
||||
display_name=payload.display_name,
|
||||
timezone=payload.timezone,
|
||||
is_admin=is_first_admin,
|
||||
# Test suite has no inbox to read a real code from - same
|
||||
# app_env == "test" precedent already used to disable rate limiting
|
||||
# (app/core/rate_limit.py). The code-generation/sending/throttle
|
||||
|
||||
@@ -20,6 +20,7 @@ from datetime import UTC, datetime
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
|
||||
from app.core.config import Settings
|
||||
from app.core.errors import NotFoundError
|
||||
from app.core.logging import get_logger
|
||||
from app.enrichment.base import EnrichmentProvider
|
||||
from app.models.company import Company
|
||||
@@ -174,3 +175,104 @@ async def enrich_company(
|
||||
failed_sections=list(errors.keys()),
|
||||
)
|
||||
return enrichment
|
||||
|
||||
|
||||
# Every entry here maps 1:1 onto one of NinjaPear's independent
|
||||
# per-company endpoints (see app/enrichment/ninjapear.py) - refreshing one
|
||||
# section never re-fetches or touches any of the others.
|
||||
REFRESHABLE_SECTIONS = ("details", "funding", "updates", "competitors", "products", "customers")
|
||||
|
||||
|
||||
async def refresh_section(
|
||||
db: AsyncSession,
|
||||
settings: Settings,
|
||||
provider: EnrichmentProvider,
|
||||
company: Company,
|
||||
section: str,
|
||||
) -> CompanyEnrichment:
|
||||
"""Re-runs a single NinjaPear endpoint for a company that already has
|
||||
an enrichment record but came back empty or failed for just this one
|
||||
section (e.g. `products` returning `[]` while everything else
|
||||
succeeded - a real, honest gap in NinjaPear's own coverage for that
|
||||
company, not a bug in this app). Only ever replaces this section's own
|
||||
slice of `data`/`errors` - every other section's stored data is left
|
||||
exactly as it was."""
|
||||
if section not in REFRESHABLE_SECTIONS:
|
||||
raise ValueError(f"Unknown enrichment section: {section}")
|
||||
|
||||
repo = CompanyEnrichmentRepository(db)
|
||||
existing = await repo.get_for_company(company.id)
|
||||
if existing is None:
|
||||
raise NotFoundError("Company has no enrichment record to refresh")
|
||||
|
||||
website = company.official_website
|
||||
data = dict(existing.data)
|
||||
errors = dict(existing.errors)
|
||||
credits_spent = existing.credits_spent or 0
|
||||
|
||||
try:
|
||||
if section == "details":
|
||||
details = await provider.get_company_details(company.name, website)
|
||||
data["employee_count"] = details.employee_count_range
|
||||
data["description"] = details.description
|
||||
data["industry"] = details.industry
|
||||
data["founded_year"] = details.founded_year
|
||||
data["specialties"] = details.specialties
|
||||
new_leadership = [m.model_dump() for m in details.leadership_team]
|
||||
# A details refresh re-fetches the leadership roster itself,
|
||||
# but not the separate per-leader work_email/person_profile
|
||||
# lookups - preserve those by matching on name so a refresh
|
||||
# never regresses contact info that was already found.
|
||||
old_by_name = {
|
||||
m.get("name"): m for m in (data.get("leadership_team") or []) if m.get("name")
|
||||
}
|
||||
for member in new_leadership:
|
||||
old = old_by_name.get(member.get("name"))
|
||||
if old:
|
||||
member["work_email"] = member.get("work_email") or old.get("work_email")
|
||||
member["profile_url"] = member.get("profile_url") or old.get("profile_url")
|
||||
member["bio"] = member.get("bio") or old.get("bio")
|
||||
data["leadership_team"] = new_leadership
|
||||
elif section == "funding":
|
||||
funding = await provider.get_funding(company.name, website)
|
||||
data["funding"] = funding.model_dump()
|
||||
elif section == "updates":
|
||||
updates = await provider.get_updates(company.name, website)
|
||||
data["recent_updates"] = [u.model_dump() for u in updates]
|
||||
elif section == "competitors":
|
||||
competitors = await provider.get_competitors(company.name, website)
|
||||
data["competitors"] = [c.model_dump() for c in competitors]
|
||||
elif section == "products":
|
||||
products = await provider.get_products(company.name, website)
|
||||
data["products"] = [p.model_dump() for p in products]
|
||||
elif section == "customers":
|
||||
customers = await provider.get_customers(company.name, website)
|
||||
data["customers"] = [c.model_dump() for c in customers]
|
||||
errors.pop(section, None)
|
||||
credits_spent += _CREDIT_COSTS.get(section, 0)
|
||||
except Exception as exc: # noqa: BLE001 - surfaced via `errors`, same as the initial run
|
||||
errors[section] = str(exc)
|
||||
logger.warning(
|
||||
"enrichment_section_refresh_failed",
|
||||
section=section,
|
||||
company_id=str(company.id),
|
||||
error=str(exc),
|
||||
)
|
||||
|
||||
new_status = EnrichmentStatus.PARTIAL if errors else EnrichmentStatus.COMPLETE
|
||||
enrichment = await repo.upsert(
|
||||
company.id,
|
||||
status=new_status,
|
||||
data=data,
|
||||
errors=errors,
|
||||
credits_spent=credits_spent,
|
||||
fetched_at=datetime.now(UTC),
|
||||
)
|
||||
await db.commit()
|
||||
logger.info(
|
||||
"company_enrichment_section_refreshed",
|
||||
company_id=str(company.id),
|
||||
section=section,
|
||||
succeeded=section not in errors,
|
||||
)
|
||||
return enrichment
|
||||
|
||||
@@ -23,7 +23,8 @@ from app.models.enums import EnrichmentStatus, ReportType, SourceStatus
|
||||
from app.models.report import Report
|
||||
from app.models.source import Source
|
||||
from app.models.source_document import SourceDocument
|
||||
from app.prompts.report_generation import generate_report
|
||||
from app.prompts.report_generation import ReportContent, generate_report
|
||||
from app.prompts.schemas import ConfidenceLabel, EvidenceRef, Finding
|
||||
from app.repositories.company_enrichment_repository import CompanyEnrichmentRepository
|
||||
from app.repositories.report_repository import ReportRepository
|
||||
from app.services import company_service
|
||||
@@ -31,6 +32,210 @@ from app.services.report_markdown import render_report_markdown
|
||||
|
||||
_MAX_DOCUMENTS = 40
|
||||
_MAX_CHANGES = 20
|
||||
_MAX_BACKFILLED_PRODUCTS = 30
|
||||
_MAX_BACKFILLED_UPDATES = 15
|
||||
_MAX_BACKFILLED_FUNDING_ROUNDS = 15
|
||||
|
||||
|
||||
def _format_amount(amount: object) -> str | None:
|
||||
if amount is None:
|
||||
return None
|
||||
text = str(amount)
|
||||
return f"${int(text):,}" if text.isdigit() else text
|
||||
|
||||
|
||||
def _backfill_from_enrichment(content: ReportContent, enrichment: dict | None) -> None:
|
||||
"""company_enrichment.products/recent_updates/funding map onto
|
||||
products_and_services/recent_developments/financial_signals with no
|
||||
inference required - a direct transcription, not a judgment call.
|
||||
Confirmed live that the LLM is nonetheless inconsistent about
|
||||
populating these Finding-list fields from enrichment alone (it
|
||||
reliably uses the same data for prose fields like company_overview,
|
||||
but repeated real API calls with increasingly explicit instructions
|
||||
still came back with these lists empty). Rather than keep fighting
|
||||
prompt compliance for a purely mechanical transform, backfill directly
|
||||
from the source data whenever the model leaves a field empty despite
|
||||
the data being available - this only ever *adds* real, non-fabricated
|
||||
content the model chose not to surface, never overwrites what the
|
||||
model did produce."""
|
||||
if not enrichment:
|
||||
return
|
||||
|
||||
if not content.products_and_services:
|
||||
content.products_and_services = [
|
||||
Finding(
|
||||
headline=product["name"],
|
||||
summary=product.get("description") or product.get("category") or product["name"],
|
||||
category=product.get("category"),
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
)
|
||||
for product in enrichment.get("products", [])[:_MAX_BACKFILLED_PRODUCTS]
|
||||
if product.get("name")
|
||||
]
|
||||
|
||||
if not content.recent_developments:
|
||||
content.recent_developments = [
|
||||
Finding(
|
||||
headline=update["text"],
|
||||
summary=f"{update.get('type', 'update').capitalize()} published by the company.",
|
||||
date=update.get("date"),
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
evidence=(
|
||||
[EvidenceRef(url=update["url"], description="Company-published update.")]
|
||||
if update.get("url")
|
||||
else []
|
||||
),
|
||||
)
|
||||
for update in enrichment.get("recent_updates", [])[:_MAX_BACKFILLED_UPDATES]
|
||||
if update.get("text")
|
||||
]
|
||||
|
||||
if not content.financial_signals:
|
||||
funding = enrichment.get("funding") or {}
|
||||
findings = []
|
||||
total_raised = _format_amount(funding.get("total_raised"))
|
||||
if total_raised:
|
||||
findings.append(
|
||||
Finding(
|
||||
headline=f"Total funding raised: {total_raised}",
|
||||
summary="Cumulative funding raised across all rounds, per company_enrichment.",
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
)
|
||||
)
|
||||
for round_ in funding.get("rounds", [])[:_MAX_BACKFILLED_FUNDING_ROUNDS]:
|
||||
name = round_.get("round_name")
|
||||
if not name:
|
||||
continue
|
||||
amount = _format_amount(round_.get("amount"))
|
||||
investors = round_.get("investors") or []
|
||||
headline = name.replace("_", " ").title() + (f" - {amount}" if amount else "")
|
||||
findings.append(
|
||||
Finding(
|
||||
headline=headline,
|
||||
summary=(
|
||||
f"Investors: {', '.join(investors)}."
|
||||
if investors
|
||||
else "No investor detail provided."
|
||||
),
|
||||
date=round_.get("date"),
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
)
|
||||
)
|
||||
content.financial_signals = findings
|
||||
|
||||
|
||||
def _mirror_swot_into_flat_lists(content: ReportContent) -> None:
|
||||
"""risks/opportunities are meant to be the same analysis as
|
||||
swot.threats/swot.opportunities, just in a flat top-level list rather
|
||||
than nested under swot - not a second, independent judgment call.
|
||||
Confirmed live: the model reliably populates the SWOT fields but is
|
||||
inconsistent about also populating these parallel flat fields with the
|
||||
same content, even though nothing about them requires different
|
||||
evidence. Mirror rather than re-derive, since the model already did
|
||||
the real analytical work once."""
|
||||
if not content.risks and content.swot.threats:
|
||||
content.risks = list(content.swot.threats)
|
||||
if not content.opportunities and content.swot.opportunities:
|
||||
content.opportunities = list(content.swot.opportunities)
|
||||
|
||||
|
||||
def _backfill_reflective_sections(
|
||||
content: ReportContent,
|
||||
*,
|
||||
enrichment: dict | None,
|
||||
documents: list[dict],
|
||||
changes: list[dict],
|
||||
company_name: str,
|
||||
) -> None:
|
||||
"""strategic_initiatives, regulatory_and_legal_signals,
|
||||
unknowns_and_missing_data, and monitoring_recommendations are the
|
||||
fields the LLM was most persistently reluctant to populate even after
|
||||
two rounds of explicit prompt strengthening (confirmed live - see
|
||||
SYSTEM_PROMPT in report_generation.py). Unlike products/recent_updates,
|
||||
these don't have a single obvious mechanical source, but each still
|
||||
has SOMETHING honest and non-fabricated to fall back on:
|
||||
- strategic_initiatives: company_enrichment.specialties names the
|
||||
company's own stated focus areas - a real, low-confidence signal,
|
||||
not invented.
|
||||
- regulatory_and_legal_signals: when truly nothing applies, a single
|
||||
insufficient_evidence Finding is what the prompt already asks the
|
||||
model to emit in this situation instead of leaving the list empty -
|
||||
the model just isn't doing it reliably, so this fills the same gap.
|
||||
- unknowns_and_missing_data / monitoring_recommendations: these are
|
||||
meta-analysis of the evidence set itself, not claims about the
|
||||
company, so they can be derived honestly from what evidence this
|
||||
pipeline actually did or didn't collect for this company."""
|
||||
enrichment = enrichment or {}
|
||||
|
||||
if not content.strategic_initiatives:
|
||||
specialties = enrichment.get("specialties") or []
|
||||
content.strategic_initiatives = [
|
||||
Finding(
|
||||
headline=f"Focus area: {specialty}",
|
||||
summary=(
|
||||
f"{company_name} lists \"{specialty}\" among its specialties, "
|
||||
"indicating an area of strategic focus."
|
||||
),
|
||||
confidence=ConfidenceLabel.POSSIBLE,
|
||||
)
|
||||
for specialty in specialties
|
||||
if specialty
|
||||
]
|
||||
|
||||
if not content.regulatory_and_legal_signals:
|
||||
content.regulatory_and_legal_signals = [
|
||||
Finding(
|
||||
headline="No regulatory or legal signals identified",
|
||||
summary=(
|
||||
"No licensing, compliance, jurisdictional, or legal-structure "
|
||||
"information was present in the available evidence for "
|
||||
f"{company_name}."
|
||||
),
|
||||
confidence=ConfidenceLabel.INSUFFICIENT_EVIDENCE,
|
||||
)
|
||||
]
|
||||
|
||||
if not content.unknowns_and_missing_data:
|
||||
unknowns = []
|
||||
if not enrichment:
|
||||
unknowns.append(
|
||||
"No third-party company enrichment data was available for this company."
|
||||
)
|
||||
else:
|
||||
if not enrichment.get("funding", {}).get("total_raised") and not enrichment.get(
|
||||
"funding", {}
|
||||
).get("rounds"):
|
||||
unknowns.append(
|
||||
"No detailed financial statements, revenue, or funding figures were found."
|
||||
)
|
||||
if not enrichment.get("leadership_team"):
|
||||
unknowns.append("No leadership or executive team data was found.")
|
||||
if not enrichment.get("customers"):
|
||||
unknowns.append("No named customers or case studies were found.")
|
||||
if not documents:
|
||||
unknowns.append(
|
||||
"No source documents have been collected yet from ongoing monitoring, so "
|
||||
"hiring, technology, patent, and manufacturing signals are not yet available."
|
||||
)
|
||||
if not changes:
|
||||
unknowns.append(
|
||||
"No changes have been detected yet between monitoring runs for this company."
|
||||
)
|
||||
content.unknowns_and_missing_data = unknowns
|
||||
|
||||
if not content.monitoring_recommendations:
|
||||
recommendations = [
|
||||
f"Monitor company_enrichment for {company_name} on its next refresh for changes "
|
||||
"to products, leadership, or funding.",
|
||||
"Track newly published company updates for announcements of new initiatives "
|
||||
"or partnerships.",
|
||||
]
|
||||
if documents:
|
||||
recommendations.append(
|
||||
"Continue reviewing newly collected source documents for signals not yet "
|
||||
"reflected in company_enrichment."
|
||||
)
|
||||
content.monitoring_recommendations = recommendations
|
||||
|
||||
|
||||
async def _gather_documents(db: AsyncSession, company_id: uuid.UUID) -> list[dict]:
|
||||
@@ -130,6 +335,15 @@ async def generate_and_persist_report(
|
||||
public_identifiers=company.public_identifiers,
|
||||
enrichment=enrichment,
|
||||
)
|
||||
_backfill_from_enrichment(content, enrichment)
|
||||
_mirror_swot_into_flat_lists(content)
|
||||
_backfill_reflective_sections(
|
||||
content,
|
||||
enrichment=enrichment,
|
||||
documents=documents,
|
||||
changes=changes,
|
||||
company_name=company.name,
|
||||
)
|
||||
|
||||
generated_at = datetime.now(UTC)
|
||||
model_name = _model_name(settings, llm)
|
||||
|
||||
Reference in New Issue
Block a user