Backfill sparse report sections, add enrichment section refresh, and bootstrap first-admin
Reports: the LLM reliably used company_enrichment for prose fields but inconsistently populated the parallel Finding-list/string-list fields from the same evidence, even with progressively more explicit prompting. Add a code-level backfill (products, recent developments, financial signals, strategic initiatives, regulatory signals, risks/opportunities mirrored from SWOT, unknowns, monitoring recommendations) that only ever fills in what the model left empty, never overwrites what it produced. Enrichment tab: reorder sections (Products/Recent updates before Customers/Competitors) and add a per-section "Refresh" button that re-fetches just one of NinjaPear's six independent per-company endpoints when it came back empty - confirmed live that a data-coverage gap (e.g. Amazon returning no products) is real provider behavior, not a bug. Auth: the first account registered on a deployment with zero existing admins is now auto-promoted to admin, closing the chicken-and-egg gap where the only path to admin access was direct DB access. Self-heals if the last admin ever deletes their account. Also bumps nginx's proxy_read_timeout for api.ciagent.org to cover the enrichment refresh's synchronous funding-endpoint call (up to 5 minutes per NinjaPear's docs). Co-Authored-By: Claude Sonnet 5 <[email protected]>
This commit is contained in:
@@ -23,7 +23,8 @@ from app.models.enums import EnrichmentStatus, ReportType, SourceStatus
|
||||
from app.models.report import Report
|
||||
from app.models.source import Source
|
||||
from app.models.source_document import SourceDocument
|
||||
from app.prompts.report_generation import generate_report
|
||||
from app.prompts.report_generation import ReportContent, generate_report
|
||||
from app.prompts.schemas import ConfidenceLabel, EvidenceRef, Finding
|
||||
from app.repositories.company_enrichment_repository import CompanyEnrichmentRepository
|
||||
from app.repositories.report_repository import ReportRepository
|
||||
from app.services import company_service
|
||||
@@ -31,6 +32,210 @@ from app.services.report_markdown import render_report_markdown
|
||||
|
||||
_MAX_DOCUMENTS = 40
|
||||
_MAX_CHANGES = 20
|
||||
_MAX_BACKFILLED_PRODUCTS = 30
|
||||
_MAX_BACKFILLED_UPDATES = 15
|
||||
_MAX_BACKFILLED_FUNDING_ROUNDS = 15
|
||||
|
||||
|
||||
def _format_amount(amount: object) -> str | None:
|
||||
if amount is None:
|
||||
return None
|
||||
text = str(amount)
|
||||
return f"${int(text):,}" if text.isdigit() else text
|
||||
|
||||
|
||||
def _backfill_from_enrichment(content: ReportContent, enrichment: dict | None) -> None:
|
||||
"""company_enrichment.products/recent_updates/funding map onto
|
||||
products_and_services/recent_developments/financial_signals with no
|
||||
inference required - a direct transcription, not a judgment call.
|
||||
Confirmed live that the LLM is nonetheless inconsistent about
|
||||
populating these Finding-list fields from enrichment alone (it
|
||||
reliably uses the same data for prose fields like company_overview,
|
||||
but repeated real API calls with increasingly explicit instructions
|
||||
still came back with these lists empty). Rather than keep fighting
|
||||
prompt compliance for a purely mechanical transform, backfill directly
|
||||
from the source data whenever the model leaves a field empty despite
|
||||
the data being available - this only ever *adds* real, non-fabricated
|
||||
content the model chose not to surface, never overwrites what the
|
||||
model did produce."""
|
||||
if not enrichment:
|
||||
return
|
||||
|
||||
if not content.products_and_services:
|
||||
content.products_and_services = [
|
||||
Finding(
|
||||
headline=product["name"],
|
||||
summary=product.get("description") or product.get("category") or product["name"],
|
||||
category=product.get("category"),
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
)
|
||||
for product in enrichment.get("products", [])[:_MAX_BACKFILLED_PRODUCTS]
|
||||
if product.get("name")
|
||||
]
|
||||
|
||||
if not content.recent_developments:
|
||||
content.recent_developments = [
|
||||
Finding(
|
||||
headline=update["text"],
|
||||
summary=f"{update.get('type', 'update').capitalize()} published by the company.",
|
||||
date=update.get("date"),
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
evidence=(
|
||||
[EvidenceRef(url=update["url"], description="Company-published update.")]
|
||||
if update.get("url")
|
||||
else []
|
||||
),
|
||||
)
|
||||
for update in enrichment.get("recent_updates", [])[:_MAX_BACKFILLED_UPDATES]
|
||||
if update.get("text")
|
||||
]
|
||||
|
||||
if not content.financial_signals:
|
||||
funding = enrichment.get("funding") or {}
|
||||
findings = []
|
||||
total_raised = _format_amount(funding.get("total_raised"))
|
||||
if total_raised:
|
||||
findings.append(
|
||||
Finding(
|
||||
headline=f"Total funding raised: {total_raised}",
|
||||
summary="Cumulative funding raised across all rounds, per company_enrichment.",
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
)
|
||||
)
|
||||
for round_ in funding.get("rounds", [])[:_MAX_BACKFILLED_FUNDING_ROUNDS]:
|
||||
name = round_.get("round_name")
|
||||
if not name:
|
||||
continue
|
||||
amount = _format_amount(round_.get("amount"))
|
||||
investors = round_.get("investors") or []
|
||||
headline = name.replace("_", " ").title() + (f" - {amount}" if amount else "")
|
||||
findings.append(
|
||||
Finding(
|
||||
headline=headline,
|
||||
summary=(
|
||||
f"Investors: {', '.join(investors)}."
|
||||
if investors
|
||||
else "No investor detail provided."
|
||||
),
|
||||
date=round_.get("date"),
|
||||
confidence=ConfidenceLabel.CONFIRMED,
|
||||
)
|
||||
)
|
||||
content.financial_signals = findings
|
||||
|
||||
|
||||
def _mirror_swot_into_flat_lists(content: ReportContent) -> None:
|
||||
"""risks/opportunities are meant to be the same analysis as
|
||||
swot.threats/swot.opportunities, just in a flat top-level list rather
|
||||
than nested under swot - not a second, independent judgment call.
|
||||
Confirmed live: the model reliably populates the SWOT fields but is
|
||||
inconsistent about also populating these parallel flat fields with the
|
||||
same content, even though nothing about them requires different
|
||||
evidence. Mirror rather than re-derive, since the model already did
|
||||
the real analytical work once."""
|
||||
if not content.risks and content.swot.threats:
|
||||
content.risks = list(content.swot.threats)
|
||||
if not content.opportunities and content.swot.opportunities:
|
||||
content.opportunities = list(content.swot.opportunities)
|
||||
|
||||
|
||||
def _backfill_reflective_sections(
|
||||
content: ReportContent,
|
||||
*,
|
||||
enrichment: dict | None,
|
||||
documents: list[dict],
|
||||
changes: list[dict],
|
||||
company_name: str,
|
||||
) -> None:
|
||||
"""strategic_initiatives, regulatory_and_legal_signals,
|
||||
unknowns_and_missing_data, and monitoring_recommendations are the
|
||||
fields the LLM was most persistently reluctant to populate even after
|
||||
two rounds of explicit prompt strengthening (confirmed live - see
|
||||
SYSTEM_PROMPT in report_generation.py). Unlike products/recent_updates,
|
||||
these don't have a single obvious mechanical source, but each still
|
||||
has SOMETHING honest and non-fabricated to fall back on:
|
||||
- strategic_initiatives: company_enrichment.specialties names the
|
||||
company's own stated focus areas - a real, low-confidence signal,
|
||||
not invented.
|
||||
- regulatory_and_legal_signals: when truly nothing applies, a single
|
||||
insufficient_evidence Finding is what the prompt already asks the
|
||||
model to emit in this situation instead of leaving the list empty -
|
||||
the model just isn't doing it reliably, so this fills the same gap.
|
||||
- unknowns_and_missing_data / monitoring_recommendations: these are
|
||||
meta-analysis of the evidence set itself, not claims about the
|
||||
company, so they can be derived honestly from what evidence this
|
||||
pipeline actually did or didn't collect for this company."""
|
||||
enrichment = enrichment or {}
|
||||
|
||||
if not content.strategic_initiatives:
|
||||
specialties = enrichment.get("specialties") or []
|
||||
content.strategic_initiatives = [
|
||||
Finding(
|
||||
headline=f"Focus area: {specialty}",
|
||||
summary=(
|
||||
f"{company_name} lists \"{specialty}\" among its specialties, "
|
||||
"indicating an area of strategic focus."
|
||||
),
|
||||
confidence=ConfidenceLabel.POSSIBLE,
|
||||
)
|
||||
for specialty in specialties
|
||||
if specialty
|
||||
]
|
||||
|
||||
if not content.regulatory_and_legal_signals:
|
||||
content.regulatory_and_legal_signals = [
|
||||
Finding(
|
||||
headline="No regulatory or legal signals identified",
|
||||
summary=(
|
||||
"No licensing, compliance, jurisdictional, or legal-structure "
|
||||
"information was present in the available evidence for "
|
||||
f"{company_name}."
|
||||
),
|
||||
confidence=ConfidenceLabel.INSUFFICIENT_EVIDENCE,
|
||||
)
|
||||
]
|
||||
|
||||
if not content.unknowns_and_missing_data:
|
||||
unknowns = []
|
||||
if not enrichment:
|
||||
unknowns.append(
|
||||
"No third-party company enrichment data was available for this company."
|
||||
)
|
||||
else:
|
||||
if not enrichment.get("funding", {}).get("total_raised") and not enrichment.get(
|
||||
"funding", {}
|
||||
).get("rounds"):
|
||||
unknowns.append(
|
||||
"No detailed financial statements, revenue, or funding figures were found."
|
||||
)
|
||||
if not enrichment.get("leadership_team"):
|
||||
unknowns.append("No leadership or executive team data was found.")
|
||||
if not enrichment.get("customers"):
|
||||
unknowns.append("No named customers or case studies were found.")
|
||||
if not documents:
|
||||
unknowns.append(
|
||||
"No source documents have been collected yet from ongoing monitoring, so "
|
||||
"hiring, technology, patent, and manufacturing signals are not yet available."
|
||||
)
|
||||
if not changes:
|
||||
unknowns.append(
|
||||
"No changes have been detected yet between monitoring runs for this company."
|
||||
)
|
||||
content.unknowns_and_missing_data = unknowns
|
||||
|
||||
if not content.monitoring_recommendations:
|
||||
recommendations = [
|
||||
f"Monitor company_enrichment for {company_name} on its next refresh for changes "
|
||||
"to products, leadership, or funding.",
|
||||
"Track newly published company updates for announcements of new initiatives "
|
||||
"or partnerships.",
|
||||
]
|
||||
if documents:
|
||||
recommendations.append(
|
||||
"Continue reviewing newly collected source documents for signals not yet "
|
||||
"reflected in company_enrichment."
|
||||
)
|
||||
content.monitoring_recommendations = recommendations
|
||||
|
||||
|
||||
async def _gather_documents(db: AsyncSession, company_id: uuid.UUID) -> list[dict]:
|
||||
@@ -130,6 +335,15 @@ async def generate_and_persist_report(
|
||||
public_identifiers=company.public_identifiers,
|
||||
enrichment=enrichment,
|
||||
)
|
||||
_backfill_from_enrichment(content, enrichment)
|
||||
_mirror_swot_into_flat_lists(content)
|
||||
_backfill_reflective_sections(
|
||||
content,
|
||||
enrichment=enrichment,
|
||||
documents=documents,
|
||||
changes=changes,
|
||||
company_name=company.name,
|
||||
)
|
||||
|
||||
generated_at = datetime.now(UTC)
|
||||
model_name = _model_name(settings, llm)
|
||||
|
||||
Reference in New Issue
Block a user