Files
Ten31-Portal/backend/ten31portal/routers/import_router.py
T
Johnny 5 758fac3ab5
Build Service / build (push) Canceled after 0s
fix: handle duplicate security names and non-JSON error responses
- Disambiguate duplicate (company, security) pairs in Carta exports:
  e.g. two 'Warrants' under BIP21 become 'Warrants' and 'Warrants (2)'
- Tested against Fund II: 33 holdings, 41 positions (3 companies with
  duplicate tranches), zero errors
- Frontend: graceful handling of 500 responses (shows status code
  instead of JSON parse crash)
2026-06-08 03:20:37 +00:00

677 lines
23 KiB
Python

"""CSV/XLSX import endpoints for entities and schedule of investments."""
import csv
import io
import re
from datetime import date, datetime
from typing import Any
import openpyxl
from fastapi import APIRouter, Depends, File, HTTPException, Query, UploadFile
from sqlmodel import Session, select
from ten31portal.audit import record_audit
from ten31portal.auth import require_role
from ten31portal.database import get_session
from ten31portal.models import (
Entity, EntityStatus, EntityType, Holding, Position,
User, UserRole, Valuation, ValuationRound, RoundStatus,
)
router = APIRouter(prefix="/api/import", tags=["import"])
# --- Column maps ---
# Confirmed against real Carta exports 2026-06-07.
# Entity CSV import (Issue 9) — still needs a real entity-level export to confirm.
ENTITY_COLUMN_MAP: dict[str, str] = {
# UNCONFIRMED: update after inspecting a real Carta entities export
"Entity Name": "name",
"Entity Type": "type",
"Vintage Year": "vintage_year",
"Fund Size": "fund_size_cents",
}
ENTITY_TYPE_MAP: dict[str, EntityType] = {
"Fund": EntityType.fund,
"fund": EntityType.fund,
"SPV": EntityType.spv,
"spv": EntityType.spv,
"GP": EntityType.gp,
"gp": EntityType.gp,
"Mgmt Co": EntityType.mgmt_co,
"mgmt_co": EntityType.mgmt_co,
"Management Company": EntityType.mgmt_co,
}
# Schedule of Investments XLSX import (Issue 10)
# CONFIRMED against Carta export: "Low Time Preference Fund I, LLC" 2026-06-07
#
# Carta XLSX layout:
# Row 1: Entity name (e.g. "Low Time Preference Fund I, LLC")
# Row 2: Metadata line ("As of MM/DD/YYYY • Generated by ...")
# Row 3: Empty
# Row 4: Headers
# Row 5: Empty
# Rows 6+: Data (company rows alternate with position rows, separated by empty rows)
# Last data row: "Total" summary
#
# Column mapping (0-indexed from Row 4 headers):
# A (0): "Investment" — company name on company/subtotal rows
# B (1): "Asset" — security name on position rows
# C (2): "Investment date" — datetime on position rows
# D (3): "Shares" — number (0 for SAFEs/membership interests)
# E (4): "Cost" — dollar amount (float)
# F (5): "Value" — dollar amount (float)
# G (6): "Last Valuation date" — datetime on position rows
# H (7): "Gain/Loss" — derived, skip
# I (8): "Cost per share" — derived, skip
# J (9): "FMV per share" — derived, skip
# K (10): "Percent of partners' capital" — skip (phase 2)
#
# Company rows: A has name, B is empty, D/E/F have subtotals
# Position rows: A is empty, B has security name, all columns populated
# SAFEs: shares = 0, cost per share = 0, FMV per share = 0
SCHEDULE_COLUMNS = {
"investment": 0, # A — company name
"asset": 1, # B — security name
"inv_date": 2, # C — investment date
"shares": 3, # D — share count
"cost": 4, # E — cost in dollars
"value": 5, # F — value in dollars
"val_date": 6, # G — last valuation date
}
def _parse_money(raw: str) -> int | None:
"""Parse dollar strings like '$3.3M', '$61.9M', '$1,234,567', '3300000' to cents."""
if not raw or not raw.strip():
return None
s = raw.strip().replace(",", "").replace("$", "")
multiplier = 1
if s.upper().endswith("M"):
multiplier = 1_000_000
s = s[:-1]
elif s.upper().endswith("K"):
multiplier = 1_000
s = s[:-1]
elif s.upper().endswith("B"):
multiplier = 1_000_000_000
s = s[:-1]
try:
return round(float(s) * multiplier * 100)
except ValueError:
return None
def _parse_date(raw: str) -> date | None:
"""Try common date formats."""
for fmt in ("%Y-%m-%d", "%m/%d/%Y", "%m/%d/%y", "%Y/%m/%d"):
try:
return datetime.strptime(raw.strip(), fmt).date()
except ValueError:
continue
return None
def _dollars_to_cents(val: float | int) -> int:
"""Convert dollar amount to integer cents."""
return round(float(val) * 100)
# --- Entity import (CSV) ---
@router.post("/entities")
def import_entities(
file: UploadFile = File(...),
commit: bool = Query(default=True),
user: User = Depends(require_role(UserRole.approver, UserRole.cfo)),
session: Session = Depends(get_session),
) -> dict[str, Any]:
content = file.file.read().decode("utf-8-sig")
reader = csv.DictReader(io.StringIO(content))
results: list[dict] = []
errors: list[dict] = []
created = 0
updated = 0
for i, row in enumerate(reader, start=2):
mapped: dict[str, Any] = {}
for csv_col, model_field in ENTITY_COLUMN_MAP.items():
val = row.get(csv_col)
if val is None:
for k, v in row.items():
if k.strip().lower() == csv_col.lower():
val = v
break
if val is not None:
mapped[model_field] = val.strip()
parsed: dict[str, Any] = {}
row_errors: list[str] = []
name = mapped.get("name")
if not name:
row_errors.append("Missing entity name")
else:
parsed["name"] = name
type_raw = mapped.get("type", "")
entity_type = ENTITY_TYPE_MAP.get(type_raw)
if entity_type is None and type_raw:
row_errors.append(f"Unknown entity type: {type_raw}")
elif entity_type:
parsed["type"] = entity_type
vy = mapped.get("vintage_year")
if vy:
try:
parsed["vintage_year"] = int(vy)
except ValueError:
row_errors.append(f"Invalid vintage year: {vy}")
fs = mapped.get("fund_size_cents")
if fs:
cents = _parse_money(fs)
if cents is None:
row_errors.append(f"Cannot parse fund size: {fs}")
else:
parsed["fund_size_cents"] = cents
if row_errors:
errors.append({"row": i, "errors": row_errors, "raw": dict(row)})
continue
if not parsed.get("name"):
continue
existing = session.exec(select(Entity).where(Entity.name == parsed["name"])).first()
action = "update" if existing else "create"
results.append({
"row": i,
"action": action,
"name": parsed["name"],
"type": parsed.get("type", EntityType.fund).value if parsed.get("type") else None,
"vintage_year": parsed.get("vintage_year"),
"fund_size_cents": parsed.get("fund_size_cents"),
})
if commit:
if existing:
if "type" in parsed:
existing.type = parsed["type"]
if "vintage_year" in parsed:
existing.vintage_year = parsed["vintage_year"]
if "fund_size_cents" in parsed:
existing.fund_size_cents = parsed["fund_size_cents"]
session.add(existing)
updated += 1
else:
entity = Entity(
name=parsed["name"],
type=parsed.get("type", EntityType.fund),
vintage_year=parsed.get("vintage_year"),
fund_size_cents=parsed.get("fund_size_cents"),
)
session.add(entity)
created += 1
if commit:
record_audit(session, user.id, "import_entities", "entity", None, {
"created": created, "updated": updated, "errors": len(errors),
})
session.commit()
return {
"committed": commit,
"preview": results,
"errors": errors,
"summary": {"created": created, "updated": updated, "error_rows": len(errors)},
}
# --- Schedule of investments import (XLSX) ---
def _parse_schedule_xlsx(file_bytes: bytes) -> tuple[str | None, list[dict], list[dict], list[dict]]:
"""
Parse a Carta Schedule of Investments XLSX export.
Returns: (entity_name, holdings_preview, positions_preview, errors)
"""
wb = openpyxl.load_workbook(io.BytesIO(file_bytes), data_only=True)
ws = wb.active
entity_name: str | None = None
holdings_preview: list[dict] = []
positions_preview: list[dict] = []
errors: list[dict] = []
# Row 1: entity name
row1_val = ws.cell(row=1, column=1).value
if row1_val:
entity_name = str(row1_val).strip()
# Row 4: headers — validate
expected_headers = {0: "Investment", 1: "Asset", 4: "Cost", 5: "Value"}
for col_idx, expected in expected_headers.items():
actual = ws.cell(row=4, column=col_idx + 1).value
if actual and str(actual).strip() != expected:
errors.append({
"row": 4,
"errors": [f"Expected header '{expected}' in column {chr(65 + col_idx)}, got '{actual}'"],
})
# Parse data rows (starting at row 6)
current_company: str | None = None
seen_companies: set[str] = set()
# Track (company, security) occurrences to disambiguate duplicate tranches
security_counts: dict[tuple[str, str], int] = {}
for row_idx in range(6, ws.max_row + 1):
col_a = ws.cell(row=row_idx, column=1).value # Investment (company)
col_b = ws.cell(row=row_idx, column=2).value # Asset (security)
col_c = ws.cell(row=row_idx, column=3).value # Investment date
col_d = ws.cell(row=row_idx, column=4).value # Shares
col_e = ws.cell(row=row_idx, column=5).value # Cost
col_f = ws.cell(row=row_idx, column=6).value # Value
col_g = ws.cell(row=row_idx, column=7).value # Last valuation date
# Skip empty rows
if col_a is None and col_b is None:
continue
# Skip total row
if col_a and str(col_a).strip().lower() == "total":
continue
# Company subtotal row: col A has name, col B is empty
if col_a and not col_b:
company_name = str(col_a).strip()
current_company = company_name
if company_name not in seen_companies:
holdings_preview.append({
"row": row_idx,
"company_name": company_name,
})
seen_companies.add(company_name)
continue
# Position row: col B has security name
if col_b:
raw_security_name = str(col_b).strip()
company = current_company or "(unknown)"
# Disambiguate duplicate (company, security) pairs
# e.g. two "Warrants" under BIP21 become "Warrants" and "Warrants (2)"
pair_key = (company, raw_security_name)
security_counts[pair_key] = security_counts.get(pair_key, 0) + 1
if security_counts[pair_key] == 1:
security_name = raw_security_name
else:
security_name = f"{raw_security_name} ({security_counts[pair_key]})"
# Parse investment date
inv_date: date | None = None
if isinstance(col_c, datetime):
inv_date = col_c.date()
elif col_c:
inv_date = _parse_date(str(col_c))
# Parse shares (could be 0 for SAFEs)
shares: str | None = None
if col_d is not None:
shares_val = float(col_d)
if shares_val != 0:
# Preserve precision: use string repr
shares = str(col_d) if not isinstance(col_d, float) else f"{col_d:g}"
# shares stays None for zero (SAFEs)
# Parse cost (dollars)
cost_cents: int | None = None
if col_e is not None:
cost_cents = _dollars_to_cents(col_e)
# Parse value (dollars)
value_cents: int | None = None
if col_f is not None:
value_cents = _dollars_to_cents(col_f)
# Parse valuation date
val_date: date | None = None
if isinstance(col_g, datetime):
val_date = col_g.date()
elif col_g:
val_date = _parse_date(str(col_g))
positions_preview.append({
"row": row_idx,
"company_name": company,
"security_name": security_name,
"investment_date": str(inv_date) if inv_date else None,
"shares": shares,
"cost_cents": cost_cents,
"value_cents": value_cents,
"valuation_date": str(val_date) if val_date else None,
})
return entity_name, holdings_preview, positions_preview, errors
@router.post("/schedule")
def import_schedule(
file: UploadFile = File(...),
as_of: date = Query(...),
entity_id: int | None = Query(default=None),
create_entity_type: EntityType = Query(default=EntityType.fund),
create_vintage_year: int | None = Query(default=None),
commit: bool = Query(default=True),
user: User = Depends(require_role(UserRole.approver, UserRole.cfo)),
session: Session = Depends(get_session),
) -> dict[str, Any]:
file_bytes = file.file.read()
filename = file.filename or ""
# Parse file first to get source_entity_name
if filename.endswith(".xlsx") or filename.endswith(".xls"):
source_entity_name, holdings_preview, positions_preview, errors = _parse_schedule_xlsx(file_bytes)
else:
source_entity_name = None
holdings_preview, positions_preview, errors = _parse_schedule_csv(file_bytes)
# Resolve entity
entity: Entity | None = None
entity_resolution: str = "existing" # "existing" | "matched" | "will_create"
if entity_id is not None:
# Explicit entity_id: use it directly
entity = session.get(Entity, entity_id)
if entity is None:
raise HTTPException(status_code=404, detail="Entity not found")
entity_resolution = "existing"
elif source_entity_name:
# Try matching by name from row 1
entity = session.exec(
select(Entity).where(Entity.name == source_entity_name)
).first()
if entity:
entity_resolution = "matched"
else:
entity_resolution = "will_create"
else:
raise HTTPException(
status_code=400,
detail="No entity_id provided and the file has no entity name in row 1. Provide entity_id or upload a Carta XLSX with the fund name.",
)
# Dry-run: report resolution without writing
if not commit:
resolution_info: dict[str, Any] = {"resolution": entity_resolution}
if entity:
resolution_info["entity_id"] = entity.id
resolution_info["entity_name"] = entity.name
resolution_info["entity_type"] = entity.type.value
else:
resolution_info["entity_name"] = source_entity_name
resolution_info["entity_type"] = create_entity_type.value
resolution_info["vintage_year"] = create_vintage_year
return {
"committed": False,
"source_entity_name": source_entity_name,
"entity": resolution_info,
"holdings": holdings_preview,
"positions": positions_preview,
"errors": errors,
"seed_round": {"entity_id": entity.id if entity else None, "quarter_end": str(as_of)},
}
# Commit path: create entity if needed
if entity_resolution == "will_create" and source_entity_name:
entity = Entity(
name=source_entity_name,
type=create_entity_type,
vintage_year=create_vintage_year,
)
session.add(entity)
session.flush()
record_audit(session, user.id, "create", "entity", entity.id, {
"name": source_entity_name,
"type": create_entity_type.value,
"source": "schedule_import",
})
assert entity is not None
resolved_entity_id = entity.id
# Check for any existing round at this quarter
existing_round = session.exec(
select(ValuationRound).where(
ValuationRound.entity_id == resolved_entity_id,
ValuationRound.quarter_end == as_of,
)
).first()
if existing_round:
kind = "seed" if existing_round.is_seed else "valuation"
raise HTTPException(
status_code=409,
detail=f"A {kind} round already exists for this entity and quarter ({as_of}). Cannot overwrite signed history.",
)
# Commit: create holdings, positions, seed round
holding_map: dict[str, Holding] = {}
for hp in holdings_preview:
name = hp["company_name"]
existing = session.exec(
select(Holding).where(
Holding.entity_id == resolved_entity_id,
Holding.company_name == name,
)
).first()
if existing:
holding_map[name] = existing
else:
h = Holding(entity_id=resolved_entity_id, company_name=name)
session.add(h)
session.flush()
holding_map[name] = h
# Create seed round
seed_round = ValuationRound(
entity_id=resolved_entity_id,
quarter_end=as_of,
status=RoundStatus.approved,
is_seed=True,
approved_by=user.id,
approved_at=datetime.utcnow(),
)
session.add(seed_round)
session.flush()
positions_created = 0
positions_updated = 0
for pp in positions_preview:
company = pp["company_name"]
holding = holding_map.get(company)
if holding is None:
existing = session.exec(
select(Holding).where(
Holding.entity_id == resolved_entity_id,
Holding.company_name == company,
)
).first()
if existing:
holding = existing
else:
holding = Holding(entity_id=resolved_entity_id, company_name=company)
session.add(holding)
session.flush()
holding_map[company] = holding
# Upsert position by (holding, security_name)
pos = session.exec(
select(Position).where(
Position.holding_id == holding.id,
Position.security_name == pp["security_name"],
)
).first()
if pos is None:
pos = Position(
holding_id=holding.id,
security_name=pp["security_name"],
investment_date=_parse_date(pp["investment_date"]) if pp["investment_date"] else as_of,
shares=pp["shares"],
cost_cents=pp["cost_cents"] or 0,
)
session.add(pos)
session.flush()
positions_created += 1
else:
if pp["investment_date"]:
pos.investment_date = _parse_date(pp["investment_date"])
if pp["shares"]:
pos.shares = pp["shares"]
if pp["cost_cents"] is not None:
pos.cost_cents = pp["cost_cents"]
session.add(pos)
session.flush()
positions_updated += 1
# Attach valuation to seed round
val = Valuation(
round_id=seed_round.id,
position_id=pos.id,
value_cents=pp["value_cents"] or 0,
)
session.add(val)
record_audit(session, user.id, "import_schedule", "entity", resolved_entity_id, {
"source_entity_name": source_entity_name,
"entity_resolution": entity_resolution,
"holdings": len(holding_map),
"positions_created": positions_created,
"positions_updated": positions_updated,
"seed_quarter": str(as_of),
})
session.commit()
return {
"committed": True,
"source_entity_name": source_entity_name,
"entity": {
"resolution": entity_resolution,
"entity_id": resolved_entity_id,
"entity_name": entity.name,
"entity_type": entity.type.value,
},
"holdings_count": len(holding_map),
"positions_count": len(positions_preview),
"positions_created": positions_created,
"positions_updated": positions_updated,
"seed_round_id": seed_round.id,
"errors": errors,
}
def _parse_schedule_csv(file_bytes: bytes) -> tuple[list[dict], list[dict], list[dict]]:
"""Legacy CSV parser fallback."""
content = file_bytes.decode("utf-8-sig")
reader = csv.DictReader(io.StringIO(content))
CSV_COLUMN_MAP = {
"Company": "company_name",
"Investment": "company_name",
"Security": "security_name",
"Asset": "security_name",
"Investment Date": "investment_date",
"Investment date": "investment_date",
"Shares": "shares",
"Cost": "cost_cents",
"Value": "value_cents",
}
holdings_preview: list[dict] = []
positions_preview: list[dict] = []
errors: list[dict] = []
current_company: str | None = None
for i, row in enumerate(reader, start=2):
mapped: dict[str, Any] = {}
for csv_col, model_field in CSV_COLUMN_MAP.items():
val = row.get(csv_col)
if val is None:
for k, v in row.items():
if k.strip().lower() == csv_col.lower():
val = v
break
if val is not None:
mapped[model_field] = val.strip()
company = mapped.get("company_name", "").strip()
security = mapped.get("security_name", "").strip()
if company and not security:
current_company = company
holdings_preview.append({"row": i, "company_name": company})
continue
if not company and current_company:
company = current_company
elif company:
if company not in [h["company_name"] for h in holdings_preview]:
holdings_preview.append({"row": i, "company_name": company})
current_company = company
if not security:
continue
row_errors: list[str] = []
inv_date = None
if mapped.get("investment_date"):
inv_date = _parse_date(mapped["investment_date"])
if inv_date is None:
row_errors.append(f"Cannot parse date: {mapped['investment_date']}")
shares = None
raw_shares = mapped.get("shares", "")
if raw_shares:
s = raw_shares.replace(",", "")
try:
sv = float(s)
if sv != 0:
shares = s
except ValueError:
row_errors.append(f"Cannot parse shares: {raw_shares}")
cost_cents = None
if mapped.get("cost_cents"):
cost_cents = _parse_money(mapped["cost_cents"])
if cost_cents is None:
row_errors.append(f"Cannot parse cost: {mapped['cost_cents']}")
value_cents = None
if mapped.get("value_cents"):
value_cents = _parse_money(mapped["value_cents"])
if value_cents is None:
row_errors.append(f"Cannot parse value: {mapped['value_cents']}")
if row_errors:
errors.append({"row": i, "errors": row_errors, "raw": dict(row)})
continue
positions_preview.append({
"row": i,
"company_name": company,
"security_name": security,
"investment_date": str(inv_date) if inv_date else None,
"shares": shares,
"cost_cents": cost_cents,
"value_cents": value_cents,
})
return holdings_preview, positions_preview, errors