"""CSV/XLSX import endpoints for entities and schedule of investments.""" import csv import io import re from datetime import date, datetime from typing import Any import openpyxl from fastapi import APIRouter, Depends, File, HTTPException, Query, UploadFile from sqlmodel import Session, select from ten31portal.audit import record_audit from ten31portal.auth import require_role from ten31portal.database import get_session from ten31portal.models import ( Entity, EntityStatus, EntityType, Holding, Position, User, UserRole, Valuation, ValuationRound, RoundStatus, ) router = APIRouter(prefix="/api/import", tags=["import"]) # --- Column maps --- # Confirmed against real Carta exports 2026-06-07. # Entity CSV import (Issue 9) — still needs a real entity-level export to confirm. ENTITY_COLUMN_MAP: dict[str, str] = { # UNCONFIRMED: update after inspecting a real Carta entities export "Entity Name": "name", "Entity Type": "type", "Vintage Year": "vintage_year", "Fund Size": "fund_size_cents", } ENTITY_TYPE_MAP: dict[str, EntityType] = { "Fund": EntityType.fund, "fund": EntityType.fund, "SPV": EntityType.spv, "spv": EntityType.spv, "GP": EntityType.gp, "gp": EntityType.gp, "Mgmt Co": EntityType.mgmt_co, "mgmt_co": EntityType.mgmt_co, "Management Company": EntityType.mgmt_co, } # Schedule of Investments XLSX import (Issue 10) # CONFIRMED against Carta export: "Low Time Preference Fund I, LLC" 2026-06-07 # # Carta XLSX layout: # Row 1: Entity name (e.g. "Low Time Preference Fund I, LLC") # Row 2: Metadata line ("As of MM/DD/YYYY • Generated by ...") # Row 3: Empty # Row 4: Headers # Row 5: Empty # Rows 6+: Data (company rows alternate with position rows, separated by empty rows) # Last data row: "Total" summary # # Column mapping (0-indexed from Row 4 headers): # A (0): "Investment" — company name on company/subtotal rows # B (1): "Asset" — security name on position rows # C (2): "Investment date" — datetime on position rows # D (3): "Shares" — number (0 for SAFEs/membership interests) # E (4): "Cost" — dollar amount (float) # F (5): "Value" — dollar amount (float) # G (6): "Last Valuation date" — datetime on position rows # H (7): "Gain/Loss" — derived, skip # I (8): "Cost per share" — derived, skip # J (9): "FMV per share" — derived, skip # K (10): "Percent of partners' capital" — skip (phase 2) # # Company rows: A has name, B is empty, D/E/F have subtotals # Position rows: A is empty, B has security name, all columns populated # SAFEs: shares = 0, cost per share = 0, FMV per share = 0 SCHEDULE_COLUMNS = { "investment": 0, # A — company name "asset": 1, # B — security name "inv_date": 2, # C — investment date "shares": 3, # D — share count "cost": 4, # E — cost in dollars "value": 5, # F — value in dollars "val_date": 6, # G — last valuation date } def _parse_money(raw: str) -> int | None: """Parse dollar strings like '$3.3M', '$61.9M', '$1,234,567', '3300000' to cents.""" if not raw or not raw.strip(): return None s = raw.strip().replace(",", "").replace("$", "") multiplier = 1 if s.upper().endswith("M"): multiplier = 1_000_000 s = s[:-1] elif s.upper().endswith("K"): multiplier = 1_000 s = s[:-1] elif s.upper().endswith("B"): multiplier = 1_000_000_000 s = s[:-1] try: return round(float(s) * multiplier * 100) except ValueError: return None def _parse_date(raw: str) -> date | None: """Try common date formats.""" for fmt in ("%Y-%m-%d", "%m/%d/%Y", "%m/%d/%y", "%Y/%m/%d"): try: return datetime.strptime(raw.strip(), fmt).date() except ValueError: continue return None def _dollars_to_cents(val: float | int) -> int: """Convert dollar amount to integer cents.""" return round(float(val) * 100) # --- Entity import (CSV) --- @router.post("/entities") def import_entities( file: UploadFile = File(...), commit: bool = Query(default=True), user: User = Depends(require_role(UserRole.approver, UserRole.cfo)), session: Session = Depends(get_session), ) -> dict[str, Any]: content = file.file.read().decode("utf-8-sig") reader = csv.DictReader(io.StringIO(content)) results: list[dict] = [] errors: list[dict] = [] created = 0 updated = 0 for i, row in enumerate(reader, start=2): mapped: dict[str, Any] = {} for csv_col, model_field in ENTITY_COLUMN_MAP.items(): val = row.get(csv_col) if val is None: for k, v in row.items(): if k.strip().lower() == csv_col.lower(): val = v break if val is not None: mapped[model_field] = val.strip() parsed: dict[str, Any] = {} row_errors: list[str] = [] name = mapped.get("name") if not name: row_errors.append("Missing entity name") else: parsed["name"] = name type_raw = mapped.get("type", "") entity_type = ENTITY_TYPE_MAP.get(type_raw) if entity_type is None and type_raw: row_errors.append(f"Unknown entity type: {type_raw}") elif entity_type: parsed["type"] = entity_type vy = mapped.get("vintage_year") if vy: try: parsed["vintage_year"] = int(vy) except ValueError: row_errors.append(f"Invalid vintage year: {vy}") fs = mapped.get("fund_size_cents") if fs: cents = _parse_money(fs) if cents is None: row_errors.append(f"Cannot parse fund size: {fs}") else: parsed["fund_size_cents"] = cents if row_errors: errors.append({"row": i, "errors": row_errors, "raw": dict(row)}) continue if not parsed.get("name"): continue existing = session.exec(select(Entity).where(Entity.name == parsed["name"])).first() action = "update" if existing else "create" results.append({ "row": i, "action": action, "name": parsed["name"], "type": parsed.get("type", EntityType.fund).value if parsed.get("type") else None, "vintage_year": parsed.get("vintage_year"), "fund_size_cents": parsed.get("fund_size_cents"), }) if commit: if existing: if "type" in parsed: existing.type = parsed["type"] if "vintage_year" in parsed: existing.vintage_year = parsed["vintage_year"] if "fund_size_cents" in parsed: existing.fund_size_cents = parsed["fund_size_cents"] session.add(existing) updated += 1 else: entity = Entity( name=parsed["name"], type=parsed.get("type", EntityType.fund), vintage_year=parsed.get("vintage_year"), fund_size_cents=parsed.get("fund_size_cents"), ) session.add(entity) created += 1 if commit: record_audit(session, user.id, "import_entities", "entity", None, { "created": created, "updated": updated, "errors": len(errors), }) session.commit() return { "committed": commit, "preview": results, "errors": errors, "summary": {"created": created, "updated": updated, "error_rows": len(errors)}, } # --- Schedule of investments import (XLSX) --- def _parse_schedule_xlsx(file_bytes: bytes) -> tuple[str | None, list[dict], list[dict], list[dict]]: """ Parse a Carta Schedule of Investments XLSX export. Returns: (entity_name, holdings_preview, positions_preview, errors) """ wb = openpyxl.load_workbook(io.BytesIO(file_bytes), data_only=True) ws = wb.active entity_name: str | None = None holdings_preview: list[dict] = [] positions_preview: list[dict] = [] errors: list[dict] = [] # Row 1: entity name row1_val = ws.cell(row=1, column=1).value if row1_val: entity_name = str(row1_val).strip() # Row 4: headers — validate expected_headers = {0: "Investment", 1: "Asset", 4: "Cost", 5: "Value"} for col_idx, expected in expected_headers.items(): actual = ws.cell(row=4, column=col_idx + 1).value if actual and str(actual).strip() != expected: errors.append({ "row": 4, "errors": [f"Expected header '{expected}' in column {chr(65 + col_idx)}, got '{actual}'"], }) # Parse data rows (starting at row 6) current_company: str | None = None seen_companies: set[str] = set() # Track (company, security) occurrences to disambiguate duplicate tranches security_counts: dict[tuple[str, str], int] = {} for row_idx in range(6, ws.max_row + 1): col_a = ws.cell(row=row_idx, column=1).value # Investment (company) col_b = ws.cell(row=row_idx, column=2).value # Asset (security) col_c = ws.cell(row=row_idx, column=3).value # Investment date col_d = ws.cell(row=row_idx, column=4).value # Shares col_e = ws.cell(row=row_idx, column=5).value # Cost col_f = ws.cell(row=row_idx, column=6).value # Value col_g = ws.cell(row=row_idx, column=7).value # Last valuation date # Skip empty rows if col_a is None and col_b is None: continue # Skip total row if col_a and str(col_a).strip().lower() == "total": continue # Company subtotal row: col A has name, col B is empty if col_a and not col_b: company_name = str(col_a).strip() current_company = company_name if company_name not in seen_companies: holdings_preview.append({ "row": row_idx, "company_name": company_name, }) seen_companies.add(company_name) continue # Position row: col B has security name if col_b: raw_security_name = str(col_b).strip() company = current_company or "(unknown)" # Disambiguate duplicate (company, security) pairs # e.g. two "Warrants" under BIP21 become "Warrants" and "Warrants (2)" pair_key = (company, raw_security_name) security_counts[pair_key] = security_counts.get(pair_key, 0) + 1 if security_counts[pair_key] == 1: security_name = raw_security_name else: security_name = f"{raw_security_name} ({security_counts[pair_key]})" # Parse investment date inv_date: date | None = None if isinstance(col_c, datetime): inv_date = col_c.date() elif col_c: inv_date = _parse_date(str(col_c)) # Parse shares (could be 0 for SAFEs) shares: str | None = None if col_d is not None: shares_val = float(col_d) if shares_val != 0: # Preserve precision: use string repr shares = str(col_d) if not isinstance(col_d, float) else f"{col_d:g}" # shares stays None for zero (SAFEs) # Parse cost (dollars) cost_cents: int | None = None if col_e is not None: cost_cents = _dollars_to_cents(col_e) # Parse value (dollars) value_cents: int | None = None if col_f is not None: value_cents = _dollars_to_cents(col_f) # Parse valuation date val_date: date | None = None if isinstance(col_g, datetime): val_date = col_g.date() elif col_g: val_date = _parse_date(str(col_g)) positions_preview.append({ "row": row_idx, "company_name": company, "security_name": security_name, "investment_date": str(inv_date) if inv_date else None, "shares": shares, "cost_cents": cost_cents, "value_cents": value_cents, "valuation_date": str(val_date) if val_date else None, }) return entity_name, holdings_preview, positions_preview, errors @router.post("/schedule") def import_schedule( file: UploadFile = File(...), as_of: date = Query(...), entity_id: int | None = Query(default=None), create_entity_type: EntityType = Query(default=EntityType.fund), create_vintage_year: int | None = Query(default=None), commit: bool = Query(default=True), user: User = Depends(require_role(UserRole.approver, UserRole.cfo)), session: Session = Depends(get_session), ) -> dict[str, Any]: file_bytes = file.file.read() filename = file.filename or "" # Parse file first to get source_entity_name if filename.endswith(".xlsx") or filename.endswith(".xls"): source_entity_name, holdings_preview, positions_preview, errors = _parse_schedule_xlsx(file_bytes) else: source_entity_name = None holdings_preview, positions_preview, errors = _parse_schedule_csv(file_bytes) # Resolve entity entity: Entity | None = None entity_resolution: str = "existing" # "existing" | "matched" | "will_create" if entity_id is not None: # Explicit entity_id: use it directly entity = session.get(Entity, entity_id) if entity is None: raise HTTPException(status_code=404, detail="Entity not found") entity_resolution = "existing" elif source_entity_name: # Try matching by name from row 1 entity = session.exec( select(Entity).where(Entity.name == source_entity_name) ).first() if entity: entity_resolution = "matched" else: entity_resolution = "will_create" else: raise HTTPException( status_code=400, detail="No entity_id provided and the file has no entity name in row 1. Provide entity_id or upload a Carta XLSX with the fund name.", ) # Dry-run: report resolution without writing if not commit: resolution_info: dict[str, Any] = {"resolution": entity_resolution} if entity: resolution_info["entity_id"] = entity.id resolution_info["entity_name"] = entity.name resolution_info["entity_type"] = entity.type.value else: resolution_info["entity_name"] = source_entity_name resolution_info["entity_type"] = create_entity_type.value resolution_info["vintage_year"] = create_vintage_year return { "committed": False, "source_entity_name": source_entity_name, "entity": resolution_info, "holdings": holdings_preview, "positions": positions_preview, "errors": errors, "seed_round": {"entity_id": entity.id if entity else None, "quarter_end": str(as_of)}, } # Commit path: create entity if needed if entity_resolution == "will_create" and source_entity_name: entity = Entity( name=source_entity_name, type=create_entity_type, vintage_year=create_vintage_year, ) session.add(entity) session.flush() record_audit(session, user.id, "create", "entity", entity.id, { "name": source_entity_name, "type": create_entity_type.value, "source": "schedule_import", }) assert entity is not None resolved_entity_id = entity.id # Check for any existing round at this quarter existing_round = session.exec( select(ValuationRound).where( ValuationRound.entity_id == resolved_entity_id, ValuationRound.quarter_end == as_of, ) ).first() if existing_round: kind = "seed" if existing_round.is_seed else "valuation" raise HTTPException( status_code=409, detail=f"A {kind} round already exists for this entity and quarter ({as_of}). Cannot overwrite signed history.", ) # Commit: create holdings, positions, seed round holding_map: dict[str, Holding] = {} for hp in holdings_preview: name = hp["company_name"] existing = session.exec( select(Holding).where( Holding.entity_id == resolved_entity_id, Holding.company_name == name, ) ).first() if existing: holding_map[name] = existing else: h = Holding(entity_id=resolved_entity_id, company_name=name) session.add(h) session.flush() holding_map[name] = h # Create seed round seed_round = ValuationRound( entity_id=resolved_entity_id, quarter_end=as_of, status=RoundStatus.approved, is_seed=True, approved_by=user.id, approved_at=datetime.utcnow(), ) session.add(seed_round) session.flush() positions_created = 0 positions_updated = 0 for pp in positions_preview: company = pp["company_name"] holding = holding_map.get(company) if holding is None: existing = session.exec( select(Holding).where( Holding.entity_id == resolved_entity_id, Holding.company_name == company, ) ).first() if existing: holding = existing else: holding = Holding(entity_id=resolved_entity_id, company_name=company) session.add(holding) session.flush() holding_map[company] = holding # Upsert position by (holding, security_name) pos = session.exec( select(Position).where( Position.holding_id == holding.id, Position.security_name == pp["security_name"], ) ).first() if pos is None: pos = Position( holding_id=holding.id, security_name=pp["security_name"], investment_date=_parse_date(pp["investment_date"]) if pp["investment_date"] else as_of, shares=pp["shares"], cost_cents=pp["cost_cents"] or 0, ) session.add(pos) session.flush() positions_created += 1 else: if pp["investment_date"]: pos.investment_date = _parse_date(pp["investment_date"]) if pp["shares"]: pos.shares = pp["shares"] if pp["cost_cents"] is not None: pos.cost_cents = pp["cost_cents"] session.add(pos) session.flush() positions_updated += 1 # Attach valuation to seed round val = Valuation( round_id=seed_round.id, position_id=pos.id, value_cents=pp["value_cents"] or 0, ) session.add(val) record_audit(session, user.id, "import_schedule", "entity", resolved_entity_id, { "source_entity_name": source_entity_name, "entity_resolution": entity_resolution, "holdings": len(holding_map), "positions_created": positions_created, "positions_updated": positions_updated, "seed_quarter": str(as_of), }) session.commit() return { "committed": True, "source_entity_name": source_entity_name, "entity": { "resolution": entity_resolution, "entity_id": resolved_entity_id, "entity_name": entity.name, "entity_type": entity.type.value, }, "holdings_count": len(holding_map), "positions_count": len(positions_preview), "positions_created": positions_created, "positions_updated": positions_updated, "seed_round_id": seed_round.id, "errors": errors, } def _parse_schedule_csv(file_bytes: bytes) -> tuple[list[dict], list[dict], list[dict]]: """Legacy CSV parser fallback.""" content = file_bytes.decode("utf-8-sig") reader = csv.DictReader(io.StringIO(content)) CSV_COLUMN_MAP = { "Company": "company_name", "Investment": "company_name", "Security": "security_name", "Asset": "security_name", "Investment Date": "investment_date", "Investment date": "investment_date", "Shares": "shares", "Cost": "cost_cents", "Value": "value_cents", } holdings_preview: list[dict] = [] positions_preview: list[dict] = [] errors: list[dict] = [] current_company: str | None = None for i, row in enumerate(reader, start=2): mapped: dict[str, Any] = {} for csv_col, model_field in CSV_COLUMN_MAP.items(): val = row.get(csv_col) if val is None: for k, v in row.items(): if k.strip().lower() == csv_col.lower(): val = v break if val is not None: mapped[model_field] = val.strip() company = mapped.get("company_name", "").strip() security = mapped.get("security_name", "").strip() if company and not security: current_company = company holdings_preview.append({"row": i, "company_name": company}) continue if not company and current_company: company = current_company elif company: if company not in [h["company_name"] for h in holdings_preview]: holdings_preview.append({"row": i, "company_name": company}) current_company = company if not security: continue row_errors: list[str] = [] inv_date = None if mapped.get("investment_date"): inv_date = _parse_date(mapped["investment_date"]) if inv_date is None: row_errors.append(f"Cannot parse date: {mapped['investment_date']}") shares = None raw_shares = mapped.get("shares", "") if raw_shares: s = raw_shares.replace(",", "") try: sv = float(s) if sv != 0: shares = s except ValueError: row_errors.append(f"Cannot parse shares: {raw_shares}") cost_cents = None if mapped.get("cost_cents"): cost_cents = _parse_money(mapped["cost_cents"]) if cost_cents is None: row_errors.append(f"Cannot parse cost: {mapped['cost_cents']}") value_cents = None if mapped.get("value_cents"): value_cents = _parse_money(mapped["value_cents"]) if value_cents is None: row_errors.append(f"Cannot parse value: {mapped['value_cents']}") if row_errors: errors.append({"row": i, "errors": row_errors, "raw": dict(row)}) continue positions_preview.append({ "row": i, "company_name": company, "security_name": security, "investment_date": str(inv_date) if inv_date else None, "shares": shares, "cost_cents": cost_cents, "value_cents": value_cents, }) return holdings_preview, positions_preview, errors