"""CV Application - Main FastAPI Server.""" import json import os import uuid import subprocess import asyncio from datetime import date, datetime from fastapi import FastAPI, UploadFile, File, HTTPException, Form, Request from fastapi.responses import HTMLResponse, JSONResponse, FileResponse, Response from fastapi.staticfiles import StaticFiles from fastapi.middleware.cors import CORSMiddleware from pathlib import Path import database as db from database import Json from config import get_settings from doc_parser import extract_text import ai_service as ai settings = get_settings() app = FastAPI(title="CV Application", version="1.0.0") app.add_middleware( CORSMiddleware, allow_origins=["*"], allow_credentials=True, allow_methods=["*"], allow_headers=["*"], ) # Ensure upload directory exists os.makedirs(settings.upload_dir, exist_ok=True) # Serve static files static_dir = Path(__file__).parent / "static" static_dir.mkdir(exist_ok=True) app.mount("/static", StaticFiles(directory=str(static_dir)), name="static") INITIALIZED = False @app.on_event("startup") async def startup(): global INITIALIZED db.init_db() INITIALIZED = True # ============================================================ # CANDIDATES / CV UPLOAD # ============================================================ @app.post("/api/candidates/upload") async def upload_cv(file: UploadFile = File(...)): """Upload a CV document, extract text, store it, and trigger AI parsing.""" # Save file file_ext = Path(file.filename).suffix saved_name = f"{uuid.uuid4()}{file_ext}" file_path = os.path.join(settings.upload_dir, saved_name) with open(file_path, "wb") as f: content = await file.read() f.write(content) # Extract text try: raw_text = extract_text(file_path) except Exception as e: raw_text = f"[Extraction error: {str(e)}]" if not raw_text or len(raw_text.strip()) < 10: raw_text = "[No text could be extracted from this document]" # Create candidate record with raw text result = db.execute(""" INSERT INTO candidates (raw_cv_text, source_filename, parse_status) VALUES (%s, %s, 'pending') RETURNING id """, (raw_text, file.filename)) candidate_id = str(result["id"]) # Trigger AI parsing (synchronous for now) try: parsed = ai.parse_cv(raw_text) # Update candidate record db.execute(""" UPDATE candidates SET first_name = %s, last_name = %s, email = %s, phone = %s, address = %s, linkedin = %s, github = %s, website = %s, summary = %s, parse_status = 'parsed', updated_at = NOW() WHERE id = %s """, ( parsed.get("first_name", ""), parsed.get("last_name", ""), parsed.get("email", ""), parsed.get("phone", ""), parsed.get("address", ""), parsed.get("linkedin", ""), parsed.get("github", ""), parsed.get("website", ""), parsed.get("summary", ""), candidate_id )) # Insert skills for skill in parsed.get("skills", []): start_d = _parse_date(skill.get("start_date")) end_d = _parse_date(skill.get("end_date")) db.execute(""" INSERT INTO skills (candidate_id, skill_name, skill_category, proficiency, start_date, end_date) VALUES (%s, %s, %s, %s, %s, %s) """, (candidate_id, skill.get("skill_name", ""), skill.get("skill_category", ""), skill.get("proficiency", ""), start_d, end_d)) # Insert experience for exp in parsed.get("experience", []): start_d = _parse_date(exp.get("start_date")) end_d = _parse_date(exp.get("end_date")) db.execute(""" INSERT INTO experience (candidate_id, company, position, location, start_date, end_date, description, achievements, skills_used) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s) RETURNING id """, (candidate_id, exp.get("company", ""), exp.get("position", ""), exp.get("location", ""), start_d, end_d, exp.get("description", ""), json.dumps(exp.get("achievements", [])), json.dumps(exp.get("skills_used", [])))) # Insert education for edu in parsed.get("education", []): start_d = _parse_date(edu.get("start_date")) end_d = _parse_date(edu.get("end_date")) db.execute(""" INSERT INTO education (candidate_id, institution, degree, field_of_study, start_date, end_date, grade, description) VALUES (%s, %s, %s, %s, %s, %s, %s, %s) """, (candidate_id, edu.get("institution", ""), edu.get("degree", ""), edu.get("field_of_study", ""), start_d, end_d, edu.get("grade", ""), edu.get("description", ""))) # Insert certifications for cert in parsed.get("certifications", []): issue_d = _parse_date(cert.get("issue_date")) expiry_d = _parse_date(cert.get("expiry_date")) db.execute(""" INSERT INTO certifications (candidate_id, name, issuer, issue_date, expiry_date, credential_id) VALUES (%s, %s, %s, %s, %s, %s) """, (candidate_id, cert.get("name", ""), cert.get("issuer", ""), issue_d, expiry_d, cert.get("credential_id", ""))) return {"status": "parsed", "candidate_id": candidate_id, "parsed_data": parsed} except Exception as e: db.execute(""" UPDATE candidates SET parse_status = 'error', parse_error = %s WHERE id = %s """, (str(e), candidate_id)) return {"status": "error", "candidate_id": candidate_id, "error": str(e)} @app.get("/api/candidates") async def list_candidates(search: str = None, page: int = 1, limit: int = 20): """List all candidates with pagination.""" offset = (page - 1) * limit params = [limit, offset] where_clause = "" if search: where_clause = "WHERE (first_name ILIKE %s OR last_name ILIKE %s OR email ILIKE %s OR summary ILIKE %s)" search_param = f"%{search}%" params = [search_param, search_param, search_param, search_param, limit, offset] count_row = db.query(f"SELECT COUNT(*) as total FROM candidates {where_clause}", [f"%{search}%"] * 4 if search else None, fetch='one') rows = db.query(f""" SELECT id, first_name, last_name, email, phone, parse_status, created_at, LEFT(raw_cv_text, 200) as preview FROM candidates {where_clause} ORDER BY created_at DESC LIMIT %s OFFSET %s """, params, fetch='all') return {"candidates": [dict(r) for r in rows], "total": count_row["total"] if count_row else 0, "page": page, "limit": limit} @app.get("/api/candidates/{candidate_id}") async def get_candidate(candidate_id: str): """Get full candidate data including skills, experience, education, certifications.""" candidate = db.query("SELECT * FROM candidates WHERE id = %s", (candidate_id,), fetch='one') if not candidate: raise HTTPException(404, "Candidate not found") skills = db.query("SELECT * FROM skills WHERE candidate_id = %s ORDER BY skill_category, skill_name", (candidate_id,), fetch='all') experience = db.query("SELECT * FROM experience WHERE candidate_id = %s ORDER BY start_date DESC", (candidate_id,), fetch='all') education = db.query("SELECT * FROM education WHERE candidate_id = %s ORDER BY start_date DESC", (candidate_id,), fetch='all') certs = db.query("SELECT * FROM certifications WHERE candidate_id = %s ORDER BY issue_date DESC", (candidate_id,), fetch='all') # Calculate dynamic skill years as of today today = date.today() skills_with_years = [] for s in skills: s_dict = dict(s) s_dict["years_experience"] = ai.calculate_years_experience( s["start_date"], s["end_date"], today ) skills_with_years.append(s_dict) batches = db.query(""" SELECT bi.id AS item_id, bi.batch_id, bi.position_title, bi.match_score, bi.status AS item_status, bi.generated_cv_path, bi.updated_at AS item_updated_at, b.name AS batch_name, b.status AS batch_status FROM batch_items bi JOIN cv_batches b ON b.id = bi.batch_id WHERE bi.candidate_id = %s ORDER BY bi.updated_at DESC """, (candidate_id,), fetch='all') return { "candidate": dict(candidate), "skills": skills_with_years, "experience": [dict(e) for e in experience], "education": [dict(e) for e in education], "certifications": [dict(c) for c in certs], "batches": [dict(b) for b in batches], } @app.delete("/api/candidates/{candidate_id}") async def delete_candidate(candidate_id: str): """Delete a candidate and all related data.""" db.execute("DELETE FROM candidates WHERE id = %s", (candidate_id,)) return {"status": "deleted"} @app.put("/api/candidates/{candidate_id}") async def update_candidate(candidate_id: str, request: Request): """Update candidate fields.""" data = await request.json() fields = ["first_name", "last_name", "email", "phone", "address", "linkedin", "github", "website", "summary"] updates = [] params = [] for f in fields: if f in data: updates.append(f"{f} = %s") params.append(data[f]) if updates: updates.append("updated_at = NOW()") params.append(candidate_id) db.execute(f"UPDATE candidates SET {', '.join(updates)} WHERE id = %s", params) return {"status": "updated"} # ============================================================ # SKILLS / EXPERIENCE / EDUCATION CRUD # ============================================================ @app.post("/api/candidates/{candidate_id}/skills") async def add_skill(candidate_id: str, request: Request): data = await request.json() result = db.execute(""" INSERT INTO skills (candidate_id, skill_name, skill_category, proficiency, start_date, end_date) VALUES (%s, %s, %s, %s, %s, %s) RETURNING id """, (candidate_id, data.get("skill_name", ""), data.get("skill_category", ""), data.get("proficiency", ""), _parse_date(data.get("start_date")), _parse_date(data.get("end_date")))) return {"id": str(result["id"]), "status": "created"} @app.put("/api/skills/{skill_id}") async def update_skill(skill_id: str, request: Request): data = await request.json() fields = ["skill_name", "skill_category", "proficiency", "start_date", "end_date"] updates = [] params = [] for f in fields: if f in data: updates.append(f"{f} = %s") params.append(_parse_date(data[f]) if "date" in f else data[f]) if updates: params.append(skill_id) db.execute(f"UPDATE skills SET {', '.join(updates)} WHERE id = %s", params) return {"status": "updated"} @app.delete("/api/skills/{skill_id}") async def delete_skill(skill_id: str): db.execute("DELETE FROM skills WHERE id = %s", (skill_id,)) return {"status": "deleted"} @app.post("/api/candidates/{candidate_id}/experience") async def add_experience(candidate_id: str, request: Request): data = await request.json() result = db.execute(""" INSERT INTO experience (candidate_id, company, position, location, start_date, end_date, description, achievements, skills_used) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s) RETURNING id """, (candidate_id, data.get("company", ""), data.get("position", ""), data.get("location", ""), _parse_date(data.get("start_date")), _parse_date(data.get("end_date")), data.get("description", ""), json.dumps(data.get("achievements", [])), json.dumps(data.get("skills_used", [])))) return {"id": str(result["id"]), "status": "created"} @app.delete("/api/experience/{exp_id}") async def delete_experience(exp_id: str): db.execute("DELETE FROM experience WHERE id = %s", (exp_id,)) return {"status": "deleted"} # ============================================================ # TEMPLATES # ============================================================ @app.get("/api/templates") async def list_templates(): rows = db.query("SELECT * FROM cv_templates ORDER BY created_at DESC", fetch='all') return {"templates": [dict(r) for r in rows]} @app.get("/api/templates/{template_id}") async def get_template(template_id: str): row = db.query("SELECT * FROM cv_templates WHERE id = %s", (template_id,), fetch='one') if not row: raise HTTPException(404, "Template not found") return dict(row) @app.post("/api/templates") async def create_template(request: Request): data = await request.json() result = db.execute(""" INSERT INTO cv_templates (name, description, template_structure, styling, created_by, generation_prompt) VALUES (%s, %s, %s, %s, %s, %s) RETURNING id """, (data.get("name", ""), data.get("description", ""), json.dumps(data.get("template_structure", {})), data.get("styling", ""), data.get("created_by", "manual"), data.get("generation_prompt", ""))) return {"id": str(result["id"]), "status": "created"} @app.post("/api/templates/generate") async def generate_template_ai(request: Request): """Generate a template using AI from a text description.""" data = await request.json() description = data.get("description", "") if not description: raise HTTPException(400, "Description required") template_structure = ai.generate_template(description) result = db.execute(""" INSERT INTO cv_templates (name, description, template_structure, styling, created_by, generation_prompt) VALUES (%s, %s, %s, %s, 'ai', %s) RETURNING id """, (data.get("name", "AI Generated Template"), description, json.dumps(template_structure.get("sections", template_structure)), template_structure.get("styling", ""), description)) return {"id": str(result["id"]), "template_structure": template_structure, "status": "generated"} @app.put("/api/templates/{template_id}") async def update_template(template_id: str, request: Request): data = await request.json() result = db.execute(""" UPDATE cv_templates SET name = %s, description = %s, template_structure = %s, styling = %s, updated_at = NOW() WHERE id = %s RETURNING id """, (data.get("name", ""), data.get("description", ""), json.dumps(data.get("template_structure", {})), data.get("styling", ""), template_id)) return {"status": "updated"} @app.delete("/api/templates/{template_id}") async def delete_template(template_id: str): db.execute("DELETE FROM cv_templates WHERE id = %s", (template_id,)) return {"status": "deleted"} # ============================================================ # REQUIREMENT REQUESTS # ============================================================ @app.get("/api/requirements") async def list_requirements(): rows = db.query("SELECT * FROM requirement_requests ORDER BY created_at DESC", fetch='all') return {"requirements": [dict(r) for r in rows]} @app.get("/api/requirements/{req_id}") async def get_requirement(req_id: str): row = db.query("SELECT * FROM requirement_requests WHERE id = %s", (req_id,), fetch='one') if not row: raise HTTPException(404, "Requirement not found") # Also get generated CVs for this requirement gen_cvs = db.query("SELECT * FROM generated_cvs WHERE requirement_request_id = %s ORDER BY created_at DESC", (req_id,), fetch='all') return {**dict(row), "generated_cvs": [dict(g) for g in gen_cvs]} @app.post("/api/requirements") async def create_requirement(request: Request): data = await request.json() result = db.execute(""" INSERT INTO requirement_requests (title, description, customer_name, requirements) VALUES (%s, %s, %s, %s) RETURNING id """, (data.get("title", ""), data.get("description", ""), data.get("customer_name", ""), json.dumps(data.get("requirements", [])))) return {"id": str(result["id"]), "status": "created"} @app.put("/api/requirements/{req_id}") async def update_requirement(req_id: str, request: Request): data = await request.json() db.execute(""" UPDATE requirement_requests SET title = %s, description = %s, customer_name = %s, requirements = %s, updated_at = NOW() WHERE id = %s """, (data.get("title", ""), data.get("description", ""), data.get("customer_name", ""), json.dumps(data.get("requirements", [])), req_id)) return {"status": "updated"} @app.delete("/api/requirements/{req_id}") async def delete_requirement(req_id: str): db.execute("DELETE FROM requirement_requests WHERE id = %s", (req_id,)) return {"status": "deleted"} # ============================================================ # MATCHING + CV GENERATION # ============================================================ @app.post("/api/requirements/{req_id}/match") async def match_candidates(req_id: str, request: Request): """Match all candidates against the requirements. Returns matches with scores.""" req = db.query("SELECT * FROM requirement_requests WHERE id = %s", (req_id,), fetch='one') if not req: raise HTTPException(404, "Requirement not found") requirements = req["requirements"] if isinstance(req["requirements"], list) else json.loads(req["requirements"]) # Get all candidates candidates = db.query("SELECT id, first_name, last_name FROM candidates WHERE parse_status = 'parsed'", fetch='all') results = [] for cand in candidates: # Get full candidate data full_data = await get_candidate(str(cand["id"])) try: matches = ai.match_candidate_to_requirements(full_data, requirements) for m in matches: m["candidate_id"] = str(cand["id"]) m["candidate_name"] = f"{cand['first_name']} {cand['last_name']}".strip() results.append(m) except Exception as e: results.append({ "candidate_id": str(cand["id"]), "candidate_name": f"{cand['first_name']} {cand['last_name']}".strip(), "error": str(e) }) # Sort by match_score descending results.sort(key=lambda x: x.get("match_score", 0), reverse=True) return {"matches": results} @app.post("/api/requirements/{req_id}/generate") async def generate_cvs_for_requirement(req_id: str, request: Request): """Generate tailored CVs for candidates that match the requirement. Body: { "candidate_ids": ["uuid1", "uuid2"], // optional, if empty uses all parsed candidates "template_id": "uuid", // optional template "position_title": "specific position" // optional, filter to one position } """ data = await request.json() req = db.query("SELECT * FROM requirement_requests WHERE id = %s", (req_id,), fetch='one') if not req: raise HTTPException(404, "Requirement not found") requirements = req["requirements"] if isinstance(req["requirements"], list) else json.loads(req["requirements"]) # Get template if specified template_structure = {"sections": []} if data.get("template_id"): tmpl = db.query("SELECT * FROM cv_templates WHERE id = %s", (data["template_id"],), fetch='one') if tmpl: template_structure = tmpl["template_structure"] if isinstance(tmpl["template_structure"], dict) else json.loads(tmpl["template_structure"]) # Get candidates candidate_ids = data.get("candidate_ids", []) if not candidate_ids: # Use all parsed candidates (could be limited by match score threshold) candidates = db.query("SELECT id FROM candidates WHERE parse_status = 'parsed'", fetch='all') candidate_ids = [str(c["id"]) for c in candidates] # Filter to specific position if requested if data.get("position_title"): requirements = [r for r in requirements if r.get("position_title") == data["position_title"]] gen_date = date.today() generated = [] for cand_id in candidate_ids: full_data = await get_candidate(cand_id) for req_item in requirements: try: result = ai.generate_aligned_cv(full_data, req_item, template_structure, gen_date) gen_row = db.execute(""" INSERT INTO generated_cvs (candidate_id, requirement_request_id, template_id, position_title, generated_content, generated_data, generation_date, match_score, match_reasoning, status) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, 'draft') RETURNING id """, (cand_id, req_id, data.get("template_id"), req_item.get("position_title", ""), result.get("generated_content", ""), json.dumps(result.get("generated_data", {})), gen_date, result.get("match_score", 0), result.get("changes_made", ""))) generated.append({ "id": str(gen_row["id"]), "candidate_id": cand_id, "candidate_name": f"{full_data['candidate'].get('first_name','')} {full_data['candidate'].get('last_name','')}".strip(), "position_title": req_item.get("position_title", ""), "status": "draft" }) except Exception as e: generated.append({ "candidate_id": cand_id, "position_title": req_item.get("position_title", ""), "error": str(e) }) return {"generated": generated, "generation_date": gen_date.isoformat()} # ============================================================ # GENERATED CVs # ============================================================ @app.get("/api/generated-cvs") async def list_generated_cvs(req_id: str = None, candidate_id: str = None, batch_id: str = None): """List batch-generated CV exports (plus any legacy generated_cvs rows).""" where_parts = [ "bi.generated_cv_path IS NOT NULL", "bi.generated_cv_path <> ''", "bi.status = 'approved'", ] params = [] if candidate_id: where_parts.append("bi.candidate_id = %s") params.append(candidate_id) if batch_id: where_parts.append("bi.batch_id = %s") params.append(batch_id) where_clause = "WHERE " + " AND ".join(where_parts) batch_rows = db.query(f""" SELECT bi.id, bi.candidate_id, bi.batch_id, bi.position_title, bi.match_score, bi.match_reasoning, bi.status AS item_status, bi.generated_cv_path, bi.created_at, bi.updated_at, c.first_name, c.last_name, b.name AS batch_name, b.status AS batch_status, b.template_id, 'batch' AS source FROM batch_items bi JOIN candidates c ON c.id = bi.candidate_id JOIN cv_batches b ON b.id = bi.batch_id {where_clause} ORDER BY bi.updated_at DESC """, params if params else None, fetch='all') # Keep legacy requirements-generated rows visible until fully retired. legacy_where = [] legacy_params = [] if req_id: legacy_where.append("gc.requirement_request_id = %s") legacy_params.append(req_id) if candidate_id: legacy_where.append("gc.candidate_id = %s") legacy_params.append(candidate_id) legacy_clause = ("WHERE " + " AND ".join(legacy_where)) if legacy_where else "" legacy_rows = db.query(f""" SELECT gc.id, gc.candidate_id, gc.requirement_request_id AS batch_id, gc.position_title, gc.match_score, gc.match_reasoning, gc.status AS item_status, NULL AS generated_cv_path, gc.created_at, gc.updated_at, c.first_name, c.last_name, rr.title AS batch_name, rr.status AS batch_status, gc.template_id, 'legacy' AS source FROM generated_cvs gc LEFT JOIN candidates c ON gc.candidate_id = c.id LEFT JOIN requirement_requests rr ON rr.id = gc.requirement_request_id {legacy_clause} ORDER BY gc.created_at DESC """, legacy_params if legacy_params else None, fetch='all') combined = [dict(r) for r in (batch_rows or [])] + [dict(r) for r in (legacy_rows or [])] combined.sort(key=lambda r: str(r.get("updated_at") or r.get("created_at") or ""), reverse=True) return {"generated_cvs": combined} @app.get("/api/generated-cvs/{gen_id}") async def get_generated_cv(gen_id: str): row = db.query(""" SELECT gc.*, c.first_name, c.last_name, c.email FROM generated_cvs gc LEFT JOIN candidates c ON gc.candidate_id = c.id WHERE gc.id = %s """, (gen_id,), fetch='one') if not row: raise HTTPException(404, "Generated CV not found") return dict(row) @app.put("/api/generated-cvs/{gen_id}") async def update_generated_cv(gen_id: str, request: Request): """Update generated CV content (user edits).""" data = await request.json() db.execute(""" UPDATE generated_cvs SET edited_content = %s, edited_at = NOW(), status = %s, updated_at = NOW() WHERE id = %s """, (data.get("edited_content", data.get("generated_content", "")), data.get("status", "draft"), gen_id)) return {"status": "updated"} @app.delete("/api/generated-cvs/{gen_id}") async def delete_generated_cv(gen_id: str): db.execute("DELETE FROM generated_cvs WHERE id = %s", (gen_id,)) return {"status": "deleted"} # ============================================================ # CHAT # ============================================================ @app.post("/api/chat") async def chat_with_ai(request: Request): """Chat with the AI assistant. Body: {message, conversation_id?, context?}""" data = await request.json() message = data.get("message", "") conversation_id = data.get("conversation_id") context = data.get("context", "") # Get conversation history history = [] if conversation_id: msgs = db.query(""" SELECT role, content FROM chat_messages WHERE conversation_id = %s ORDER BY created_at ASC LIMIT 20 """, (conversation_id,), fetch='all') history = [dict(m) for m in msgs] else: # Create new conversation conv = db.execute(""" INSERT INTO chat_conversations (context_type, title) VALUES (%s, %s) RETURNING id """, (data.get("context_type", "general"), message[:50])) conversation_id = str(conv["id"]) # Save user message db.execute(""" INSERT INTO chat_messages (conversation_id, role, content) VALUES (%s, 'user', %s) """, (conversation_id, message)) # Get AI response response = ai.chat(message, history, context) # Save AI response db.execute(""" INSERT INTO chat_messages (conversation_id, role, content) VALUES (%s, 'assistant', %s) """, (conversation_id, response)) return {"response": response, "conversation_id": conversation_id} @app.get("/api/chat/conversations") async def list_conversations(): rows = db.query(""" SELECT c.*, (SELECT content FROM chat_messages WHERE conversation_id = c.id ORDER BY created_at DESC LIMIT 1) as last_message FROM chat_conversations ORDER BY updated_at DESC """, fetch='all') return {"conversations": [dict(r) for r in rows]} @app.get("/api/chat/conversations/{conv_id}") async def get_conversation(conv_id: str): msgs = db.query(""" SELECT * FROM chat_messages WHERE conversation_id = %s ORDER BY created_at ASC """, (conv_id,), fetch='all') return {"messages": [dict(m) for m in msgs]} # ============================================================ # DASHBOARD STATS # ============================================================ @app.get("/api/stats") async def get_stats(): candidates = db.query("SELECT COUNT(*) as count FROM candidates", fetch='one') parsed = db.query("SELECT COUNT(*) as count FROM candidates WHERE parse_status = 'parsed'", fetch='one') templates = db.query("SELECT COUNT(*) as count FROM cv_templates", fetch='one') requirements = db.query("SELECT COUNT(*) as count FROM requirement_requests", fetch='one') generated = db.query("SELECT COUNT(*) as count FROM generated_cvs", fetch='one') skills_count = db.query("SELECT COUNT(DISTINCT skill_name) as count FROM skills", fetch='one') return { "total_candidates": candidates["count"] if candidates else 0, "parsed_candidates": parsed["count"] if parsed else 0, "total_templates": templates["count"] if templates else 0, "total_requirements": requirements["count"] if requirements else 0, "total_generated_cvs": generated["count"] if generated else 0, "unique_skills": skills_count["count"] if skills_count else 0 } # ============================================================ # UTILITY # ============================================================ def _parse_date(d): """Parse a date string or return None.""" if d is None or d == "" or d == "null": return None if isinstance(d, date): return d try: return datetime.strptime(d, "%Y-%m-%d").date() except (ValueError, TypeError): try: return datetime.strptime(d, "%Y-%m-%dT%H:%M:%S").date() except (ValueError, TypeError): try: # Try just year return datetime.strptime(d, "%Y").date() except (ValueError, TypeError): return None # ============================================================ # PDF RENDERING (Puppeteer) # ============================================================ @app.post("/api/render-pdf") async def render_pdf(request: Request): """Render a CV template + candidate data to PDF using the Puppeteer renderer. Body: { "template": {JSON CV Template schema}, "cv_data": {candidate data}, "generation_date": "YYYY-MM-DD" (optional) } Returns: {"pdf_url": "/api/pdf/", "html_url": "/api/html/"} """ data = await request.json() template = data.get("template", {}) cv_data = data.get("cv_data", {}) gen_date = data.get("generation_date", date.today().isoformat()) return await _do_render_pdf(template, cv_data, gen_date) async def _do_render_pdf(template, cv_data, gen_date): """Core PDF rendering logic — calls the Node Puppeteer renderer.""" output_dir = os.path.join(os.path.dirname(os.path.abspath(__file__)), "renderer", "output") os.makedirs(output_dir, exist_ok=True) template_path = os.path.join(output_dir, f"template_{uuid.uuid4().hex[:8]}.json") data_path = os.path.join(output_dir, f"cvdata_{uuid.uuid4().hex[:8]}.json") with open(template_path, "w") as f: json.dump(template, f) with open(data_path, "w") as f: json.dump(cv_data, f, default=str) # Create a Node script that calls the renderer script = f""" const {{ renderToPDF }} = require('/root/workspace/cv-app/renderer/render.js'); const template = require('{template_path}'); const cvData = require('{data_path}'); renderToPDF(template, cvData, {{ generationDate: '{gen_date}' }}) .then(r => console.log(JSON.stringify({{ htmlPath: r.htmlPath, pdfPath: r.pdfPath }}))) .catch(e => {{ console.error(e.message); process.exit(1); }}); """ script_path = os.path.join(output_dir, f"render_{uuid.uuid4().hex[:8]}.js") with open(script_path, "w") as f: f.write(script) try: result = subprocess.run( ["node", script_path], capture_output=True, text=True, timeout=60, cwd="/root/workspace/cv-app/renderer" ) if result.returncode != 0: raise HTTPException(500, f"Render failed: {result.stderr[:500]}") output = json.loads(result.stdout.strip()) pdf_filename = os.path.basename(output["pdfPath"]) html_filename = os.path.basename(output["htmlPath"]) return { "pdf_url": f"/api/pdf/{pdf_filename}", "html_url": f"/api/html/{html_filename}", "pdf_path": output["pdfPath"] } except subprocess.TimeoutExpired: raise HTTPException(500, "PDF render timed out") finally: # Clean up temp files for p in [template_path, data_path, script_path]: try: os.unlink(p) except: pass @app.get("/api/pdf/{filename}") async def serve_pdf(filename: str): """Serve a generated PDF file.""" pdf_path = os.path.join(os.path.dirname(os.path.abspath(__file__)), "renderer", "output", filename) if not os.path.exists(pdf_path): raise HTTPException(404, "PDF not found") return FileResponse(pdf_path, media_type="application/pdf", filename=filename) @app.get("/api/html/{filename}") async def serve_html(filename: str): """Serve a generated HTML file.""" html_path = os.path.join(os.path.dirname(os.path.abspath(__file__)), "renderer", "output", filename) if not os.path.exists(html_path): raise HTTPException(404, "HTML not found") return FileResponse(html_path, media_type="text/html") @app.post("/api/generated-cvs/{gen_id}/render-pdf") async def render_generated_cv_pdf(gen_id: str): """Render an existing generated CV to PDF using its stored template + data.""" row = db.query(""" SELECT gc.*, c.first_name, c.last_name FROM generated_cvs gc LEFT JOIN candidates c ON gc.candidate_id = c.id WHERE gc.id = %s """, (gen_id,), fetch='one') if not row: raise HTTPException(404, "Generated CV not found") # Get the template if one was used template_schema = {"canvas": {"width": 794, "height": 1123, "columns": 12}, "blocks": []} if row.get("template_id"): tmpl = db.query("SELECT * FROM cv_templates WHERE id = %s", (str(row["template_id"]),), fetch='one') if tmpl: tstruct = tmpl["template_structure"] if isinstance(tstruct, str): tstruct = json.loads(tstruct) template_schema = tstruct # If no blocks in template, build a default layout if not template_schema.get("blocks"): template_schema["blocks"] = [ {"blockId": "header", "type": "PersonalDetails", "title": "Header", "x": 0, "y": 0, "w": 12, "h": 8}, {"blockId": "summary", "type": "ProfessionalSummary", "title": "Summary", "x": 0, "y": 8, "w": 12, "h": 5}, {"blockId": "experience", "type": "WorkExperience", "title": "Experience", "x": 0, "y": 13, "w": 8, "h": 40, "config": {"showAchievements": True}}, {"blockId": "skills", "type": "SkillsList", "title": "Skills", "x": 8, "y": 13, "w": 4, "h": 25, "config": {"groupByCategory": True, "showYears": True, "sidebar": True}}, {"blockId": "education", "type": "Education", "title": "Education", "x": 8, "y": 38, "w": 4, "h": 15}, {"blockId": "footer", "type": "Footer", "title": "Footer", "x": 0, "y": 105, "w": 12, "h": 3, "config": {"content": f"Generated {row['generation_date']}"}} ] # Get full candidate data full_data = await get_candidate(str(row["candidate_id"])) # Call the render logic directly return await _do_render_pdf(template_schema, full_data, str(row["generation_date"])) # ============================================================ # CV BATCHES # ============================================================ @app.get("/api/batches") async def list_batches(): """List all CV batches with item/export summary chips.""" rows = db.query(""" SELECT b.*, COALESCE(stats.total_items, 0) AS total_items, COALESCE(stats.approved_items, 0) AS approved_items, COALESCE(stats.proposed_items, 0) AS proposed_items, COALESCE(stats.removed_items, 0) AS removed_items, COALESCE(stats.generated_items, 0) AS generated_items FROM cv_batches b LEFT JOIN ( SELECT batch_id, COUNT(*) AS total_items, COUNT(*) FILTER (WHERE status = 'approved') AS approved_items, COUNT(*) FILTER (WHERE status = 'proposed') AS proposed_items, COUNT(*) FILTER (WHERE status = 'removed') AS removed_items, COUNT(*) FILTER ( WHERE generated_cv_path IS NOT NULL AND generated_cv_path <> '' AND status = 'approved' ) AS generated_items FROM batch_items GROUP BY batch_id ) stats ON stats.batch_id = b.id ORDER BY b.created_at DESC """) return {"batches": [dict(r) for r in rows]} @app.post("/api/batches") async def create_batch(request: Request): """Create a new batch manually (name + description only).""" body = await request.json() name = body.get("name", "").strip() if not name: raise HTTPException(400, "Batch name is required") row = db.execute( "INSERT INTO cv_batches (name, description) VALUES (%s, %s) RETURNING *", (name, body.get("description", "")) ) return dict(row) @app.post("/api/batches/upload") async def upload_batch_document(file: UploadFile = File(...)): """Upload a requirements document, extract positions with AI, create a batch.""" # Extract text from the document content = await file.read() import tempfile with tempfile.NamedTemporaryFile(delete=False, suffix=f"_{file.filename}") as tmp: tmp.write(content) tmp_path = tmp.name try: doc_text = extract_text(tmp_path) finally: os.unlink(tmp_path) if not doc_text or len(doc_text.strip()) < 50: raise HTTPException(400, "Could not extract enough text from the document") # AI extracts requirements try: result = ai.extract_requirements(doc_text) except Exception as e: # If AI fails, create the batch with just the raw text result = {"batch_name": file.filename.replace(".pdf", "").replace(".docx", ""), "description": "AI extraction failed — edit manually", "positions": []} batch_name = result.get("batch_name", file.filename) description = result.get("description", "") positions = result.get("positions", []) row = db.execute( "INSERT INTO cv_batches (name, description, requirements_text, positions) VALUES (%s, %s, %s, %s) RETURNING *", (batch_name, description, doc_text, Json(positions)) ) return dict(row) @app.get("/api/batches/{batch_id}") async def get_batch(batch_id: str): """Get a single batch with its items.""" batch = db.query("SELECT * FROM cv_batches WHERE id = %s", (batch_id,), fetch='one') if not batch: raise HTTPException(404, "Batch not found") items = db.query(""" SELECT bi.*, c.first_name, c.last_name, c.email FROM batch_items bi JOIN candidates c ON bi.candidate_id = c.id WHERE bi.batch_id = %s ORDER BY bi.position_title, bi.match_score DESC """, (batch_id,)) chat = db.query("SELECT * FROM batch_chat WHERE batch_id = %s ORDER BY created_at ASC", (batch_id,)) return { "batch": dict(batch), "items": [dict(r) for r in items], "chat": [dict(r) for r in chat] } @app.put("/api/batches/{batch_id}") async def update_batch(batch_id: str, request: Request): """Update batch name/description/template.""" body = await request.json() row = db.execute( """UPDATE cv_batches SET name = %s, description = %s, template_id = %s, updated_at = NOW() WHERE id = %s RETURNING *""", (body.get("name"), body.get("description"), body.get("template_id"), batch_id) ) if not row: raise HTTPException(404, "Batch not found") return dict(row) @app.delete("/api/batches/{batch_id}") async def delete_batch(batch_id: str): """Delete a batch (cascades to items and chat).""" db.execute("DELETE FROM cv_batches WHERE id = %s", (batch_id,)) return {"success": True} @app.post("/api/batches/{batch_id}/analyze") async def analyze_batch(batch_id: str): """Run AI matching against all candidates for each position in the batch.""" batch = db.query("SELECT * FROM cv_batches WHERE id = %s", (batch_id,), fetch='one') if not batch: raise HTTPException(404, "Batch not found") positions = batch["positions"] or [] if not positions: raise HTTPException(400, "No positions defined in this batch") # Fetch all candidates with their full data candidates_raw = db.query("SELECT * FROM candidates ORDER BY created_at DESC") candidates = [] for c in candidates_raw: c_dict = dict(c) c_dict["skills"] = [dict(s) for s in db.query("SELECT * FROM skills WHERE candidate_id = %s", (c["id"],))] c_dict["experience"] = [dict(e) for e in db.query("SELECT * FROM experience WHERE candidate_id = %s", (c["id"],))] c_dict["education"] = [dict(e) for e in db.query("SELECT * FROM education WHERE candidate_id = %s", (c["id"],))] c_dict["certifications"] = [dict(cert) for cert in db.query("SELECT * FROM certifications WHERE candidate_id = %s", (c["id"],))] candidates.append(c_dict) if not candidates: raise HTTPException(400, "No candidates in the database to match against") # Clear existing proposed items for this batch db.execute("DELETE FROM batch_items WHERE batch_id = %s AND status = 'proposed'", (batch_id,)) total_matched = 0 for position in positions: try: matches = ai.match_candidates_for_position(position, candidates) except Exception as e: print(f"Match error for position {position.get('job_title', '?')}: {e}") continue for match in matches: candidate_id = match.get("candidate_id") if not candidate_id: continue # Verify candidate exists exists = db.query("SELECT 1 FROM candidates WHERE id = %s", (candidate_id,), fetch='one') if not exists: continue db.execute( """INSERT INTO batch_items (batch_id, candidate_id, position_title, match_score, match_reasoning, realigned_cv_data, status) VALUES (%s, %s, %s, %s, %s, %s, 'proposed')""", (batch_id, candidate_id, position.get("job_title", ""), match.get("match_score", 0), match.get("reasoning", ""), Json({"realignment_suggestion": match.get("realignment_suggestion", "")})) ) total_matched += 1 # Update batch status db.execute("UPDATE cv_batches SET status = 'active', updated_at = NOW() WHERE id = %s", (batch_id,)) return {"success": True, "matched": total_matched} @app.put("/api/batches/{batch_id}/items/{item_id}") async def update_batch_item(batch_id: str, item_id: str, request: Request): """Update a batch item (approve, remove, edit realignment).""" body = await request.json() status = body.get("status", "proposed") realigned = body.get("realigned_cv_data") # If removing a candidate, clear any stale generated PDF path so exports/ZIPs stay clean. if status == "removed": existing = db.query( "SELECT generated_cv_path FROM batch_items WHERE id = %s AND batch_id = %s", (item_id, batch_id), fetch="one", ) if existing and existing.get("generated_cv_path"): rel = str(existing["generated_cv_path"]).lstrip("/") if rel.startswith("static/"): file_path = static_dir / rel[len("static/"):] else: file_path = static_dir / "generated" / Path(rel).name try: if file_path.exists() and file_path.is_file(): # Only delete files that belong to this item id tag when possible. if str(item_id)[:8] in file_path.name: file_path.unlink() except Exception: pass row = db.execute( """UPDATE batch_items SET status = %s, realigned_cv_data = %s, generated_cv_path = NULL, updated_at = NOW() WHERE id = %s AND batch_id = %s RETURNING *""", (status, Json(realigned), item_id, batch_id), ) else: row = db.execute( """UPDATE batch_items SET status = %s, realigned_cv_data = %s, updated_at = NOW() WHERE id = %s AND batch_id = %s RETURNING *""", (status, Json(realigned), item_id, batch_id), ) if not row: raise HTTPException(404, "Batch item not found") return dict(row) @app.delete("/api/batches/{batch_id}/items/{item_id}") async def delete_batch_item(batch_id: str, item_id: str): """Delete a batch item.""" db.execute("DELETE FROM batch_items WHERE id = %s AND batch_id = %s", (item_id, batch_id)) return {"success": True} @app.post("/api/batches/{batch_id}/chat") async def batch_chat_endpoint(batch_id: str, request: Request): """Send a message to the batch chat and get AI response.""" body = await request.json() user_message = body.get("message", "").strip() if not user_message: raise HTTPException(400, "Message is required") batch = db.query("SELECT * FROM cv_batches WHERE id = %s", (batch_id,), fetch='one') if not batch: raise HTTPException(404, "Batch not found") # Save user message db.execute( "INSERT INTO batch_chat (batch_id, role, content) VALUES (%s, 'user', %s)", (batch_id, user_message) ) # Build batch context for AI items = db.query(""" SELECT bi.*, c.first_name, c.last_name FROM batch_items bi JOIN candidates c ON bi.candidate_id = c.id WHERE bi.batch_id = %s AND bi.status != 'removed' ORDER BY bi.position_title, bi.match_score DESC """, (batch_id,)) context_parts = [ f"Batch: {batch['name']}", f"Description: {batch.get('description', '')}", f"Positions: {json.dumps(batch.get('positions', []), indent=2)}", f"\nMatched Candidates:", ] for item in items: context_parts.append( f" - {item['first_name']} {item['last_name']} → {item['position_title']} " f"(score: {item['match_score']}, status: {item['status']})" ) batch_context = "\n".join(context_parts) # Get chat history history = db.query("SELECT role, content FROM batch_chat WHERE batch_id = %s ORDER BY created_at ASC", (batch_id,)) history_list = [dict(r) for r in history] # Get AI response try: ai_response = ai.batch_chat(user_message, batch_context, history_list) except Exception as e: ai_response = f"Sorry, I couldn't process that: {str(e)}" # Save AI response db.execute( "INSERT INTO batch_chat (batch_id, role, content) VALUES (%s, 'assistant', %s)", (batch_id, ai_response) ) return {"response": ai_response} @app.put("/api/batches/{batch_id}/positions") async def update_batch_positions(batch_id: str, request: Request): """Update the positions array on a batch (add/edit/remove positions).""" body = await request.json() positions = body.get("positions", []) title_renames = body.get("title_renames") or {} # {old_title: new_title} batch = db.query("SELECT * FROM cv_batches WHERE id = %s", (batch_id,), fetch='one') if not batch: raise HTTPException(404, "Batch not found") row = db.execute( "UPDATE cv_batches SET positions = %s, updated_at = NOW() WHERE id = %s RETURNING *", (Json(positions), batch_id) ) if not row: raise HTTPException(404, "Batch not found") # When a position title is renamed, remap linked batch items. if isinstance(title_renames, dict): for old_title, new_title in title_renames.items(): if not old_title or not new_title or old_title == new_title: continue db.execute( """UPDATE batch_items SET position_title = %s, updated_at = NOW() WHERE batch_id = %s AND position_title = %s""", (new_title, batch_id, old_title) ) return dict(row) @app.post("/api/batches/{batch_id}/generate") async def generate_batch_cvs(batch_id: str): """Generate CVs for all approved items in the batch using the selected template.""" import httpx from datetime import date as _date, datetime as _datetime from decimal import Decimal from uuid import UUID def _jsonable(value): """Recursively convert DB row values into JSON-safe types.""" if isinstance(value, dict): return {k: _jsonable(v) for k, v in value.items()} if isinstance(value, (list, tuple)): return [_jsonable(v) for v in value] if isinstance(value, (_datetime, _date)): return value.isoformat() if isinstance(value, UUID): return str(value) if isinstance(value, Decimal): return float(value) if isinstance(value, (bytes, bytearray)): return value.decode("utf-8", errors="replace") return value batch = db.query("SELECT * FROM cv_batches WHERE id = %s", (batch_id,), fetch='one') if not batch: raise HTTPException(404, "Batch not found") template_id = batch.get("template_id") if not template_id: raise HTTPException(400, "No template selected for this batch") # Get approved items with candidate data items = db.query(""" SELECT bi.*, c.first_name, c.last_name, c.email, c.phone, c.address, c.linkedin, c.github, c.website, c.summary, c.raw_cv_text FROM batch_items bi JOIN candidates c ON bi.candidate_id = c.id WHERE bi.batch_id = %s AND bi.status = 'approved' ORDER BY bi.position_title, bi.match_score DESC """, (batch_id,)) if not items: raise HTTPException(400, "No approved candidates to generate CVs for") # For each approved item, fetch full candidate data and call Carbone render results = [] carbone_url = f"http://localhost:8771/api/carbone/templates/{template_id}/render" for item in items: candidate_id = str(item["candidate_id"]) # Fetch full candidate data (skills, experience, education, certs) cand = dict(db.query("SELECT * FROM candidates WHERE id = %s", (candidate_id,), fetch='one')) cand["skills"] = [dict(s) for s in db.query("SELECT * FROM skills WHERE candidate_id = %s", (candidate_id,))] cand["experience"] = [dict(e) for e in db.query("SELECT * FROM experience WHERE candidate_id = %s", (candidate_id,))] cand["education"] = [dict(e) for e in db.query("SELECT * FROM education WHERE candidate_id = %s", (candidate_id,))] cand["certifications"] = [dict(c) for c in db.query("SELECT * FROM certifications WHERE candidate_id = %s", (candidate_id,))] cand = _jsonable(cand) # Use realigned data if available, else original realigned = item.get("realigned_cv_data") if realigned and isinstance(realigned, dict) and realigned.get("realignment_suggestion"): # For now, use original data — realignment is advisory only pass try: async with httpx.AsyncClient(timeout=120) as client: resp = await client.post(carbone_url, json={"candidateData": cand}) if resp.status_code == 200: # Save the PDF pdf_dir = os.path.join(str(static_dir), "generated") os.makedirs(pdf_dir, exist_ok=True) safe_name = f"{item['first_name']}_{item['last_name']}".replace(" ", "_") safe_batch = str(batch['name']).replace(" ", "_").replace("/", "-") safe_pos = str(item.get("position_title") or "role").replace(" ", "_").replace("/", "-")[:40] item_tag = str(item["id"])[:8] filename = f"{safe_batch}_{safe_pos}_{safe_name}_{item_tag}.pdf" filepath = os.path.join(pdf_dir, filename) with open(filepath, "wb") as f: f.write(resp.content) # Update batch item with path db.execute( "UPDATE batch_items SET generated_cv_path = %s, updated_at = NOW() WHERE id = %s", (f"/static/generated/{filename}", item["id"]) ) results.append({ "candidate": f"{item['first_name']} {item['last_name']}", "position_title": item.get("position_title"), "status": "ok", "file": filename, "path": f"/static/generated/{filename}", }) else: results.append({"candidate": f"{item['first_name']} {item['last_name']}", "status": "error", "error": resp.text[:200]}) except Exception as e: results.append({"candidate": f"{item['first_name']} {item['last_name']}", "status": "error", "error": str(e)}) ok_count = len([r for r in results if r["status"] == "ok"]) # Only mark exported if at least one CV succeeded if ok_count: db.execute("UPDATE cv_batches SET status = 'exported', updated_at = NOW() WHERE id = %s", (batch_id,)) return {"success": True, "generated": ok_count, "results": results} @app.get("/api/batches/{batch_id}/download-zip") async def download_batch_zip(batch_id: str): """Zip all generated PDFs for a batch into one download.""" import io import zipfile from urllib.parse import unquote batch = db.query("SELECT * FROM cv_batches WHERE id = %s", (batch_id,), fetch='one') if not batch: raise HTTPException(404, "Batch not found") items = db.query(""" SELECT bi.generated_cv_path, bi.position_title, c.first_name, c.last_name FROM batch_items bi JOIN candidates c ON c.id = bi.candidate_id WHERE bi.batch_id = %s AND bi.status = 'approved' AND bi.generated_cv_path IS NOT NULL AND bi.generated_cv_path <> '' ORDER BY bi.position_title, c.last_name, c.first_name """, (batch_id,)) if not items: raise HTTPException(400, "No generated CVs available to zip") buf = io.BytesIO() added = 0 with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf: for item in items: rel = unquote(str(item["generated_cv_path"] or "")).lstrip("/") # Paths are stored as /static/generated/ if rel.startswith("static/"): file_path = static_dir / rel[len("static/"):] else: file_path = Path(rel) if not file_path.exists(): # Also try direct under static/generated name_only = Path(rel).name file_path = static_dir / "generated" / name_only if not file_path.exists(): continue safe_person = f"{item.get('first_name') or ''}_{item.get('last_name') or ''}".strip("_").replace(" ", "_") or "candidate" safe_pos = (item.get("position_title") or "role").replace(" ", "_") arcname = f"{safe_pos}/{safe_person}_{file_path.name}" zf.write(str(file_path), arcname=arcname) added += 1 if added == 0: raise HTTPException(404, "Generated CV files not found on disk") buf.seek(0) safe_batch = str(batch.get("name") or "batch").replace(" ", "_") return Response( content=buf.getvalue(), media_type="application/zip", headers={"Content-Disposition": f'attachment; filename="{safe_batch}_CVs.zip"'} ) @app.delete("/api/batches/{batch_id}/items/{item_id}/generated") async def clear_batch_item_generated(batch_id: str, item_id: str): """Clear a generated CV path (optional file cleanup for batch exports).""" item = db.query( "SELECT * FROM batch_items WHERE id = %s AND batch_id = %s", (item_id, batch_id), fetch='one' ) if not item: raise HTTPException(404, "Batch item not found") path = item.get("generated_cv_path") if path: rel = str(path).lstrip("/") if rel.startswith("static/"): file_path = static_dir / rel[len("static/"):] else: file_path = static_dir / "generated" / Path(rel).name try: if file_path.exists() and file_path.is_file(): file_path.unlink() except Exception: pass row = db.execute( """UPDATE batch_items SET generated_cv_path = NULL, updated_at = NOW() WHERE id = %s AND batch_id = %s RETURNING *""", (item_id, batch_id) ) return dict(row) if row else {"success": True} # ============================================================ # SERVE FRONTEND # ============================================================ @app.get("/") async def index(): return FileResponse(str(static_dir / "index.html")) @app.get("/{path:path}") async def serve_static(path: str): """Serve static files or return index.html for SPA routes.""" file_path = static_dir / path if file_path.exists() and file_path.is_file(): response = FileResponse(str(file_path)) # Prevent caching of JS/CSS so code changes take effect without hard refresh if path.endswith(('.js', '.css')): response.headers['Cache-Control'] = 'no-cache, no-store, must-revalidate' return response return FileResponse(str(static_dir / "index.html"))