feat: CV Application - AI-powered CV management, parsing, matching, and generation

This commit is contained in:
root
2026-07-15 14:23:23 +00:00
commit 76f0d9efba
5691 changed files with 699219 additions and 0 deletions

61
doc_parser.py Normal file
View File

@@ -0,0 +1,61 @@
"""Document text extraction utilities (PDF, DOCX, TXT, etc.)."""
import os
from PyPDF2 import PdfReader
from docx import Document
def extract_text_from_pdf(file_path: str) -> str:
"""Extract text from a PDF file."""
reader = PdfReader(file_path)
text_parts = []
for page in reader.pages:
text = page.extract_text()
if text:
text_parts.append(text)
return "\n\n".join(text_parts)
def extract_text_from_docx(file_path: str) -> str:
"""Extract text from a DOCX file."""
doc = Document(file_path)
text_parts = []
for para in doc.paragraphs:
if para.text.strip():
text_parts.append(para.text)
# Also extract tables
for table in doc.tables:
for row in table.rows:
row_text = " | ".join(cell.text.strip() for cell in row.cells)
if row_text.strip():
text_parts.append(row_text)
return "\n".join(text_parts)
def extract_text_from_txt(file_path: str) -> str:
"""Read a plain text file."""
with open(file_path, "r", encoding="utf-8", errors="replace") as f:
return f.read()
def extract_text(file_path: str) -> str:
"""Extract text from a file based on its extension."""
ext = os.path.splitext(file_path)[1].lower()
if ext == ".pdf":
return extract_text_from_pdf(file_path)
elif ext == ".docx":
return extract_text_from_docx(file_path)
elif ext in (".txt", ".md", ".rtf"):
return extract_text_from_txt(file_path)
elif ext == ".doc":
# Old .doc format — try reading as text (best effort)
try:
return extract_text_from_txt(file_path)
except Exception:
return f"[Could not extract text from .doc file: {file_path}]"
else:
# Try reading as text
try:
return extract_text_from_txt(file_path)
except Exception:
return f"[Unsupported file format: {ext}]"