Spaces:
Sleeping
Sleeping
| from fastapi import FastAPI, HTTPException | |
| from pydantic import BaseModel | |
| from typing import List, Dict, Optional, Union | |
| import pandas as pd | |
| import numpy as np | |
| import json | |
| import logging | |
| from sentence_transformers import SentenceTransformer | |
| from sklearn.metrics.pairwise import cosine_similarity | |
| from fastapi.middleware.cors import CORSMiddleware | |
| from langchain_google_genai import ChatGoogleGenerativeAI | |
| from langchain.schema import HumanMessage, SystemMessage | |
| import re | |
| from collections import defaultdict | |
| import os | |
| import asyncio | |
| from urllib.parse import urlparse | |
| import time | |
| from functools import lru_cache | |
| import gradio as gr | |
| from datetime import datetime, timedelta | |
| import hashlib | |
| import urllib.request | |
| import urllib.error | |
| from html.parser import HTMLParser | |
| import re as _re_for_url_extract | |
| app = FastAPI( | |
| title="Advanced Job Skill & Certification Matcher", | |
| description="Comprehensive certification analysis with reliability scoring, skill matching, and career path recommendations", | |
| version="10.0" | |
| ) | |
| logger = logging.getLogger("uvicorn") | |
| logging.basicConfig(level=logging.INFO) | |
| # Configuration - Using only skil.xlsx file | |
| SKILL_FILE = "skil.xlsx" | |
| GOOGLE_API_KEY = "AIzaSyBZNWhMXa9SG0WDKbK5uhLc5ewxFmOyH_Y" | |
| # Reliability scoring factors | |
| CERT_RELIABILITY_FACTORS = { | |
| "provider_reputation": { | |
| "weight": 0.25, | |
| "providers": { | |
| "microsoft": 0.9, "aws": 0.95, "google": 0.9, "cisco": 0.85, | |
| "comptia": 0.8, "pmi": 0.85, "isc2": 0.9, "oracle": 0.85, | |
| "ibm": 0.8, "salesforce": 0.85, "apple": 0.8, "linux": 0.8 | |
| } | |
| }, | |
| "industry_demand": {"weight": 0.2}, | |
| "exam_rigor": { | |
| "weight": 0.15, | |
| "indicators": ["proctored", "practical", "hands-on", "lab", "performance-based"] | |
| }, | |
| "validity_period": { | |
| "weight": 0.1, | |
| "scale": {"lifetime": 1.0, "3_years": 0.8, "2_years": 0.7, "1_year": 0.5} | |
| }, | |
| "renewal_requirements": { | |
| "weight": 0.1, | |
| "scale": {"none": 0.3, "exam": 0.8, "continuing_education": 0.7, "both": 0.9} | |
| }, | |
| "market_recognition": {"weight": 0.2} | |
| } | |
| # Models | |
| class InputText(BaseModel): | |
| text: str | |
| min_relevance: float = 0.4 | |
| max_internal_results: int = 20 | |
| max_external_results: int = 10 | |
| analyze: bool = True | |
| class CertificateURL(BaseModel): | |
| url: str | |
| extract_skills: bool = True | |
| analyze_reliability: bool = True | |
| class CertificationComparison(BaseModel): | |
| cert_names: List[str] | |
| compare_by: str = "reliability" | |
| class LearningPathRequest(BaseModel): | |
| target_role: str | |
| current_skills: List[str] | |
| timeframe: str = "6_months" | |
| budget: str = "moderate" | |
| class SkillGapAnalysis(BaseModel): | |
| current_skills: List[str] | |
| target_skills: List[str] | |
| current_role: Optional[str] = None | |
| target_role: Optional[str] = None | |
| class CertificationTracker(BaseModel): | |
| cert_name: str | |
| achieved_date: str | |
| expiration_date: Optional[str] = None | |
| renewal_requirements: Optional[List[str]] = None | |
| class AnalyzeCertificateURL(BaseModel): | |
| url: str | |
| min_relevance: float = 0.4 | |
| max_internal_results: int = 20 | |
| max_external_results: int = 10 | |
| analyze: bool = True | |
| # New model for job ontology extraction | |
| class JobOntologyRequest(BaseModel): | |
| job_description: str | |
| include_certificates: bool = True | |
| min_relevance: float = 0.4 | |
| # Initialize LLM | |
| llm = ChatGoogleGenerativeAI( | |
| model="gemini-2.0-flash", | |
| temperature=0.2, | |
| google_api_key=GOOGLE_API_KEY | |
| ) | |
| # Global cached models and throttling | |
| _embedding_model_cached = None | |
| LLM_MAX_CONCURRENCY = 2 | |
| _llm_semaphore = asyncio.Semaphore(LLM_MAX_CONCURRENCY) | |
| def get_embedding_model(): | |
| """Return a singleton embedding model instance.""" | |
| global _embedding_model_cached | |
| if _embedding_model_cached is None: | |
| _embedding_model_cached = SentenceTransformer('BAAI/bge-small-en-v1.5') | |
| return _embedding_model_cached | |
| async def ainvoke_llm(messages, retries: int = 2, delay: float = 0.8): | |
| """LLM call with concurrency guard and simple retries.""" | |
| attempt = 0 | |
| while True: | |
| try: | |
| async with _llm_semaphore: | |
| return await llm.ainvoke(messages) | |
| except Exception as e: | |
| attempt += 1 | |
| if attempt > retries: | |
| raise | |
| await asyncio.sleep(delay * attempt) | |
| # Global variables | |
| certs_df = None | |
| cert_embeddings = None | |
| skills_df = None | |
| skill_embeddings = None | |
| id_col = None | |
| skill_col = None | |
| # -------------------------- | |
| # ENHANCED FUNCTIONS | |
| # -------------------------- | |
| # Utilities to cope with duplicate column names and non-scalar cells | |
| def get_column_series(df: pd.DataFrame, column_name: str) -> Optional[pd.Series]: | |
| """Return a single Series for a column name even if duplicates exist.""" | |
| if df is None or column_name not in df.columns: | |
| return None | |
| col = df[column_name] | |
| if isinstance(col, pd.DataFrame): | |
| # Pick the first occurrence | |
| return col.iloc[:, 0] | |
| return col | |
| def get_row_scalar(row: pd.Series, key: str) -> str: | |
| """Safely extract a scalar string from a row that may have duplicate column names.""" | |
| if row is None: | |
| return "" | |
| try: | |
| value = row.get(key, "") | |
| except Exception: | |
| value = "" | |
| if isinstance(value, pd.Series): | |
| # Choose first non-empty string among duplicates | |
| for v in value.values: | |
| s = str(v).strip() | |
| if s and s.lower() != 'nan': | |
| return s | |
| return "" | |
| return str(value) | |
| # Robust getter to fetch a field using multiple candidate column names | |
| def get_field_with_fallback(row: pd.Series, candidate_keys: List[str]) -> str: | |
| """Try multiple column names and return the first non-empty scalar string.""" | |
| for key in candidate_keys: | |
| if key in row.index: | |
| val = get_row_scalar(row, key) | |
| s = str(val).strip() | |
| if s and s.lower() != 'nan' and s.lower() != 'none': | |
| return s | |
| return "" | |
| async def calculate_certificate_reliability(cert_data: Dict) -> Dict: | |
| """Calculate comprehensive reliability score for a certification""" | |
| scores = {} | |
| total_score = 0.0 | |
| # 1. Provider reputation scoring | |
| provider = str(cert_data.get("provider", "")).lower() | |
| provider_score = 0.5 # Default for unknown providers | |
| for known_provider, score in CERT_RELIABILITY_FACTORS["provider_reputation"]["providers"].items(): | |
| if known_provider in provider: | |
| provider_score = score | |
| break | |
| provider_weight = CERT_RELIABILITY_FACTORS["provider_reputation"]["weight"] | |
| scores["provider_reputation"] = provider_score | |
| total_score += provider_score * provider_weight | |
| # 2. Exam rigor scoring | |
| description = str(cert_data.get("description", "")).lower() | |
| exam_rigor_score = 0.3 # Default | |
| for indicator in CERT_RELIABILITY_FACTORS["exam_rigor"]["indicators"]: | |
| if indicator in description: | |
| exam_rigor_score = min(1.0, exam_rigor_score + 0.2) | |
| exam_rigor_weight = CERT_RELIABILITY_FACTORS["exam_rigor"]["weight"] | |
| scores["exam_rigor"] = exam_rigor_score | |
| total_score += exam_rigor_score * exam_rigor_weight | |
| # 3. Validity period scoring (if available) | |
| validity_score = 0.5 | |
| validity_weight = CERT_RELIABILITY_FACTORS["validity_period"]["weight"] | |
| scores["validity_period"] = validity_score | |
| total_score += validity_score * validity_weight | |
| # 4. Renewal requirements scoring (if available) | |
| renewal_score = 0.5 | |
| renewal_weight = CERT_RELIABILITY_FACTORS["renewal_requirements"]["weight"] | |
| scores["renewal_requirements"] = renewal_score | |
| total_score += renewal_score * renewal_weight | |
| # 5. Industry demand (would require external data source) | |
| demand_score = 0.7 # Placeholder | |
| demand_weight = CERT_RELIABILITY_FACTORS["industry_demand"]["weight"] | |
| scores["industry_demand"] = demand_score | |
| total_score += demand_score * demand_weight | |
| # 6. Market recognition (would require external data source) | |
| recognition_score = 0.7 # Placeholder | |
| recognition_weight = CERT_RELIABILITY_FACTORS["market_recognition"]["weight"] | |
| scores["market_recognition"] = recognition_score | |
| total_score += recognition_score * recognition_weight | |
| return { | |
| "overall_score": round(total_score, 2), | |
| "component_scores": scores, | |
| "confidence": "medium" | |
| } | |
| class _TextExtractor(HTMLParser): | |
| """Simple HTML text extractor using stdlib only.""" | |
| def __init__(self): | |
| super().__init__() | |
| self._texts = [] | |
| self._in_script_style = False | |
| def handle_starttag(self, tag, attrs): | |
| if tag in ("script", "style", "noscript"): | |
| self._in_script_style = True | |
| def handle_endtag(self, tag): | |
| if tag in ("script", "style", "noscript"): | |
| self._in_script_style = False | |
| def handle_data(self, data): | |
| if not self._in_script_style: | |
| text = data.strip() | |
| if text: | |
| self._texts.append(text) | |
| def get_text(self) -> str: | |
| return " ".join(self._texts) | |
| def fetch_url_content(url: str, timeout: int = 12) -> Dict: | |
| """Fetch URL content using stdlib with a browser-like User-Agent. | |
| Returns: {"html": str, "text": str, "title": str, "meta": Dict} | |
| """ | |
| try: | |
| req = urllib.request.Request( | |
| url, | |
| headers={ | |
| "User-Agent": ( | |
| "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " | |
| "(KHTML, like Gecko) Chrome/124.0 Safari/537.36" | |
| ) | |
| }, | |
| ) | |
| with urllib.request.urlopen(req, timeout=timeout) as resp: | |
| raw = resp.read() | |
| # Try to decode; fall back to utf-8 with errors ignored | |
| try: | |
| charset = resp.headers.get_content_charset() or "utf-8" | |
| except Exception: | |
| charset = "utf-8" | |
| html = raw.decode(charset, errors="ignore") | |
| # Extract title and simple meta tags | |
| title = "" | |
| meta: Dict[str, str] = {} | |
| # crude title extraction | |
| lower = html.lower() | |
| start = lower.find("<title>") | |
| end = lower.find("</title>") | |
| if start != -1 and end != -1 and end > start: | |
| title = html[start + 7:end].strip() | |
| # meta description | |
| for needle in ( | |
| 'name="description"', | |
| "name='description'", | |
| 'property="og:description"', | |
| "property='og:description'", | |
| ): | |
| idx = lower.find("<meta ") | |
| search_from = 0 | |
| while idx != -1: | |
| close = lower.find(">", idx) | |
| if close == -1: | |
| break | |
| chunk = html[idx:close + 1] | |
| if needle in chunk.lower(): | |
| # extract content="..." | |
| c_start = chunk.lower().find('content="') | |
| if c_start != -1: | |
| c_start += len('content="') | |
| c_end = chunk.find('"', c_start) | |
| if c_end != -1: | |
| meta["description"] = chunk[c_start:c_end].strip() | |
| break | |
| c_start = chunk.lower().find("content='") | |
| if c_start != -1: | |
| c_start += len("content='") | |
| c_end = chunk.find("'", c_start) | |
| if c_end != -1: | |
| meta["description"] = chunk[c_start:c_end].strip() | |
| break | |
| search_from = close + 1 | |
| idx = lower.find("<meta ", search_from) | |
| if "description" in meta: | |
| break | |
| # Extract plain text | |
| parser = _TextExtractor() | |
| try: | |
| parser.feed(html) | |
| except Exception: | |
| pass | |
| text = parser.get_text() | |
| return {"html": html, "text": text, "title": title, "meta": meta} | |
| except (urllib.error.URLError, urllib.error.HTTPError, ValueError) as e: | |
| logger.warning(f"Failed to fetch URL '{url}': {e}") | |
| return {"html": "", "text": "", "title": "", "meta": {}} | |
| async def process_certificate_url(url: str) -> Dict: | |
| """Extract information from a certificate URL""" | |
| try: | |
| parsed_url = urlparse(url) | |
| domain = parsed_url.netloc.lower() | |
| # Check if URL is from our internal database | |
| internal_match = None | |
| if certs_df is not None and 'hyperlink' in certs_df.columns: | |
| try: | |
| hyperlink_series = get_column_series(certs_df, 'hyperlink') | |
| if hyperlink_series is not None: | |
| match_mask = hyperlink_series.astype(str).str.contains(parsed_url.path, na=False, case=False, regex=False) | |
| match = certs_df[match_mask] | |
| if not match.empty: | |
| internal_match = match.iloc[0].to_dict() | |
| except Exception as _: | |
| internal_match = None | |
| # Try to fetch page content up-front for external URLs | |
| fetched = fetch_url_content(url) | |
| fetched_summary = { | |
| "title": fetched.get("title", ""), | |
| "description": fetched.get("meta", {}).get("description", ""), | |
| "text_preview": (fetched.get("text", "")[:1200] + "...") if len(fetched.get("text", "")) > 1200 else fetched.get("text", ""), | |
| "domain": domain, | |
| } | |
| # Use Gemini to extract information from the page if not found internally | |
| if internal_match is None: | |
| prompt = f"""Analyze this certification page and extract information. | |
| URL: {url} | |
| Page Title: {fetched_summary.get('title','')} | |
| Meta Description: {fetched_summary.get('description','')} | |
| Text Preview (truncated): {fetched_summary.get('text_preview','')} | |
| Extract the following information: | |
| - name: Certification name | |
| - provider: Organization offering the certification | |
| - description: Brief description of the certification | |
| - skills: List of skills covered (comma-separated or array) | |
| - requirements: Any prerequisites or requirements | |
| - exam_details: Information about the exam format | |
| - validity: How long the certification is valid | |
| - renewal: Renewal policy if available | |
| Return strict JSON only. | |
| """ | |
| messages = [SystemMessage(content="Return valid JSON with certification information."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| cert_info: Dict = {"name": "Unknown", "provider": domain} | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| json_str = response_text[start_idx:end_idx] | |
| try: | |
| cert_info = json.loads(json_str) | |
| except Exception: | |
| pass | |
| # Fallbacks using fetched content | |
| if not cert_info.get("name"): | |
| cert_info["name"] = fetched_summary.get("title") or domain | |
| if not cert_info.get("description"): | |
| desc = fetched_summary.get("description") or fetched_summary.get("text_preview", "")[:280] | |
| cert_info["description"] = desc | |
| if not cert_info.get("provider"): | |
| cert_info["provider"] = domain | |
| else: | |
| cert_info = internal_match | |
| # Add URL and source information | |
| cert_info["url"] = url | |
| cert_info["source"] = "internal" if internal_match else "external" | |
| # Normalize skills field to list of strings | |
| if "skills" in cert_info: | |
| if isinstance(cert_info["skills"], str): | |
| # split by comma or pipe | |
| if "," in cert_info["skills"]: | |
| cert_info["skills"] = [s.strip() for s in cert_info["skills"].split(",") if s.strip()] | |
| elif "|" in cert_info["skills"]: | |
| cert_info["skills"] = [s.strip() for s in cert_info["skills"].split("|") if s.strip()] | |
| else: | |
| cert_info["skills"] = [cert_info["skills"].strip()] if cert_info["skills"].strip() else [] | |
| elif isinstance(cert_info["skills"], list): | |
| cert_info["skills"] = [str(s).strip() for s in cert_info["skills"] if str(s).strip()] | |
| else: | |
| cert_info["skills"] = [] | |
| return cert_info | |
| except Exception as e: | |
| logger.error(f"URL processing failed: {e}") | |
| return {"error": str(e), "url": url} | |
| def extract_first_url_from_row(row: pd.Series) -> str: | |
| """Scan all string fields in a row and return the first http(s) URL if found.""" | |
| url_pattern = _re_for_url_extract.compile(r"https?://[^\s]+", flags=_re_for_url_extract.IGNORECASE) | |
| try: | |
| for value in row.values: | |
| s = str(value) | |
| if not s or s.lower() == 'nan': | |
| continue | |
| m = url_pattern.search(s) | |
| if m: | |
| return m.group(0) | |
| except Exception: | |
| return "" | |
| return "" | |
| async def search_internal_certificates(query: str, min_score: float, max_results: int) -> List[Dict]: | |
| """Search internal certificate database with enhanced filtering for unique skills""" | |
| try: | |
| embedding_model = get_embedding_model() | |
| query_embedding = embedding_model.encode([query]) | |
| scores = cosine_similarity(query_embedding, cert_embeddings)[0] | |
| # Get more results for filtering | |
| top_indices = np.argsort(scores)[-max_results * 5:][::-1] | |
| results = [] | |
| used_skills = set() | |
| for idx in top_indices: | |
| if scores[idx] >= min_score: | |
| row = certs_df.iloc[idx] | |
| # Get skills from this certificate (handle possible duplicate 'skills' columns) | |
| cert_skills = set() | |
| try: | |
| raw_skills_str = "" | |
| # Try multiple possible skills columns | |
| possible_skill_cols = [ | |
| 'skills', 'skill', 'skill_list', 'competencies', 'capabilities' | |
| ] | |
| for key in possible_skill_cols: | |
| if key in row.index: | |
| val = row[key] | |
| if isinstance(val, pd.Series): | |
| parts = [str(v) for v in val.values if pd.notna(v) and str(v).strip()] | |
| if parts: | |
| raw_skills_str = '|'.join(parts) | |
| break | |
| else: | |
| s = str(val) | |
| if s and s.strip() and s.lower() != 'nan': | |
| raw_skills_str = s | |
| break | |
| if raw_skills_str: | |
| # Support both '|' and ',' separated lists | |
| tokens = [] | |
| for sep in ['|', ',']: | |
| if sep in raw_skills_str: | |
| tokens = [t.strip() for t in raw_skills_str.split(sep)] | |
| break | |
| if not tokens: | |
| tokens = [raw_skills_str.strip()] | |
| # Normalize each token to a canonical internal skill name if possible | |
| normalized = set() | |
| for tok in tokens: | |
| if not tok: | |
| continue | |
| if skills_df is not None and skill_col is not None: | |
| matched = await strict_internal_skill_match(tok) | |
| if matched and matched.get('name'): | |
| normalized.add(str(matched['name'])) | |
| else: | |
| normalized.add(tok) | |
| else: | |
| normalized.add(tok) | |
| cert_skills = {s for s in normalized if s and s.lower() != 'nan'} | |
| except Exception: | |
| # Fallback: no skills extracted | |
| cert_skills = set() | |
| # Check if this certificate adds new skills | |
| if cert_skills and not cert_skills.issubset(used_skills): | |
| # Calculate reliability score | |
| name_val = get_field_with_fallback(row, ['name', 'title']) | |
| provider_val = get_field_with_fallback(row, ['provider', 'vendor', 'organization', 'issuer']) | |
| description_val = get_field_with_fallback(row, ['description', 'desc', 'details', 'summary']) | |
| hyperlink_val = get_field_with_fallback(row, ['hyperlink', 'link', 'url', 'website']) | |
| if not hyperlink_val: | |
| # Try to parse any URL from the row values | |
| hyperlink_val = extract_first_url_from_row(row) | |
| if not hyperlink_val and name_val: | |
| # Last resort: provide a search link so user can navigate | |
| q = urllib.parse.quote_plus(f"{name_val} {provider_val} certification") | |
| hyperlink_val = f"https://www.google.com/search?q={q}" | |
| cert_data = { | |
| "name": name_val, | |
| "provider": provider_val, | |
| "description": description_val, | |
| "skills": list(cert_skills) | |
| } | |
| reliability = await calculate_certificate_reliability(cert_data) | |
| result_data = { | |
| "type": "certification", | |
| "name": name_val, | |
| "provider": provider_val, | |
| "score": float(scores[idx]), | |
| "hyperlink": hyperlink_val, | |
| "description": (lambda d: (d[:200] + "...") if len(d) > 200 else d)(description_val), | |
| "skills": list(cert_skills), | |
| "new_skills_count": len(cert_skills - used_skills), | |
| "bucket": "internal", | |
| "reliability_score": reliability | |
| } | |
| results.append(result_data) | |
| used_skills.update(cert_skills) | |
| if len(results) >= max_results: | |
| break | |
| return results | |
| except Exception as e: | |
| logger.error(f"Internal certificate search failed: {e}") | |
| return [] | |
| async def search_external_certificates(query: str, max_results: int) -> List[Dict]: | |
| """Search for external certificates using Gemini API""" | |
| try: | |
| prompt = f"""Find {max_results} relevant professional certifications for the skill: "{query}". | |
| For each certification, provide: | |
| - name: The full name of the certification | |
| - provider: The organization that offers it | |
| - description: A brief description of what it covers | |
| - skills: A list of specific skills covered (comma-separated) | |
| - url: The official URL for more information | |
| - relevance_score: A score from 0-1 indicating relevance to the query | |
| Return the results in JSON format with this structure: | |
| {{ | |
| "certifications": [ | |
| {{ | |
| "name": "Certification Name", | |
| "provider": "Provider Name", | |
| "description": "Brief description", | |
| "skills": ["skill1", "skill2", "skill3"], | |
| "url": "https://example.com", | |
| "relevance_score": 0.85 | |
| }} | |
| ] | |
| }} | |
| """ | |
| messages = [SystemMessage(content="Return valid JSON with certification information."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| # Extract JSON from response | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| json_str = response_text[start_idx:end_idx] | |
| result = json.loads(json_str) | |
| certifications = result.get("certifications", []) | |
| # Format the results | |
| formatted_results = [] | |
| used_skills = set() # Track skills to calculate new_skills_count | |
| for cert in certifications[:max_results]: | |
| # Calculate reliability score | |
| # Ensure provider/description are strings in case of Series types | |
| safe_cert = dict(cert) | |
| safe_cert['provider'] = str(safe_cert.get('provider', '')) | |
| safe_cert['description'] = str(safe_cert.get('description', '')) | |
| reliability = await calculate_certificate_reliability(safe_cert) | |
| # Get skills for this certificate | |
| cert_skills = set() | |
| if "skills" in cert and cert["skills"]: | |
| if isinstance(cert["skills"], str): | |
| cert_skills = set(s.strip() for s in cert["skills"].split(",") if s.strip()) | |
| elif isinstance(cert["skills"], list): | |
| cert_skills = set(str(s).strip() for s in cert["skills"] if str(s).strip()) | |
| # Calculate new skills count | |
| new_skills_count = len(cert_skills - used_skills) if cert_skills else 0 | |
| formatted_results.append({ | |
| "type": "certification", | |
| "name": cert.get("name", ""), | |
| "provider": cert.get("provider", ""), | |
| "score": float(cert.get("relevance_score", 0.5)), | |
| "hyperlink": cert.get("url", ""), | |
| "description": (lambda d: (d[:200] + "...") if len(d) > 200 else d)(str(cert.get("description", ""))), | |
| "skills": list(cert_skills), | |
| "new_skills_count": new_skills_count, | |
| "bucket": "external", | |
| "source": "gemini_api", | |
| "reliability_score": reliability | |
| }) | |
| # Update used skills for next iteration | |
| used_skills.update(cert_skills) | |
| return formatted_results | |
| return [] | |
| except Exception as e: | |
| logger.error(f"External certificate search failed: {e}") | |
| return [] | |
| async def get_certificate_recommendations(query: str, min_score: float, | |
| max_internal: int, max_external: int) -> Dict: | |
| """Get certificate recommendations from both internal and external sources""" | |
| # Get internal recommendations | |
| internal_certs = await search_internal_certificates(query, min_score, max_internal) | |
| # Get external recommendations | |
| external_certs = await search_external_certificates(query, max_external) | |
| # Remove duplicates between internal and external | |
| internal_names = {cert["name"].lower() for cert in internal_certs} | |
| external_certs = [cert for cert in external_certs | |
| if cert["name"].lower() not in internal_names] | |
| return { | |
| "internal_bucket": internal_certs, | |
| "external_bucket": external_certs | |
| } | |
| async def determine_input_type(text: str) -> str: | |
| """Determine if the input is a job description or a skill""" | |
| try: | |
| prompt = f"Analyze: {text}\nReturn ONLY 'job_description' or 'skill'" | |
| messages = [SystemMessage(content="Return only 'job_description' or 'skill'"), HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| return "job_description" if "job" in response.content.lower() else "skill" | |
| except Exception as e: | |
| logger.error(f"Input type determination failed: {e}") | |
| return "job_description" | |
| def skills_to_json(df: pd.DataFrame, max_items: int = 1000) -> str: | |
| """Convert skills DataFrame to JSON string""" | |
| if df.empty: | |
| return "[]" | |
| records = [] | |
| for _, row in df.iterrows(): | |
| try: | |
| rec = {"id": str(row[id_col]), "skill": str(row[skill_col])} | |
| for c in df.columns: | |
| if c not in [id_col, skill_col] and pd.notna(row[c]): | |
| rec[c] = str(row[c]) | |
| records.append(rec) | |
| if len(records) >= max_items: break | |
| except Exception as e: | |
| continue | |
| return json.dumps(records, ensure_ascii=False, indent=2) | |
| async def extract_skills_with_gemini(job_description: str) -> List[Dict]: | |
| """Extract skills using Gemini (only for job description analysis)""" | |
| try: | |
| skills_json = skills_to_json(skills_df) | |
| prompt = f"""Analyze this job description and extract relevant skills from the provided list. | |
| Available skills: {skills_json} | |
| Job Description: {job_description} | |
| Return JSON format: {{"skills": [{{"id": "id", "skill": "name", "relevance_score": 1-10}}]}} | |
| """ | |
| messages = [SystemMessage(content="Return valid JSON with skills from the provided list."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| # Extract JSON from response | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| json_str = response_text[start_idx:end_idx] | |
| result = json.loads(json_str) | |
| return [ | |
| { | |
| "type": "skill", | |
| "name": s["skill"], | |
| "id": s["id"], | |
| "score": float(s.get("relevance_score", 5)) / 10 | |
| } | |
| for s in result.get("skills", []) | |
| if "id" in s and "skill" in s | |
| ] | |
| return [] | |
| except Exception as e: | |
| logger.error(f"Skill extraction failed: {e}") | |
| return [] | |
| async def analyze_with_gemini(job_desc: str, skills: List[Dict], certs: Dict) -> Dict: | |
| """Get comprehensive analysis from Gemini""" | |
| try: | |
| prompt = f"""Analyze this job matching result: | |
| Job Description: {job_desc} | |
| Matched Skills: {json.dumps(skills, indent=2)} | |
| Recommended Certifications: {json.dumps(certs, indent=2)} | |
| Provide analysis including skill gaps, certification relevance, and recommendations. | |
| Return in JSON format with analysis of both internal and external certification buckets. | |
| """ | |
| messages = [SystemMessage(content="Return comprehensive analysis in JSON format."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| return json.loads(response_text[start_idx:end_idx]) | |
| return {"error": "Failed to parse analysis"} | |
| except Exception as e: | |
| logger.error(f"Analysis failed: {e}") | |
| return {"error": str(e)} | |
| # -------------------------- | |
| # JOB ONTOLOGY FUNCTIONS | |
| # -------------------------- | |
| async def extract_job_ontology(job_description: str, include_certificates: bool = True, | |
| min_relevance: float = 0.4) -> Dict: | |
| """Extract job ontology with categorized skills from internal database only""" | |
| try: | |
| # Convert internal skills to JSON for the prompt | |
| skills_json = skills_to_json(skills_df) | |
| # Use Gemini to extract and categorize skills (only from internal database) | |
| prompt = f"""Analyze this job description and extract skills categorized into: | |
| 1. Must to Have (essential technical skills) | |
| 2. Good to Have (nice-to-have technical skills) | |
| 3. Behavioral Skills (soft skills, personality traits) | |
| IMPORTANT: Only use skills from this internal skills list: {skills_json} | |
| If a skill in the job description is not in this list, DO NOT include it. | |
| Job Description: {job_description} | |
| Return in JSON format: | |
| {{ | |
| "job_title": "extracted job title", | |
| "must_have_skills": ["skill_name1", "skill_name2", ...], | |
| "good_to_have_skills": ["skill_name1", "skill_name2", ...], | |
| "behavioral_skills": ["skill_name1", "skill_name2", ...], | |
| "overall_analysis": "brief analysis of the role" | |
| }} | |
| """ | |
| messages = [SystemMessage( | |
| content="Return valid JSON with categorized skills using ONLY skills from the provided internal list."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| # Extract JSON from response | |
| ontology = { | |
| "job_title": "Unknown Role", | |
| "must_have_skills": [], | |
| "good_to_have_skills": [], | |
| "behavioral_skills": [], | |
| "overall_analysis": "Failed to extract detailed ontology" | |
| } | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| json_str = response_text[start_idx:end_idx] | |
| try: | |
| extracted_ontology = json.loads(json_str) | |
| # Validate and use the extracted ontology | |
| if isinstance(extracted_ontology, dict): | |
| ontology = extracted_ontology | |
| except json.JSONDecodeError: | |
| logger.warning("Failed to parse LLM response as JSON") | |
| # Enhance ontology with skill IDs (only internal skills) | |
| ontology = await enhance_ontology_with_skill_ids(ontology) | |
| # Get certificates for must-have skills if requested | |
| certificates = {} | |
| if include_certificates and ontology.get("must_have_skills"): | |
| must_have_certs = {} | |
| for skill_obj in ontology["must_have_skills"][:5]: # Limit to top 5 skills | |
| skill_name = skill_obj.get("name", "") | |
| if skill_name: | |
| cert_recommendations = await get_certificate_recommendations( | |
| skill_name, min_relevance, 3, 2 # Get 3 internal and 2 external certs per skill | |
| ) | |
| # Combine internal and external certs, remove duplicates | |
| all_certs = cert_recommendations["internal_bucket"] + cert_recommendations["external_bucket"] | |
| unique_certs = [] | |
| seen_names = set() | |
| for cert in all_certs: | |
| if cert["name"] not in seen_names: | |
| unique_certs.append(cert) | |
| seen_names.add(cert["name"]) | |
| must_have_certs[skill_name] = unique_certs[:3] # Limit to 3 certs per skill | |
| certificates["must_have_certificates"] = must_have_certs | |
| return { | |
| "status": "success", | |
| "ontology": ontology, | |
| "certificates": certificates if include_certificates else {} | |
| } | |
| except Exception as e: | |
| logger.error(f"Job ontology extraction failed: {e}") | |
| return { | |
| "status": "error", | |
| "message": f"Failed to extract job ontology: {str(e)}" | |
| } | |
| async def match_skills_with_ids(skills_list: List[str]) -> List[Dict]: | |
| """Match skill names with their IDs from the internal database only""" | |
| matched_skills = [] | |
| for skill_name in skills_list: | |
| # Try to find the skill in the internal database using strict matching | |
| matched_skill = await strict_internal_skill_match(skill_name) | |
| if matched_skill: | |
| matched_skills.append(matched_skill) | |
| else: | |
| # Log skipped external skills for debugging | |
| logger.debug(f"Skipped external skill: {skill_name}") | |
| return matched_skills | |
| # Add this function to enhance the ontology with better skill matching | |
| async def enhance_ontology_with_skill_ids(ontology: Dict) -> Dict: | |
| """Enhance ontology with proper skill IDs from internal database only""" | |
| try: | |
| # Process must-have skills | |
| if ontology.get("must_have_skills"): | |
| if isinstance(ontology["must_have_skills"][0], str): | |
| # Convert string list to object list with IDs (internal only) | |
| must_have_skills = await match_skills_with_ids(ontology["must_have_skills"]) | |
| ontology["must_have_skills"] = must_have_skills | |
| elif isinstance(ontology["must_have_skills"][0], dict) and "id" not in ontology["must_have_skills"][0]: | |
| # Convert dict list without IDs to include IDs (internal only) | |
| skill_names = [skill.get("name", "") for skill in ontology["must_have_skills"]] | |
| must_have_skills = await match_skills_with_ids(skill_names) | |
| ontology["must_have_skills"] = must_have_skills | |
| # Process good-to-have skills | |
| if ontology.get("good_to_have_skills"): | |
| if isinstance(ontology["good_to_have_skills"][0], str): | |
| good_to_have_skills = await match_skills_with_ids(ontology["good_to_have_skills"]) | |
| ontology["good_to_have_skills"] = good_to_have_skills | |
| elif isinstance(ontology["good_to_have_skills"][0], dict) and "id" not in ontology["good_to_have_skills"][ | |
| 0]: | |
| skill_names = [skill.get("name", "") for skill in ontology["good_to_have_skills"]] | |
| good_to_have_skills = await match_skills_with_ids(skill_names) | |
| ontology["good_to_have_skills"] = good_to_have_skills | |
| # Process behavioral skills | |
| if ontology.get("behavioral_skills"): | |
| if isinstance(ontology["behavioral_skills"][0], str): | |
| behavioral_skills = await match_skills_with_ids(ontology["behavioral_skills"]) | |
| ontology["behavioral_skills"] = behavioral_skills | |
| elif isinstance(ontology["behavioral_skills"][0], dict) and "id" not in ontology["behavioral_skills"][0]: | |
| skill_names = [skill.get("name", "") for skill in ontology["behavioral_skills"]] | |
| behavioral_skills = await match_skills_with_ids(skill_names) | |
| ontology["behavioral_skills"] = behavioral_skills | |
| return ontology | |
| except Exception as e: | |
| logger.error(f"Skill ID enhancement failed: {e}") | |
| return ontology | |
| async def strict_internal_skill_match(skill_name: str) -> Optional[Dict]: | |
| """Enhanced matching of skill name against internal database with semantic similarity""" | |
| if skills_df is None or skill_embeddings is None: | |
| return None | |
| # Try exact match first | |
| exact_matches = skills_df[skills_df[skill_col].str.lower() == skill_name.lower()] | |
| if not exact_matches.empty: | |
| row = exact_matches.iloc[0] | |
| return { | |
| "id": str(row[id_col]), | |
| "name": str(row[skill_col]), | |
| "source": "internal", | |
| "match_type": "exact" | |
| } | |
| # Try partial match with word boundaries | |
| word_boundary_matches = skills_df[ | |
| skills_df[skill_col].str.contains(r'\b' + re.escape(skill_name.lower()) + r'\b', | |
| case=False, na=False, regex=True) | |
| ] | |
| if not word_boundary_matches.empty: | |
| row = word_boundary_matches.iloc[0] | |
| return { | |
| "id": str(row[id_col]), | |
| "name": str(row[skill_col]), | |
| "source": "internal", | |
| "match_type": "word_boundary" | |
| } | |
| # Try contains match | |
| contains_matches = skills_df[ | |
| skills_df[skill_col].str.contains(skill_name.lower(), case=False, na=False, regex=False) | |
| ] | |
| if not contains_matches.empty: | |
| row = contains_matches.iloc[0] | |
| return { | |
| "id": str(row[id_col]), | |
| "name": str(row[skill_col]), | |
| "source": "internal", | |
| "match_type": "contains" | |
| } | |
| # Try semantic/related match using embeddings | |
| try: | |
| embedding_model = get_embedding_model() | |
| query_embedding = embedding_model.encode([skill_name]) | |
| scores = cosine_similarity(query_embedding, skill_embeddings)[0] | |
| # Find the best semantic match with a minimum similarity threshold | |
| best_match_idx = np.argmax(scores) | |
| best_score = scores[best_match_idx] | |
| # Only return semantic match if similarity is above threshold (0.6 = 60% similar) | |
| if best_score >= 0.6: | |
| row = skills_df.iloc[best_match_idx] | |
| return { | |
| "id": str(row[id_col]), | |
| "name": str(row[skill_col]), | |
| "source": "internal", | |
| "match_type": "semantic_related", | |
| "similarity_score": float(best_score) | |
| } | |
| except Exception as e: | |
| logger.warning(f"Semantic matching failed for '{skill_name}': {e}") | |
| return None | |
| async def process_job_description_url(url: str) -> str: | |
| """Extract job description text from a URL""" | |
| try: | |
| # Use Gemini to extract job description from the URL | |
| prompt = f"""Extract the job description text from this URL: {url} | |
| Return ONLY the job description text, nothing else. | |
| If this is not a job description URL, return "NOT_A_JOB_DESCRIPTION". | |
| """ | |
| messages = [SystemMessage(content="Return only the job description text."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| job_description = response.content.strip() | |
| if job_description == "NOT_A_JOB_DESCRIPTION": | |
| raise ValueError("The provided URL does not contain a job description") | |
| return job_description | |
| except Exception as e: | |
| logger.error(f"Job description URL processing failed: {e}") | |
| raise HTTPException(status_code=400, detail=f"Failed to extract job description from URL: {e}") | |
| # -------------------------- | |
| # NEW FUNCTIONS FOR CERTIFICATE URL ANALYSIS | |
| # -------------------------- | |
| async def analyze_certificate_with_gemini(cert_info: Dict, skills: List[str], recommendations: Dict) -> Dict: | |
| """Get comprehensive analysis of a certificate from Gemini""" | |
| try: | |
| prompt = f"""Analyze this certification and provide insights: | |
| Certification: {json.dumps(cert_info, indent=2)} | |
| Skills Covered: {skills} | |
| Related Certifications: {json.dumps(recommendations, indent=2)} | |
| Provide analysis including: | |
| 1. Career value and ROI | |
| 2. Skill development opportunities | |
| 3. Market demand and trends | |
| 4. Comparison with related certifications | |
| 5. Recommended learning path | |
| Return in JSON format. | |
| """ | |
| messages = [SystemMessage(content="Return comprehensive analysis in JSON format."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| return json.loads(response_text[start_idx:end_idx]) | |
| return {"error": "Failed to parse analysis"} | |
| except Exception as e: | |
| logger.error(f"Certificate analysis failed: {e}") | |
| return {"error": str(e)} | |
| async def generate_radar_data(cert_info: Dict, recommendations: Dict) -> Dict: | |
| """Generate radar chart data points for visualization""" | |
| try: | |
| # Calculate reliability score for the source certificate | |
| reliability_score = await calculate_certificate_reliability(cert_info) | |
| # Calculate average scores for recommendations | |
| internal_scores = [c.get("score", 0.5) for c in recommendations["internal_bucket"]] | |
| external_scores = [c.get("score", 0.5) for c in recommendations["external_bucket"]] | |
| avg_internal_score = sum(internal_scores) / len(internal_scores) if internal_scores else 0 | |
| avg_external_score = sum(external_scores) / len(external_scores) if external_scores else 0 | |
| # Count skills coverage | |
| all_skills = set() | |
| for cert in recommendations["internal_bucket"] + recommendations["external_bucket"]: | |
| if "skills" in cert and cert["skills"]: | |
| all_skills.update(cert["skills"]) | |
| # Calculate market alignment (placeholder - would use real data) | |
| market_alignment = await calculate_job_market_alignment(cert_info) | |
| # Prepare radar points | |
| radar_points = { | |
| "reliability": reliability_score["overall_score"], | |
| "internal_relevance": avg_internal_score, | |
| "external_relevance": avg_external_score, | |
| "skills_coverage": min(1.0, len(all_skills) / 20), # Normalized to 0-1 scale | |
| "market_alignment": market_alignment, | |
| "cost_effectiveness": 0.7, # Placeholder - would calculate based on actual cost data | |
| "learning_curve": 0.6 # Placeholder - would calculate based on complexity | |
| } | |
| return radar_points | |
| except Exception as e: | |
| logger.error(f"Radar data generation failed: {e}") | |
| return { | |
| "reliability": 0.5, | |
| "internal_relevance": 0.5, | |
| "external_relevance": 0.5, | |
| "skills_coverage": 0.5, | |
| "market_alignment": 0.5, | |
| "cost_effectiveness": 0.5, | |
| "learning_curve": 0.5 | |
| } | |
| # -------------------------- | |
| # DATA LOADING | |
| # -------------------------- | |
| def load_skills(): | |
| """Load skills data from skil.xlsx and create embeddings for semantic matching""" | |
| try: | |
| df = pd.read_excel(SKILL_FILE).fillna('') | |
| original_columns = df.columns.tolist() | |
| df.columns = [str(col).strip().lower().replace(' ', '_') for col in df.columns] | |
| # Better column detection logic | |
| # Look for ID column - prioritize 'skill_id' or columns with 'id' in them | |
| id_col = None | |
| for col in df.columns: | |
| if 'skill_id' in col.lower(): | |
| id_col = col | |
| break | |
| if id_col is None: | |
| id_col = next((c for c in df.columns if 'id' in c.lower() and 'skill' in c.lower()), None) | |
| if id_col is None: | |
| id_col = next((c for c in df.columns if 'id' in c.lower()), df.columns[0]) | |
| # Look for skill name column - prioritize 'skill_name' | |
| skill_col = None | |
| for col in df.columns: | |
| if 'skill_name' in col.lower(): | |
| skill_col = col | |
| break | |
| if skill_col is None: | |
| skill_col = next((c for c in df.columns if 'name' in c.lower() and 'skill' in c.lower()), None) | |
| if skill_col is None: | |
| skill_col = next((c for c in df.columns if 'name' in c.lower()), None) | |
| if skill_col is None: | |
| skill_col = df.columns[1] if len(df.columns) > 1 else df.columns[0] | |
| logger.info(f"Using ID column: '{id_col}' and Skill column: '{skill_col}'") | |
| logger.info(f"Available columns: {original_columns}") | |
| # Create embeddings for semantic matching | |
| logger.info("Creating skill embeddings for semantic matching...") | |
| embedding_model = get_embedding_model() | |
| # Create search text combining skill name and definition if available | |
| search_texts = [] | |
| for _, row in df.iterrows(): | |
| skill_name = str(row[skill_col]) | |
| skill_def = str(row.get('skill_definition', '')) | |
| if skill_def and skill_def != 'nan': | |
| search_text = f"{skill_name} {skill_def}" | |
| else: | |
| search_text = skill_name | |
| search_texts.append(search_text) | |
| skill_embeddings = embedding_model.encode(search_texts) | |
| logger.info(f"Created embeddings for {len(skill_embeddings)} skills") | |
| return df, id_col, skill_col, skill_embeddings | |
| except Exception as e: | |
| logger.error(f"Failed to load skills: {e}") | |
| raise | |
| async def load_certifications(): | |
| """Load certification data from skil.xlsx""" | |
| try: | |
| certs_df = pd.read_excel(SKILL_FILE).fillna('') | |
| original_columns = certs_df.columns.tolist() | |
| logger.info(f"Original columns in skil.xlsx: {original_columns}") | |
| # Map to expected columns - ADJUST THIS BASED ON ACTUAL COLUMNS | |
| column_mapping = {} | |
| for col in certs_df.columns: | |
| col_lower = col.lower() | |
| if 'title' in col_lower or 'name' in col_lower: | |
| column_mapping[col] = 'name' | |
| elif 'provider' in col_lower or 'vendor' in col_lower: | |
| column_mapping[col] = 'provider' | |
| elif 'link' in col_lower or 'url' in col_lower or 'hyperlink' in col_lower: | |
| column_mapping[col] = 'hyperlink' | |
| elif 'desc' in col_lower: | |
| column_mapping[col] = 'description' | |
| elif 'skill' in col_lower and 'name' not in col_lower: | |
| column_mapping[col] = 'skills' | |
| certs_df.rename(columns=column_mapping, inplace=True) | |
| # Ensure required columns exist | |
| for col in ['name', 'hyperlink']: | |
| if col not in certs_df.columns: | |
| certs_df[col] = '' | |
| logger.info(f"Mapped columns: {certs_df.columns.tolist()}") | |
| # Create search text from available columns | |
| search_parts = [] | |
| for col in ['name', 'provider', 'skills', 'description']: | |
| if col in certs_df.columns: | |
| search_parts.append(col) | |
| # If there are duplicate columns with same name, select first occurrence for search text | |
| parts_for_search = [] | |
| for col in search_parts: | |
| series = get_column_series(certs_df, col) | |
| if series is not None: | |
| parts_for_search.append(series.astype(str)) | |
| if parts_for_search: | |
| certs_df['search_text'] = pd.concat(parts_for_search, axis=1).agg('|'.join, axis=1) | |
| else: | |
| certs_df['search_text'] = '' | |
| # Initialize embeddings | |
| embedding_model = get_embedding_model() | |
| cert_embeddings = embedding_model.encode(certs_df['search_text'].tolist()) | |
| logger.info(f"Loaded {len(certs_df)} certifications with embeddings") | |
| return certs_df, cert_embeddings | |
| except Exception as e: | |
| logger.error(f"Failed to load certification data: {e}") | |
| # Return empty dataframe instead of raising error | |
| return pd.DataFrame(), None | |
| # -------------------------- | |
| # API ENDPOINTS | |
| # -------------------------- | |
| async def extract_job_ontology_endpoint(request: JobOntologyRequest): | |
| """Extract job ontology with categorized skills and relevant certificates""" | |
| try: | |
| result = await extract_job_ontology( | |
| request.job_description, | |
| request.include_certificates, | |
| request.min_relevance | |
| ) | |
| return result | |
| except Exception as e: | |
| logger.error(f"Job ontology endpoint failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Job ontology extraction failed: {e}") | |
| async def analyze_job_url_endpoint(url: str, include_certificates: bool = True, min_relevance: float = 0.4): | |
| """Analyze job description from a URL and extract ontology""" | |
| try: | |
| # Extract job description from URL | |
| job_description = await process_job_description_url(url) | |
| # Extract ontology | |
| result = await extract_job_ontology(job_description, include_certificates, min_relevance) | |
| # Add URL information to the result | |
| result["source_url"] = url | |
| result["extracted_description"] = job_description[:500] + "..." if len( | |
| job_description) > 500 else job_description | |
| return result | |
| except Exception as e: | |
| logger.error(f"Job URL analysis failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Job URL analysis failed: {e}") | |
| async def analyze_certificate_endpoint(cert_data: AnalyzeCertificateURL): | |
| """Analyze a certificate URL and extract skills from internal database and external sources""" | |
| try: | |
| # Process the certificate URL | |
| cert_info = await process_certificate_url(cert_data.url) | |
| if "error" in cert_info: | |
| raise HTTPException(status_code=400, detail=f"Failed to process URL: {cert_info['error']}") | |
| # Extract skills from the certificate | |
| cert_skills = [] | |
| if "skills" in cert_info: | |
| if isinstance(cert_info["skills"], str): | |
| cert_skills = [s.strip() for s in cert_info["skills"].split(",")] | |
| elif isinstance(cert_info["skills"], list): | |
| cert_skills = cert_info["skills"] | |
| # If no skills found in certificate, try to extract them from description | |
| if not cert_skills and "description" in cert_info: | |
| # Use Gemini to extract skills from description | |
| prompt = f"""Extract technical skills from this certification description: {cert_info['description']} | |
| Return a JSON array of skills: ["skill1", "skill2", "skill3"] | |
| """ | |
| messages = [SystemMessage(content="Return valid JSON array of skills."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| if response_text.startswith('[') and response_text.endswith(']'): | |
| try: | |
| cert_skills = json.loads(response_text) | |
| except: | |
| # Fallback: simple keyword extraction | |
| description = cert_info['description'].lower() | |
| tech_keywords = ["python", "java", "cloud", "aws", "azure", "security", | |
| "network", "database", "linux", "windows", "docker", "kubernetes"] | |
| cert_skills = [kw for kw in tech_keywords if kw in description] | |
| # Match extracted skills with internal database (skil.xlsx) to get IDs | |
| matched_internal_skills = [] | |
| external_skills = [] | |
| for skill in cert_skills: | |
| # Try to match with internal skills database | |
| matched_skill = await strict_internal_skill_match(skill) | |
| if matched_skill: | |
| matched_internal_skills.append(matched_skill) | |
| else: | |
| # Keep as external skill if not found in internal database | |
| external_skills.append({ | |
| "name": skill, | |
| "source": "external", | |
| "match_type": "extracted_from_certificate" | |
| }) | |
| # Calculate reliability score for the certificate itself | |
| reliability_score = await calculate_certificate_reliability(cert_info) | |
| response = { | |
| "status": "success", | |
| "source_certificate": { | |
| "name": cert_info.get("name", "Unknown Certification"), | |
| "provider": cert_info.get("provider", "Unknown Provider"), | |
| "url": cert_data.url, | |
| "description": cert_info.get("description", ""), | |
| "reliability_score": reliability_score | |
| }, | |
| "skills_analysis": { | |
| "total_skills_extracted": len(cert_skills), | |
| "internal_skills_matched": len(matched_internal_skills), | |
| "external_skills_found": len(external_skills), | |
| "internal_skills": matched_internal_skills, | |
| "external_skills": external_skills | |
| }, | |
| "analysis_method": "skill_extraction_only" | |
| } | |
| # Optional detailed analysis using Gemini | |
| if cert_data.analyze: | |
| analysis_prompt = f"""Analyze the skills extracted from this certification: | |
| Certificate: {cert_info.get('name', 'Unknown')} | |
| Provider: {cert_info.get('provider', 'Unknown')} | |
| Internal Skills (from database): {json.dumps(matched_internal_skills, indent=2)} | |
| External Skills: {json.dumps(external_skills, indent=2)} | |
| Provide insights about: | |
| 1. Skill relevance and market demand | |
| 2. Career opportunities these skills enable | |
| 3. Skill categories and technical domains covered | |
| 4. Learning progression and prerequisites | |
| Return in JSON format. | |
| """ | |
| messages = [SystemMessage(content="Return comprehensive skill analysis in JSON format."), | |
| HumanMessage(content=analysis_prompt)] | |
| analysis_response = await ainvoke_llm(messages) | |
| analysis_text = analysis_response.content.strip() | |
| if '{' in analysis_text and '}' in analysis_text: | |
| start_idx = analysis_text.find('{') | |
| end_idx = analysis_text.rfind('}') + 1 | |
| try: | |
| response["detailed_analysis"] = json.loads(analysis_text[start_idx:end_idx]) | |
| except: | |
| response["detailed_analysis"] = {"error": "Failed to parse analysis"} | |
| return response | |
| except Exception as e: | |
| logger.error(f"Certificate analysis failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Certificate analysis failed: {e}") | |
| async def get_certificate_reliability(cert_name: str): | |
| """Get reliability score for a specific certificate""" | |
| try: | |
| # Search for certificate in internal database | |
| cert_data = {} | |
| if certs_df is not None and 'name' in certs_df.columns: | |
| matches = certs_df[certs_df['name'].str.contains(cert_name, na=False, case=False, regex=False)] | |
| if not matches.empty: | |
| cert_data = matches.iloc[0].to_dict() | |
| # If not found internally, try to find information | |
| if not cert_data: | |
| # Use Gemini to get certificate information | |
| prompt = f"""Get information about this certification: {cert_name} | |
| Return JSON with: name, provider, description, typical skills covered. | |
| """ | |
| messages = [SystemMessage(content="Return valid JSON with certification information."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| json_str = response_text[start_idx:end_idx] | |
| cert_data = json.loads(json_str) | |
| # Calculate reliability score | |
| reliability_score = await calculate_certificate_reliability(cert_data) | |
| return { | |
| "status": "success", | |
| "certificate": cert_name, | |
| "reliability_score": reliability_score, | |
| "factors": CERT_RELIABILITY_FACTORS, | |
| "certificate_data": cert_data | |
| } | |
| except Exception as e: | |
| logger.error(f"Reliability check failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Reliability check failed: {e}") | |
| async def compare_certifications(comparison: CertificationComparison): | |
| """Compare multiple certifications across various dimensions""" | |
| try: | |
| comparison_results = {} | |
| for cert_name in comparison.cert_names: | |
| # Get certificate information | |
| cert_info = await get_certificate_info(cert_name) | |
| reliability = await calculate_certificate_reliability(cert_info) | |
| comparison_results[cert_name] = { | |
| "reliability": reliability, | |
| "skills": cert_info.get("skills", []), | |
| "provider": cert_info.get("provider", ""), | |
| "estimated_cost": await estimate_certification_cost(cert_info), | |
| "duration": await estimate_preparation_time(cert_info), | |
| "job_market_alignment": await calculate_job_market_alignment(cert_info) | |
| } | |
| # Generate comparative analysis | |
| analysis = await generate_comparative_analysis(comparison_results, comparison.compare_by) | |
| return { | |
| "status": "success", | |
| "comparison": comparison_results, | |
| "analysis": analysis, | |
| "recommendation": await generate_recommendation(comparison_results) | |
| } | |
| except Exception as e: | |
| logger.error(f"Certification comparison failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Comparison failed: {e}") | |
| async def skill_gap_analysis(analysis: SkillGapAnalysis): | |
| """Analyze skill gaps between current and target skills/roles""" | |
| try: | |
| # Get missing skills | |
| missing_skills = list(set(analysis.target_skills) - set(analysis.current_skills)) | |
| # Get overlapping skills | |
| overlapping_skills = list(set(analysis.target_skills) & set(analysis.current_skills)) | |
| # Find certifications to bridge the gap | |
| cert_recommendations = await recommend_certifications_for_skills(missing_skills) | |
| # Generate learning recommendations | |
| learning_recommendations = await generate_learning_recommendations(missing_skills) | |
| return { | |
| "status": "success", | |
| "missing_skills": missing_skills, | |
| "overlapping_skills": overlapping_skills, | |
| "skill_gap_percentage": len(missing_skills) / len( | |
| analysis.target_skills) * 100 if analysis.target_skills else 0, | |
| "certification_recommendations": cert_recommendations, | |
| "learning_recommendations": learning_recommendations, | |
| "timeline_estimation": await estimate_timeline_for_skills(missing_skills) | |
| } | |
| except Exception as e: | |
| logger.error(f"Skill gap analysis failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Skill gap analysis failed: {e}") | |
| async def track_certification(tracker: CertificationTracker): | |
| """Track certification status and renewal requirements""" | |
| try: | |
| cert_info = await get_certificate_info(tracker.cert_name) | |
| # Calculate days until expiration | |
| days_until_expiration = None | |
| if tracker.expiration_date: | |
| expiration = datetime.strptime(tracker.expiration_date, "%Y-%m-%d") | |
| days_until_expiration = (expiration - datetime.now()).days | |
| # Get renewal requirements if not provided | |
| if not tracker.renewal_requirements: | |
| tracker.renewal_requirements = await extract_renewal_requirements(cert_info) | |
| return { | |
| "status": "success", | |
| "certification": tracker.cert_name, | |
| "days_until_expiration": days_until_expiration, | |
| "renewal_requirements": tracker.renewal_requirements, | |
| "renewal_cost_estimate": await estimate_renewal_cost(tracker.cert_name), | |
| "recommended_renewal_timeline": await generate_renewal_timeline(tracker), | |
| "alternative_renewal_options": await find_alternative_renewal_options(tracker.cert_name) | |
| } | |
| except Exception as e: | |
| logger.error(f"Certification tracking failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Certification tracking failed: {e}") | |
| async def analyze_input(input_data: Union[InputText, AnalyzeCertificateURL, JobOntologyRequest]): | |
| # Check if it's a job ontology request | |
| if isinstance(input_data, JobOntologyRequest): | |
| return await extract_job_ontology_endpoint(input_data) | |
| # Check if it's a URL (existing functionality) | |
| elif isinstance(input_data, AnalyzeCertificateURL) or (hasattr(input_data, 'text') and | |
| (input_data.text.startswith('http://') or | |
| input_data.text.startswith('https://'))): | |
| # Handle as certificate URL or job URL | |
| url_to_check = input_data.url if isinstance(input_data, AnalyzeCertificateURL) else input_data.text | |
| if "linkedin.com" in url_to_check or "indeed.com" in url_to_check or "job" in url_to_check.lower(): | |
| # Likely a job description URL | |
| try: | |
| return await analyze_job_url_endpoint( | |
| url_to_check, | |
| include_certificates=True, | |
| min_relevance=input_data.min_relevance if hasattr(input_data, 'min_relevance') else 0.4 | |
| ) | |
| except: | |
| # Fall back to certificate analysis | |
| pass | |
| # Handle as certificate URL (existing code) | |
| if isinstance(input_data, InputText): | |
| cert_data = AnalyzeCertificateURL( | |
| url=input_data.text, | |
| min_relevance=input_data.min_relevance, | |
| max_internal_results=input_data.max_internal_results, | |
| max_external_results=input_data.max_external_results, | |
| analyze=input_data.analyze | |
| ) | |
| else: | |
| cert_data = input_data | |
| return await analyze_certificate_endpoint(cert_data) | |
| else: | |
| # Handle as text input (existing functionality with job description detection) | |
| try: | |
| input_type = await determine_input_type(input_data.text) | |
| if input_type == "job_description": | |
| # Use the new ontology extraction for job descriptions | |
| return await extract_job_ontology( | |
| input_data.text, | |
| include_certificates=True, | |
| min_relevance=input_data.min_relevance | |
| ) | |
| else: | |
| # Existing skill-based processing | |
| cert_recommendations = await get_certificate_recommendations( | |
| input_data.text, | |
| input_data.min_relevance, | |
| input_data.max_internal_results, | |
| input_data.max_external_results | |
| ) | |
| response = { | |
| "status": "success", | |
| "input_type": "skill", | |
| "matches": { | |
| "certifications": cert_recommendations | |
| }, | |
| "analysis_method": "internal_external_buckets" | |
| } | |
| if input_data.analyze: | |
| try: | |
| analysis_prompt = f"Analyze these certification recommendations for skill '{input_data.text}': {json.dumps(cert_recommendations)}" | |
| messages = [SystemMessage(content="Return valid JSON analysis."), | |
| HumanMessage(content=analysis_prompt)] | |
| response_content = (await ainvoke_llm(messages)).content | |
| if '{' in response_content and '}' in response_content: | |
| start_idx = response_content.find('{') | |
| end_idx = response_content.rfind('}') + 1 | |
| response["analysis"] = json.loads(response_content[start_idx:end_idx]) | |
| except Exception as e: | |
| response["analysis"] = {"error": str(e)} | |
| return response | |
| except Exception as e: | |
| logger.error(f"Analysis failed: {e}") | |
| raise HTTPException(status_code=500, detail=f"Analysis failed: {e}") | |
| async def health_check(): | |
| """Health check endpoint""" | |
| return { | |
| "status": "healthy", | |
| "skills_loaded": len(skills_df) if skills_df is not None else 0, | |
| "certifications_loaded": len(certs_df) if certs_df is not None else 0 | |
| } | |
| # -------------------------- | |
| # HELPER FUNCTIONS | |
| # -------------------------- | |
| async def get_certificate_info(cert_name: str) -> Dict: | |
| """Get certificate information from internal database or external source""" | |
| cert_data = {} | |
| if certs_df is not None and 'name' in certs_df.columns: | |
| try: | |
| name_series = get_column_series(certs_df, 'name') | |
| if name_series is not None: | |
| mask = name_series.astype(str).str.contains(cert_name, na=False, case=False, regex=False) | |
| matches = certs_df[mask] | |
| else: | |
| matches = pd.DataFrame() | |
| except Exception: | |
| matches = pd.DataFrame() | |
| if not matches.empty: | |
| cert_data = matches.iloc[0].to_dict() | |
| # If not found internally, use Gemini to get information | |
| if not cert_data: | |
| prompt = f"""Get information about this certification: {cert_name} | |
| Return JSON with: name, provider, description, skills covered. | |
| """ | |
| messages = [SystemMessage(content="Return valid JSON with certification information."), | |
| HumanMessage(content=prompt)] | |
| response = await ainvoke_llm(messages) | |
| response_text = response.content.strip() | |
| if '{' in response_text and '}' in response_text: | |
| start_idx = response_text.find('{') | |
| end_idx = response_text.rfind('}') + 1 | |
| json_str = response_text[start_idx:end_idx] | |
| cert_data = json.loads(json_str) | |
| return cert_data | |
| async def estimate_certification_cost(cert_info: Dict) -> Dict: | |
| """Estimate total cost of certification""" | |
| # Placeholder implementation - would integrate with actual cost data | |
| return { | |
| "exam_fee": 150, | |
| "study_materials": 100, | |
| "training_course": 300, | |
| "total": 550, | |
| "currency": "USD" | |
| } | |
| async def estimate_preparation_time(cert_info: Dict) -> Dict: | |
| """Estimate preparation time for certification""" | |
| # Placeholder implementation | |
| return { | |
| "hours_required": 40, | |
| "weeks_recommended": 8, | |
| "intensity": "moderate" | |
| } | |
| async def calculate_job_market_alignment(cert_info: Dict) -> float: | |
| """Calculate how well the certification aligns with job market demands""" | |
| # Placeholder implementation | |
| return 0.75 | |
| async def recommend_certifications_for_skills(skills: List[str], max_results: int = 5) -> List[Dict]: | |
| """Recommend certifications for a set of skills""" | |
| recommendations = [] | |
| for skill in skills[:3]: # Limit to top 3 skills | |
| certs = await get_certificate_recommendations(skill, 0.3, 2, 2) | |
| for cert in certs["internal_bucket"] + certs["external_bucket"]: | |
| if cert["name"] not in [r["name"] for r in recommendations]: | |
| recommendations.append(cert) | |
| if len(recommendations) >= max_results: | |
| break | |
| if len(recommendations) >= max_results: | |
| break | |
| return recommendations | |
| async def generate_learning_recommendations(skills: List[str]) -> List[Dict]: | |
| """Generate learning recommendations for skills""" | |
| # Placeholder implementation | |
| return [ | |
| { | |
| "skill": skill, | |
| "resources": ["Online course", "Practice labs", "Documentation"], | |
| "estimated_time": "2-4 weeks" | |
| } | |
| for skill in skills[:3] # Limit to top 3 skills | |
| ] | |
| async def estimate_timeline_for_skills(skills: List[str]) -> Dict: | |
| """Estimate timeline for acquiring skills""" | |
| # Placeholder implementation | |
| return { | |
| "total_weeks": len(skills) * 3, | |
| "total_hours": len(skills) * 20, | |
| "recommended_schedule": "2-3 hours per week per skill" | |
| } | |
| async def extract_renewal_requirements(cert_info: Dict) -> List[str]: | |
| """Extract renewal requirements from certificate information""" | |
| # Placeholder implementation | |
| return ["Continuing education units", "Annual fee", "Periodic exam"] | |
| async def estimate_renewal_cost(cert_name: str) -> Dict: | |
| """Estimate renewal cost for certification""" | |
| # Placeholder implementation | |
| return { | |
| "fee": 100, | |
| "continuing_education": 200, | |
| "total": 300, | |
| "currency": "USD" | |
| } | |
| async def generate_renewal_timeline(tracker: CertificationTracker) -> Dict: | |
| """Generate renewal timeline for certification""" | |
| # Placeholder implementation | |
| return { | |
| "recommended_start": "90 days before expiration", | |
| "steps": [ | |
| "Complete continuing education", | |
| "Submit renewal application", | |
| "Pay renewal fee" | |
| ] | |
| } | |
| async def find_alternative_renewal_options(cert_name: str) -> List[Dict]: | |
| """Find alternative renewal options for certification""" | |
| # Placeholder implementation | |
| return [ | |
| { | |
| "option": "Higher-level certification", | |
| "description": "Earn a more advanced certification instead of renewing", | |
| "benefits": "Demonstrates continued growth and expertise" | |
| } | |
| ] | |
| async def generate_comparative_analysis(comparison_data: Dict, compare_by: str) -> Dict: | |
| """Generate comparative analysis of certifications""" | |
| # Placeholder implementation | |
| return { | |
| "summary": "Comparative analysis based on " + compare_by, | |
| "key_findings": ["Certification A has higher reliability", "Certification B covers more skills"], | |
| "recommendation": "Consider your specific career goals when choosing" | |
| } | |
| async def generate_recommendation(comparison_data: Dict) -> str: | |
| """Generate recommendation based on comparison data""" | |
| # Placeholder implementation | |
| return "Based on the analysis, Certification A is recommended for most users due to its higher reliability score." | |
| # CORS middleware | |
| app.add_middleware( | |
| CORSMiddleware, | |
| allow_origins=["*"], | |
| allow_methods=["*"], | |
| allow_headers=["*"], | |
| ) | |
| # -------------------------- | |
| # ENHANCED GRADIO UI | |
| # -------------------------- | |
| def create_gradio_interface(): | |
| """Create Gradio interface for the application""" | |
| with gr.Blocks(title="Advanced Certification Analyzer", theme=gr.themes.Soft()) as demo: | |
| gr.Markdown("# 🎯 Advanced Certification Analyzer") | |
| gr.Markdown( | |
| "Comprehensive certification analysis with reliability scoring, skill matching, and career path recommendations") | |
| with gr.Tab("Job Ontology Analysis"): | |
| with gr.Row(): | |
| with gr.Column(scale=2): | |
| job_desc_input = gr.Textbox( | |
| label="Job Description", | |
| placeholder="Paste a job description here...", | |
| lines=5, | |
| max_lines=10 | |
| ) | |
| with gr.Row(): | |
| min_relevance_ontology = gr.Slider( | |
| minimum=0.1, | |
| maximum=1.0, | |
| value=0.4, | |
| step=0.1, | |
| label="Minimum Relevance Score" | |
| ) | |
| include_certs_check = gr.Checkbox( | |
| value=True, | |
| label="Include Certificates" | |
| ) | |
| analyze_job_btn = gr.Button("Extract Job Ontology", variant="primary") | |
| clear_job_btn = gr.Button("Clear") | |
| with gr.Column(scale=3): | |
| job_title_output = gr.Textbox(label="Job Title") | |
| with gr.Tab("Must Have Skills"): | |
| must_have_skills = gr.JSON(label="Must Have Skills") | |
| with gr.Tab("Good to Have Skills"): | |
| good_to_have_skills = gr.JSON(label="Good to Have Skills") | |
| with gr.Tab("Behavioral Skills"): | |
| behavioral_skills = gr.JSON(label="Behavioral Skills") | |
| with gr.Tab("Certificates"): | |
| cert_recommendations = gr.JSON(label="Certificate Recommendations") | |
| with gr.Tab("Analysis"): | |
| job_analysis = gr.JSON(label="Job Analysis") | |
| with gr.Tab("Main Analysis"): | |
| with gr.Row(): | |
| with gr.Column(scale=2): | |
| input_text = gr.Textbox( | |
| label="Input Text", | |
| placeholder="Paste a job description or enter a specific skill...", | |
| lines=5, | |
| max_lines=10 | |
| ) | |
| with gr.Row(): | |
| min_relevance = gr.Slider( | |
| minimum=0.1, | |
| maximum=1.0, | |
| value=0.4, | |
| step=0.1, | |
| label="Minimum Relevance Score" | |
| ) | |
| analyze_checkbox = gr.Checkbox( | |
| value=True, | |
| label="Enable Detailed Analysis" | |
| ) | |
| with gr.Row(): | |
| max_internal = gr.Slider( | |
| minimum=1, | |
| maximum=50, | |
| value=20, | |
| step=1, | |
| label="Max Internal Results" | |
| ) | |
| max_external = gr.Slider( | |
| minimum=1, | |
| maximum=20, | |
| value=10, | |
| step=1, | |
| label="Max External Results" | |
| ) | |
| submit_btn = gr.Button("Analyze", variant="primary") | |
| clear_btn = gr.Button("Clear") | |
| with gr.Column(scale=3): | |
| output_type = gr.Textbox(label="Input Type Detected") | |
| status = gr.Textbox(label="Status") | |
| with gr.Tab("Internal Certifications"): | |
| internal_certs = gr.JSON(label="Internal Certification Recommendations") | |
| with gr.Tab("External Certifications"): | |
| external_certs = gr.JSON(label="External Certification Recommendations") | |
| with gr.Tab("Skills"): | |
| skills_output = gr.JSON(label="Extracted Skills") | |
| with gr.Tab("Analysis"): | |
| analysis_output = gr.JSON(label="Detailed Analysis") | |
| with gr.Tab("Raw Output"): | |
| raw_output = gr.JSON(label="Complete Response") | |
| with gr.Tab("Certificate URL Analysis"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| cert_url_input = gr.Textbox(label="Certificate URL", placeholder="Enter certificate URL...") | |
| extract_skills_check = gr.Checkbox(value=True, label="Extract Skills") | |
| analyze_reliability_check = gr.Checkbox(value=True, label="Analyze Reliability") | |
| analyze_url_btn = gr.Button("Analyze URL", variant="primary") | |
| with gr.Column(scale=2): | |
| cert_info_output = gr.JSON(label="Certificate Information") | |
| cert_skills_output = gr.JSON(label="Skills from Certificate") | |
| cert_reliability_output = gr.JSON(label="Reliability Analysis") | |
| with gr.Tab("Reliability Check"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| cert_name_input = gr.Textbox(label="Certificate Name", placeholder="Enter certificate name...") | |
| check_reliability_btn = gr.Button("Check Reliability", variant="primary") | |
| with gr.Column(scale=2): | |
| reliability_output = gr.JSON(label="Reliability Score") | |
| with gr.Tab("Skill Gap Analysis"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| current_skills_input = gr.Textbox(label="Current Skills", | |
| placeholder="Enter your current skills (comma-separated)...") | |
| target_skills_input = gr.Textbox(label="Target Skills", | |
| placeholder="Enter target skills (comma-separated)...") | |
| analyze_gap_btn = gr.Button("Analyze Gap", variant="primary") | |
| with gr.Column(scale=2): | |
| gap_analysis_output = gr.JSON(label="Gap Analysis Results") | |
| # Examples | |
| gr.Examples( | |
| examples=[ | |
| ["Python programming and data analysis with machine learning experience"], | |
| ["Cloud computing and AWS infrastructure management"], | |
| ["Cybersecurity and network security protocols"], | |
| ["Project management with agile methodology experience"], | |
| ["Frontend development with React and JavaScript"] | |
| ], | |
| inputs=input_text | |
| ) | |
| def analyze_job_ontology(job_desc, min_rel, include_certs): | |
| """Wrapper function for Gradio to call the job ontology API""" | |
| if not job_desc.strip(): | |
| return "Please enter a job description", {}, {}, {}, {}, {} | |
| try: | |
| # Call the internal function | |
| result = asyncio.run(extract_job_ontology(job_desc, include_certs, min_rel)) | |
| if result["status"] == "success": | |
| ontology = result["ontology"] | |
| certs = result.get("certificates", {}) | |
| return ( | |
| ontology.get("job_title", "Unknown"), | |
| ontology.get("must_have_skills", []), | |
| ontology.get("good_to_have_skills", []), | |
| ontology.get("behavioral_skills", []), | |
| certs.get("must_have_certificates", {}), | |
| ontology.get("overall_analysis", "") | |
| ) | |
| else: | |
| return "Error", {}, {}, {}, {}, result.get("message", "Unknown error") | |
| except Exception as e: | |
| return f"Error: {str(e)}", {}, {}, {}, {}, {} | |
| def analyze_text(text, min_rel, max_int, max_ext, analyze_flag): | |
| """Wrapper function for Gradio to call the API""" | |
| if not text.strip(): | |
| return "Please enter some text", "Error", {}, {}, {}, {}, {} | |
| try: | |
| # Prepare the request | |
| payload = { | |
| "text": text, | |
| "min_relevance": min_rel, | |
| "max_internal_results": int(max_int), | |
| "max_external_results": int(max_ext), | |
| "analyze": analyze_flag | |
| } | |
| # For Gradio, we'll call the function directly instead of HTTP | |
| result = asyncio.run(analyze_input_internal(payload)) | |
| if result["status"] == "success": | |
| input_type = result["input_type"] | |
| matches = result["matches"] | |
| # Extract different components | |
| internal_bucket = matches.get("certifications", {}).get("internal_bucket", []) | |
| external_bucket = matches.get("certifications", {}).get("external_bucket", []) | |
| skills_list = matches.get("skills", []) | |
| analysis = result.get("analysis", {}) | |
| return ( | |
| input_type, | |
| "Analysis completed successfully", | |
| internal_bucket, | |
| external_bucket, | |
| skills_list, | |
| analysis, | |
| result | |
| ) | |
| else: | |
| return "error", "Analysis failed", {}, {}, {}, {}, {} | |
| except Exception as e: | |
| logger.error(f"Gradio analysis error: {e}") | |
| return "error", f"Error: {str(e)}", {}, {}, {}, {}, {} | |
| def analyze_cert_url(url, extract_skills, analyze_reliability): | |
| """Analyze certificate URL""" | |
| if not url.strip(): | |
| return {}, {}, {} | |
| try: | |
| # Call the internal function | |
| result = asyncio.run(process_certificate_url(url)) | |
| # Extract skills | |
| skills_from_cert = [] | |
| if extract_skills and "skills" in result: | |
| cert_skills = result["skills"] | |
| if isinstance(cert_skills, str): | |
| cert_skills = [s.strip() for s in cert_skills.split(",")] | |
| for skill in cert_skills: | |
| skills_from_cert.append({ | |
| "skill": skill, | |
| "source": "certificate" | |
| }) | |
| # Calculate reliability | |
| reliability_score = {} | |
| if analyze_reliability: | |
| reliability_score = asyncio.run(calculate_certificate_reliability(result)) | |
| return result, skills_from_cert, reliability_score | |
| except Exception as e: | |
| return {"error": str(e)}, {}, {} | |
| def check_reliability(cert_name): | |
| """Check reliability of a certificate""" | |
| if not cert_name.strip(): | |
| return {} | |
| try: | |
| # Search for certificate in internal database | |
| cert_data = {} | |
| if certs_df is not None and 'name' in certs_df.columns: | |
| matches = certs_df[certs_df['name'].str.contains(cert_name, na=False, case=False, regex=False)] | |
| if not matches.empty: | |
| cert_data = matches.iloc[0].to_dict() | |
| # Calculate reliability score | |
| reliability_score = asyncio.run(calculate_certificate_reliability(cert_data)) | |
| return { | |
| "certificate": cert_name, | |
| "reliability_score": reliability_score, | |
| "factors": CERT_RELIABILITY_FACTORS | |
| } | |
| except Exception as e: | |
| return {"error": str(e)} | |
| def analyze_skill_gap(current_skills, target_skills): | |
| """Analyze skill gap""" | |
| if not current_skills.strip() or not target_skills.strip(): | |
| return {} | |
| try: | |
| current_skills_list = [s.strip() for s in current_skills.split(",")] | |
| target_skills_list = [s.strip() for s in target_skills.split(",")] | |
| # Get missing skills | |
| missing_skills = list(set(target_skills_list) - set(current_skills_list)) | |
| # Get overlapping skills | |
| overlapping_skills = list(set(target_skills_list) & set(current_skills_list)) | |
| return { | |
| "missing_skills": missing_skills, | |
| "overlapping_skills": overlapping_skills, | |
| "gap_percentage": f"{(len(missing_skills) / len(target_skills_list) * 100):.1f}%" | |
| } | |
| except Exception as e: | |
| return {"error": str(e)} | |
| # Connect the interface | |
| analyze_job_btn.click( | |
| fn=analyze_job_ontology, | |
| inputs=[job_desc_input, min_relevance_ontology, include_certs_check], | |
| outputs=[job_title_output, must_have_skills, good_to_have_skills, behavioral_skills, cert_recommendations, | |
| job_analysis] | |
| ) | |
| clear_job_btn.click( | |
| fn=lambda: ["", {}, {}, {}, {}, {}], | |
| outputs=[job_desc_input, job_title_output, must_have_skills, good_to_have_skills, behavioral_skills, | |
| job_analysis] | |
| ) | |
| submit_btn.click( | |
| fn=analyze_text, | |
| inputs=[input_text, min_relevance, max_internal, max_external, analyze_checkbox], | |
| outputs=[output_type, status, internal_certs, external_certs, skills_output, analysis_output, raw_output] | |
| ) | |
| clear_btn.click( | |
| fn=lambda: ["", "", "", {}, {}, {}, {}, {}], | |
| outputs=[input_text, output_type, status, internal_certs, external_certs, skills_output, analysis_output, | |
| raw_output] | |
| ) | |
| analyze_url_btn.click( | |
| fn=analyze_cert_url, | |
| inputs=[cert_url_input, extract_skills_check, analyze_reliability_check], | |
| outputs=[cert_info_output, cert_skills_output, cert_reliability_output] | |
| ) | |
| check_reliability_btn.click( | |
| fn=check_reliability, | |
| inputs=cert_name_input, | |
| outputs=reliability_output | |
| ) | |
| analyze_gap_btn.click( | |
| fn=analyze_skill_gap, | |
| inputs=[current_skills_input, target_skills_input], | |
| outputs=gap_analysis_output | |
| ) | |
| return demo | |
| async def analyze_input_internal(payload): | |
| """Internal analysis function for Gradio""" | |
| class InputTextInternal: | |
| def __init__(self, text, min_relevance=0.4, max_internal_results=20, max_external_results=10, analyze=True): | |
| self.text = text | |
| self.min_relevance = min_relevance | |
| self.max_internal_results = max_internal_results | |
| self.max_external_results = max_external_results | |
| self.analyze = analyze | |
| input_data = InputTextInternal(**payload) | |
| return await analyze_input(input_data) | |
| # Mount Gradio app to FastAPI | |
| gradio_app = create_gradio_interface() | |
| app = gr.mount_gradio_app(app, gradio_app, path="/") | |
| async def startup_event(): | |
| """Initialize data on startup""" | |
| try: | |
| logger.info("Loading data...") | |
| global certs_df, cert_embeddings, skills_df, skill_embeddings, id_col, skill_col | |
| # Load skills with embeddings | |
| skills_df, id_col, skill_col, skill_embeddings = load_skills() | |
| logger.info(f"Loaded {len(skills_df)} skills with embeddings") | |
| # Load certifications from the same file | |
| certs_df, cert_embeddings = await load_certifications() | |
| logger.info(f"Loaded {len(certs_df)} certifications") | |
| except Exception as e: | |
| logger.error(f"Data loading failed: {e}") | |
| raise RuntimeError(f"Service initialization failed: {e}") | |
| if __name__ == "__main__": | |
| import uvicorn | |
| uvicorn.run(app, host="0.0.0.0", port=7860) | |
| async def match_certificate_skills_with_internal(cert_skills: List[str]) -> List[Dict]: | |
| """Match certificate skills with internal skills database and return only matching skills""" | |
| matched_skills = [] | |
| if not cert_skills or not skills_df is not None: | |
| return matched_skills | |
| for cert_skill in cert_skills: | |
| if not cert_skill or cert_skill.strip() == '': | |
| continue | |
| # Try to find matching skill in internal database | |
| matched_skill = await strict_internal_skill_match(cert_skill.strip()) | |
| if matched_skill: | |
| matched_skills.append({ | |
| "id": matched_skill["id"], | |
| "name": matched_skill["name"], | |
| "source": "internal", | |
| "match_type": matched_skill.get("match_type", "unknown"), | |
| "similarity_score": matched_skill.get("similarity_score", 1.0), | |
| "original_cert_skill": cert_skill.strip() | |
| }) | |
| else: | |
| # Log skipped external/unmatched skills | |
| logger.debug(f"Skipped unmatched certificate skill: {cert_skill}") | |
| return matched_skills | |
| async def filter_relevant_certificates_only(cert_recommendations: Dict, extracted_skills: List[str]) -> Dict: | |
| """Filter certificate recommendations to only include those with skills matching extracted skills""" | |
| if not extracted_skills: | |
| return {"internal_bucket": [], "external_bucket": []} | |
| # Create set of extracted skill names for comparison | |
| extracted_skill_names = {skill.lower().strip() for skill in extracted_skills if skill and skill.strip()} | |
| filtered_recommendations = {"internal_bucket": [], "external_bucket": []} | |
| for bucket_name in ["internal_bucket", "external_bucket"]: | |
| for cert in cert_recommendations.get(bucket_name, []): | |
| cert_skills = cert.get("skills", []) | |
| if not cert_skills: | |
| continue | |
| # Check if any certificate skills match our extracted skills | |
| cert_skill_names = {skill.lower().strip() for skill in cert_skills if skill and skill.strip()} | |
| # Calculate overlap | |
| overlap = extracted_skill_names.intersection(cert_skill_names) | |
| overlap_ratio = len(overlap) / len(extracted_skill_names) if extracted_skill_names else 0 | |
| # Only include certificates with meaningful skill overlap | |
| if overlap_ratio >= 0.3: # At least 30% overlap | |
| cert["skill_overlap_ratio"] = overlap_ratio | |
| cert["matching_skills"] = list(overlap) | |
| filtered_recommendations[bucket_name].append(cert) | |
| return filtered_recommendations |