Spaces:
Running
Running
File size: 29,117 Bytes
bc32894 02c72ce bc32894 02c72ce bc32894 02c72ce bc32894 02c72ce bc32894 f899b0a bc32894 02c72ce bc32894 02c72ce bc32894 02c72ce | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 | #!/usr/bin/env python3
"""
app.py β Gradio Sandbox for V7 Elite Ranker
HuggingFace Spaces compatible. Upload β€100 candidates β get ranked CSV.
"""
import json
import csv
import sys
import time
import tempfile
import traceback
from pathlib import Path
from io import StringIO
import numpy as np
import pandas as pd
import gradio as gr
# Ensure lib/ is importable
sys.path.insert(0, str(Path(__file__).resolve().parent))
from lib import schema, features, honeypot, embeddings as emb_mod, scoring, reasoning
from lib.jd_parser import get_jd
from lib.hiring_intent import get_intent
from lib.query_expansion import get_expanded_text
from precompute import extract_features_v7
TOP_N = 100
# ββ Built-in 10-candidate sample for quick demo ββββββββββββββββββββββββββ
SAMPLE_CANDIDATES = [
{
"candidate_id": "CAND_0001001",
"profile": {
"anonymized_name": "Demo Elite", "headline": "Senior AI Engineer | RAG, Vector Search, Ranking",
"summary": "8 years building production search and ranking systems. Led the recommendation engine at Razorpay serving 50M+ users. Deep expertise in embeddings, hybrid retrieval, and LLM fine-tuning. Shipped end-to-end systems from data pipeline to production API.",
"location": "Bangalore", "country": "India", "years_of_experience": 8.2,
"current_title": "Lead AI Engineer", "current_company": "Razorpay",
"current_company_size": "1001-5000", "current_industry": "Fintech",
"notice_period_days": 15, "open_to_relocation": True,
"expected_salary_range_inr_lpa": {"min": 55, "max": 75},
"recruiter_response_rate": 0.85, "days_since_last_active": 2,
"platform_trust_score": 0.9, "interview_completion_rate": 0.8,
},
"career_history": [
{"company": "Razorpay", "title": "Lead AI Engineer", "start_date": "2022-03-01", "end_date": None,
"duration_months": 38, "is_current": True, "industry": "Fintech", "company_size": "1001-5000",
"description": "Built and owned the hybrid vector search ranking system serving 50M+ users. Designed the two-tower retrieval pipeline with FAISS, implemented learning-to-rank with XGBoost, and fine-tuned BGE embeddings for domain-specific retrieval. Reduced search latency by 40% while improving NDCG@10 by 15%. Managed a team of 3 engineers."},
{"company": "Amazon", "title": "SDE II", "start_date": "2018-06-01", "end_date": "2022-02-28",
"duration_months": 45, "is_current": False, "industry": "E-commerce", "company_size": "10001+",
"description": "Worked on A9 product search ranking. Implemented query understanding features and built the candidate generation pipeline for product search. Used Word2Vec embeddings for product similarity and improved search relevance metrics."},
{"company": "Flipkart", "title": "SDE I", "start_date": "2016-07-01", "end_date": "2018-05-31",
"duration_months": 23, "is_current": False, "industry": "E-commerce", "company_size": "10001+",
"description": "Built data pipelines for the recommendation system. Worked on collaborative filtering and content-based recommendation algorithms using Python and Spark."}
],
"education": [{"institution": "IIT Bombay", "degree": "B.Tech", "field_of_study": "Computer Science",
"start_year": 2012, "end_year": 2016, "grade": "9.1 CGPA", "tier": "tier_1"}],
"skills": [{"name": "Python", "proficiency": "expert", "duration_months": 96},
{"name": "Machine Learning", "proficiency": "expert", "duration_months": 84},
{"name": "Vector Search", "proficiency": "advanced", "duration_months": 42},
{"name": "RAG", "proficiency": "advanced", "duration_months": 30},
{"name": "Embeddings", "proficiency": "expert", "duration_months": 60},
{"name": "FAISS", "proficiency": "advanced", "duration_months": 36},
{"name": "XGBoost", "proficiency": "advanced", "duration_months": 72},
{"name": "LLM Fine-tuning", "proficiency": "advanced", "duration_months": 24}],
"assessments": [{"name": "ML System Design", "score": 92, "completed": True}],
"endorsements_count": 28, "github_profile": {"has_public_repos": True, "public_repo_count": 12, "stargazers_total": 450}
},
{
"candidate_id": "CAND_0001002",
"profile": {
"anonymized_name": "LangChain Tourist", "headline": "AI Developer | LangChain, OpenAI, ChatGPT",
"summary": "Recently transitioned to AI development. Built multiple LangChain prototypes and RAG chatbots using OpenAI APIs. Passionate about generative AI.",
"location": "Mumbai", "country": "India", "years_of_experience": 4.5,
"current_title": "AI Developer", "current_company": "TCS",
"current_company_size": "10001+", "current_industry": "IT Services",
"notice_period_days": 90, "open_to_relocation": False,
"expected_salary_range_inr_lpa": {"min": 25, "max": 40},
"recruiter_response_rate": 0.15, "days_since_last_active": 180,
"platform_trust_score": 0.4, "interview_completion_rate": 0.0,
},
"career_history": [
{"company": "TCS", "title": "AI Developer", "start_date": "2024-01-01", "end_date": None,
"duration_months": 18, "is_current": True, "industry": "IT Services", "company_size": "10001+",
"description": "Building RAG chatbots using LangChain and OpenAI APIs for internal clients. Created a proof-of-concept document Q&A system using GPT-4 and Pinecone."},
{"company": "Infosys", "title": "Software Engineer", "start_date": "2021-07-01", "end_date": "2023-12-31",
"duration_months": 30, "is_current": False, "industry": "IT Services", "company_size": "10001+",
"description": "Java backend development for enterprise clients. Built REST APIs and maintained legacy systems."}
],
"education": [{"institution": "VTU", "degree": "B.E.", "field_of_study": "Information Science",
"start_year": 2017, "end_year": 2021, "grade": "7.2 CGPA", "tier": "tier_3"}],
"skills": [{"name": "Python", "proficiency": "intermediate", "duration_months": 42},
{"name": "LangChain", "proficiency": "advanced", "duration_months": 14},
{"name": "OpenAI API", "proficiency": "advanced", "duration_months": 14},
{"name": "RAG", "proficiency": "intermediate", "duration_months": 10},
{"name": "Pinecone", "proficiency": "intermediate", "duration_months": 8}],
"endorsements_count": 3, "github_profile": {"has_public_repos": False, "public_repo_count": 0, "stargazers_total": 0}
},
{
"candidate_id": "CAND_0001003",
"profile": {
"anonymized_name": "HR Manager Trap", "headline": "HR Manager | Recruitment, Talent Acquisition",
"summary": "Experienced HR manager with 12 years in talent acquisition and recruitment. Recently completed a Coursera course on AI.",
"location": "Delhi", "country": "India", "years_of_experience": 12.0,
"current_title": "HR Manager", "current_company": "Wipro",
"current_company_size": "10001+", "current_industry": "IT Services",
"notice_period_days": 60, "open_to_relocation": False,
"expected_salary_range_inr_lpa": {"min": 20, "max": 35},
"recruiter_response_rate": 0.9, "days_since_last_active": 1,
"platform_trust_score": 0.8, "interview_completion_rate": 0.5,
},
"career_history": [
{"company": "Wipro", "title": "HR Manager", "start_date": "2019-01-01", "end_date": None,
"duration_months": 77, "is_current": True, "industry": "IT Services", "company_size": "10001+",
"description": "Managing end-to-end recruitment for the AI/ML division. Hiring 50+ engineers per quarter. Completed Coursera AI for Everyone course."},
{"company": "Infosys", "title": "Talent Acquisition Lead", "start_date": "2014-01-01", "end_date": "2018-12-31",
"duration_months": 60, "is_current": False, "industry": "IT Services", "company_size": "10001+",
"description": "Led campus and lateral hiring for technology roles."}
],
"education": [{"institution": "Delhi University", "degree": "MBA", "field_of_study": "HR",
"start_year": 2012, "end_year": 2014, "grade": "7.5 CGPA", "tier": "tier_2"}],
"skills": [{"name": "Recruitment", "proficiency": "expert", "duration_months": 144},
{"name": "Machine Learning", "proficiency": "beginner", "duration_months": 2},
{"name": "Python", "proficiency": "beginner", "duration_months": 3}],
"endorsements_count": 45, "github_profile": {"has_public_repos": False, "public_repo_count": 0, "stargazers_total": 0}
},
{
"candidate_id": "CAND_0001004",
"profile": {
"anonymized_name": "Pre-LLM Veteran", "headline": "Staff ML Engineer | IR, Ranking, Recommendations",
"summary": "10 years in information retrieval and ranking. Built search engines at Microsoft and recommendation systems at Netflix before the LLM era. Deep expertise in classic IR: BM25, learning-to-rank, two-tower models, approximate nearest neighbor search.",
"location": "Pune", "country": "India", "years_of_experience": 10.5,
"current_title": "Staff ML Engineer", "current_company": "Sarvam AI",
"current_company_size": "51-200", "current_industry": "AI/ML",
"notice_period_days": 20, "open_to_relocation": True,
"expected_salary_range_inr_lpa": {"min": 60, "max": 85},
"recruiter_response_rate": 0.70, "days_since_last_active": 5,
"platform_trust_score": 0.85, "interview_completion_rate": 0.75,
},
"career_history": [
{"company": "Sarvam AI", "title": "Staff ML Engineer", "start_date": "2023-06-01", "end_date": None,
"duration_months": 36, "is_current": True, "industry": "AI/ML", "company_size": "51-200",
"description": "Leading the retrieval and ranking infrastructure for Indic language AI products. Built the hybrid retrieval pipeline combining sparse BM25 with dense BGE embeddings. Designed the evaluation framework with NDCG@10 and MRR metrics. Owns the entire ranking stack end-to-end."},
{"company": "Microsoft", "title": "Senior SDE", "start_date": "2018-04-01", "end_date": "2023-05-31",
"duration_months": 62, "is_current": False, "industry": "Technology", "company_size": "10001+",
"description": "Core contributor to Bing's document ranking pipeline. Implemented learning-to-rank features using LambdaMART. Built the query understanding module using transformer-based re-ranking. Optimized serving latency for 100+ billion document index."},
{"company": "Netflix", "title": "ML Engineer", "start_date": "2015-01-01", "end_date": "2018-03-31",
"duration_months": 39, "is_current": False, "industry": "Entertainment", "company_size": "10001+",
"description": "Built the recommendation ranking model for the home page. Worked on collaborative filtering, content-based filtering, and hybrid approaches. Improved click-through rate by 12% through better candidate generation and re-ranking."}
],
"education": [{"institution": "IIT Delhi", "degree": "M.Tech", "field_of_study": "Computer Science",
"start_year": 2013, "end_year": 2015, "grade": "9.4 CGPA", "tier": "tier_1"}],
"skills": [{"name": "Python", "proficiency": "expert", "duration_months": 126},
{"name": "Machine Learning", "proficiency": "expert", "duration_months": 120},
{"name": "Information Retrieval", "proficiency": "expert", "duration_months": 120},
{"name": "Learning to Rank", "proficiency": "expert", "duration_months": 84},
{"name": "Vector Search", "proficiency": "expert", "duration_months": 48},
{"name": "Embeddings", "proficiency": "expert", "duration_months": 72},
{"name": "FAISS", "proficiency": "advanced", "duration_months": 36},
{"name": "BM25", "proficiency": "expert", "duration_months": 96}],
"assessments": [{"name": "System Design", "score": 88, "completed": True}],
"endorsements_count": 52, "github_profile": {"has_public_repos": True, "public_repo_count": 18, "stargazers_total": 1200}
},
{
"candidate_id": "CAND_0001005",
"profile": {
"anonymized_name": "Graphic Designer", "headline": "Senior Graphic Designer | Adobe, Figma, AI Art",
"summary": "Creative professional with 7 years in graphic design. Recently started using Midjourney and DALL-E for AI art generation. Interested in AI roles.",
"location": "Hyderabad", "country": "India", "years_of_experience": 7.0,
"current_title": "Senior Graphic Designer", "current_company": "Accenture",
"current_company_size": "10001+", "current_industry": "IT Services",
"notice_period_days": 45, "open_to_relocation": False,
"expected_salary_range_inr_lpa": {"min": 15, "max": 25},
"recruiter_response_rate": 0.50, "days_since_last_active": 30,
"platform_trust_score": 0.6, "interview_completion_rate": 0.3,
},
"career_history": [
{"company": "Accenture", "title": "Senior Graphic Designer", "start_date": "2021-01-01", "end_date": None,
"duration_months": 53, "is_current": True, "industry": "IT Services", "company_size": "10001+",
"description": "Leading design projects for Fortune 500 clients. Using AI tools like Midjourney and DALL-E for rapid prototyping."},
{"company": "Ogilvy", "title": "Graphic Designer", "start_date": "2018-06-01", "end_date": "2020-12-31",
"duration_months": 31, "is_current": False, "industry": "Advertising", "company_size": "10001+",
"description": "Created visual campaigns for major brands."}
],
"education": [{"institution": "NID", "degree": "B.Des", "field_of_study": "Graphic Design",
"start_year": 2014, "end_year": 2018, "grade": "8.5 CGPA", "tier": "tier_1"}],
"skills": [{"name": "Adobe Creative Suite", "proficiency": "expert", "duration_months": 84},
{"name": "Figma", "proficiency": "advanced", "duration_months": 48},
{"name": "Machine Learning", "proficiency": "beginner", "duration_months": 6},
{"name": "Python", "proficiency": "beginner", "duration_months": 8}],
"endorsements_count": 35, "github_profile": {"has_public_repos": False, "public_repo_count": 0, "stargazers_total": 0}
},
]
def load_candidates_from_upload(file_obj) -> list[dict]:
"""Load candidates from uploaded file (JSONL, JSON array, or .jsonl.gz)."""
if file_obj is None:
return []
path = Path(file_obj.name if hasattr(file_obj, "name") else str(file_obj))
suffix = path.suffix.lower()
# JSON array format
if suffix == ".json":
with open(file_obj.name, "r", encoding="utf-8") as f:
data = json.load(f)
if isinstance(data, list):
return data
elif isinstance(data, dict):
return [data]
return []
# JSONL format
candidates = []
import gzip
if suffix == ".gz":
opener = gzip.open
else:
opener = open
with opener(file_obj.name, "rt", encoding="utf-8") as f:
for line in f:
line = line.strip()
if not line or line.startswith("#") or line.startswith("candidate_id,"):
continue
try:
candidates.append(json.loads(line))
except json.JSONDecodeError:
continue
return candidates
def run_ranking(candidates: list[dict]) -> tuple[str, str, str, str]:
"""Run the full V7 pipeline on candidates. Returns (csv_text, stats, logs, csv_path)."""
t0 = time.time()
logs = []
if not candidates:
return "", "No candidates loaded.", "", ""
n = len(candidates)
log = f"[sandbox] Loaded {n} candidates"
logs.append(log)
print(log)
if n > 100:
log = f"[sandbox] WARNING: {n} > 100 limit. Truncating to first 100."
logs.append(log)
print(log)
candidates = candidates[:100]
# ββ Step 1: Feature Extraction βββββββββββββββββββββββββββββββββββββ
t1 = time.time()
log = "[sandbox] Step 1/3: Extracting V7 features..."
logs.append(log)
print(log)
rows = []
texts = []
for c in candidates:
row = extract_features_v7(c)
rows.append(row)
texts.append(schema.unified_text_blob(c))
log = f"[sandbox] Extracted features for {len(rows)} candidates in {time.time()-t1:.1f}s"
logs.append(log)
print(log)
# ββ Step 2: TF-IDF + SVD Embeddings ββββββββββββββββββββββββββββββββ
t2 = time.time()
log = "[sandbox] Step 2/3: Computing TF-IDF+SVD embeddings..."
logs.append(log)
print(log)
jd = get_jd()
expanded_ideal = get_expanded_text(jd.ideal_text)
all_texts = texts + [expanded_ideal]
n_components = min(100, len(all_texts) - 1)
embedder = emb_mod.TfidfSvdEmbedder(n_components=max(n_components, 2))
embedder.fit(all_texts)
doc_emb = embedder.transform(all_texts)[:len(texts)]
sims = embedder.similarity_to_query(doc_emb, expanded_ideal)
for row, sim in zip(rows, sims):
row["embedding_sim"] = float(sim)
df = pd.DataFrame(rows)
log = f"[sandbox] Embeddings computed in {time.time()-t2:.1f}s ({len(df.columns)} features)"
logs.append(log)
print(log)
# ββ Step 3: Scoring + Ranking ββββββββββββββββββββββββββββββββββββββ
t3 = time.time()
log = "[sandbox] Step 3/3: Scoring and ranking..."
logs.append(log)
print(log)
# Print hiring intent
try:
intent = get_intent()
log = (f"[sandbox] Hiring Intent: {intent.philosophy} | "
f"ownership={intent.ownership_expectation:.2f} | "
f"need={intent.primary_need} | team={intent.team_context}")
logs.append(log)
print(log)
except Exception as e:
log = f"[sandbox] Warning: hiring intent failed: {e}"
logs.append(log)
# Compute scores
elite = scoring.elite_score_vec(df)
final = scoring.final_score_vec(df)
df = df.assign(elite_score=elite.values, raw_score=final.values)
top_n = min(TOP_N, len(df))
top = df.sort_values("raw_score", ascending=False).head(top_n).copy()
# Sigmoid stretch
raw = top["raw_score"].values.astype(float)
raw_min, raw_max = raw.min(), raw.max()
if raw_max > raw_min:
norm = (raw - raw_min) / (raw_max - raw_min)
stretched = 1.0 / (1.0 + np.exp(-10.0 * (norm - 0.5)))
final_scores = 0.52 + 0.47 * stretched
else:
final_scores = np.full(top_n, 0.75)
top["score"] = final_scores
top["score"] = top["score"].round(6)
top = top.sort_values(["score", "candidate_id"], ascending=[False, True]).reset_index(drop=True)
top["rank"] = top.index + 1
honeypots_in_top = int(top["is_honeypot"].sum())
scores = top["score"].values
# Generate reasoning
log = f"[sandbox] Generating reasoning for {top_n} candidates..."
logs.append(log)
print(log)
t_reason = time.time()
out_rows = []
for _, row in top.iterrows():
cid = row["candidate_id"]
candidate = json.loads(row["_candidate_json"])
disq_reasons = json.loads(row["_disq_reasons"])
beh_evidence = json.loads(row["_behaviour_evidence"])
narrative_suspicious = json.loads(row.get("narrative_suspicious", "[]"))
feat = {
"current_title": row["current_title"],
"current_company": row["current_company"],
"years_of_experience": row["years_of_experience"],
"skill_coverage": float(row.get("skill_coverage", 0)),
"impact_magnitude": float(row.get("impact_magnitude", 0)),
"ownership_hierarchy": float(row.get("ownership_hierarchy", 0)),
"evaluation_experience": float(row.get("evaluation_experience", 0)),
"pre_llm_months": float(row.get("pre_llm_months", 0)),
"evidence_strength": float(row.get("evidence_strength", 0)),
"notice_period_days": beh_evidence.get("notice_period_days", 45),
"days_since_active": beh_evidence.get("days_since_active", 90),
"recruiter_response_rate": beh_evidence.get("recruiter_response_rate", 0.3),
"disqualifier_reasons": disq_reasons,
"tier5_signature": float(row.get("tier5_signature", 0)),
"behavioral_twin_penalty": float(row.get("behavioral_twin_penalty", 1.0)),
"langchain_only_penalty": float(row.get("langchain_only_penalty", 1.0)),
"closed_source_penalty": float(row.get("closed_source_penalty", 1.0)),
"pre_llm_x_ownership": float(row.get("pre_llm_x_ownership", 0)),
"salary_compatibility": float(row.get("salary_compatibility", 0.7)),
"is_honeypot": bool(row.get("is_honeypot", False)),
"assessment_signal": float(row.get("assessment_signal", 0.5)),
"endorsement_signal": float(row.get("endorsement_signal", 0.5)),
"education_tier": float(row.get("education_tier", 0.5)),
"cross_validation": float(row.get("cross_validation", 0.5)),
"_candidate_json": row.get("_candidate_json", "{}"),
"_beh_twin_evidence": row.get("_beh_twin_evidence", "{}"),
"_langchain_evidence": row.get("_langchain_evidence", "{}"),
"_closed_source_evidence": row.get("_closed_source_evidence", "{}"),
"_salary_evidence": row.get("_salary_evidence", "{}"),
"_assessment_evidence": row.get("_assessment_evidence", "{}"),
"_endorsement_evidence": row.get("_endorsement_evidence", "{}"),
"_education_evidence": row.get("_education_evidence", "{}"),
}
text = reasoning.generate(cid, candidate, feat, narrative_suspicious)
out_rows.append({
"candidate_id": cid,
"rank": int(row["rank"]),
"score": f'{row["score"]:.6f}',
"reasoning": text,
})
log = f"[sandbox] Reasoning generated in {time.time()-t_reason:.1f}s"
logs.append(log)
print(log)
# Build CSV string
csv_buf = StringIO()
writer = csv.DictWriter(csv_buf, fieldnames=["candidate_id", "rank", "score", "reasoning"])
writer.writeheader()
writer.writerows(out_rows)
csv_text = csv_buf.getvalue()
# Write temp file for download
tmp_path = Path(tempfile.gettempdir()) / "sandbox_submission.csv"
with open(tmp_path, "w", encoding="utf-8") as f:
f.write(csv_text)
# Stats
total_time = time.time() - t0
unique_scores = len(set(scores))
non_mono = sum(1 for i in range(len(scores) - 1) if scores[i] < scores[i + 1])
honeypots_total = int(df["is_honeypot"].sum())
stats = (
f"**Pipeline Stats**\n\n"
f"| Metric | Value |\n|---|---|\n"
f"| Candidates input | {n} |\n"
f"| Candidates ranked | {top_n} |\n"
f"| Honeypots detected | {honeypots_total} / {n} |\n"
f"| Honeypots in top {top_n} | {honeypots_in_top} |\n"
f"| Score range | {scores[0]:.6f} β {scores[-1]:.6f} |\n"
f"| Unique scores | {unique_scores} / {top_n} |\n"
f"| Monotonicity violations | {non_mono} |\n"
f"| Total time | {total_time:.1f}s |\n"
f"| Under 5-min budget | {'β
Yes' if total_time < 300 else 'β No'} |\n"
)
top5_table = "| Rank | Candidate ID | Score | Title | Company |\n|---|---|---|---|---|\n"
for _, row in top.head(5).iterrows():
top5_table += f"| {int(row['rank'])} | {row['candidate_id']} | {row['score']:.6f} | {row['current_title']} | {row['current_company']} |\n"
stats += f"\n\n**Top 5 Candidates**\n\n{top5_table}"
log = f"[sandbox] Total time: {total_time:.1f}s β CSV ready for download"
logs.append(log)
print(log)
# Return pandas DataFrame for Gradio Dataframe component
result_df = pd.DataFrame(out_rows)
return result_df, stats, "\n".join(logs), str(tmp_path)
def on_upload_and_rank(file_obj):
"""Handle file upload + rank button."""
if file_obj is None:
return pd.DataFrame(columns=["candidate_id", "rank", "score", "reasoning"]), "", "", None
try:
candidates = load_candidates_from_upload(file_obj)
if not candidates:
return pd.DataFrame(columns=["candidate_id", "rank", "score", "reasoning"]), "No valid candidates found in file. Use JSONL (one JSON per line) or JSON array format.", "", None
return run_ranking(candidates)
except Exception as e:
tb = traceback.format_exc()
return pd.DataFrame(columns=["candidate_id", "rank", "score", "reasoning"]), f"ERROR: {e}\n\n{tb}", "", None
def on_demo_rank():
"""Run with built-in 5-candidate demo set."""
return run_ranking(SAMPLE_CANDIDATES)
# ββ Gradio UI βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ
with gr.Blocks(
title="V7 Elite Ranker β Redrob Hackathon Sandbox",
theme=gr.themes.Soft(),
) as demo:
gr.Markdown(
"""
# V7 Elite Ranker β Hackathon Sandbox
**Redrob AI Hackathon:** India Runs Data & AI Challenge
**Role:** Senior AI Engineer β Founding Team
Upload a candidate file (β€100 candidates) or use the built-in demo to see the ranking pipeline in action.
The sandbox runs the **exact same code** that runs at Stage 3 reproduction.
"""
)
with gr.Row():
with gr.Column(scale=1):
gr.Markdown("### Input")
file_input = gr.File(
label="Upload Candidates (.jsonl, .json, .jsonl.gz)",
file_types=[".json", ".jsonl", ".gz"],
)
gr.Markdown(
"""
**Accepted formats:**
- `.jsonl` β one JSON object per line
- `.json` β JSON array of candidate objects
- `.jsonl.gz` β gzipped JSONL
- Max **100 candidates** (sandbox limit)
**Or click "Run Demo" to use 5 built-in test candidates** (1 elite, 1 LangChain tourist, 1 HR trap, 1 pre-LLM veteran, 1 graphic designer).
"""
)
with gr.Row():
btn_demo = gr.Button("Run Demo (5 Candidates)", variant="secondary")
btn_run = gr.Button("Rank Uploaded File", variant="primary")
gr.Markdown("### Logs")
log_output = gr.Textbox(label="Pipeline Logs", lines=12, interactive=False, max_lines=20)
with gr.Column(scale=2):
gr.Markdown("### Output")
stats_output = gr.Markdown(label="Stats")
csv_output = gr.Dataframe(
label="Ranked Candidates (submission.csv)",
headers=["candidate_id", "rank", "score", "reasoning"],
interactive=False,
)
download_btn = gr.File(
label="Download submission.csv",
interactive=False,
)
gr.Markdown(
"""
---
### Pipeline Architecture
```
candidates.jsonl β Feature Extraction (55 features) β TF-IDF+SVD Embeddings
β Intent-Aware Scoring (Impact 42% + Ownership 33% + Search 15% + Behaviour 10%)
β 4 Additive Boosts β 5 Multiplicative Penalties β Honeypot Γ0.01
β 13-term Tiebreaker β Sigmoid Stretch β Evidence-Graph Reasoning
β submission.csv
```
**Constraints met:** CPU-only | No network | <5 min | β€16 GB RAM | Deterministic
"""
)
# Wire up events
btn_demo.click(fn=on_demo_rank, outputs=[csv_output, stats_output, log_output, download_btn])
btn_run.click(fn=on_upload_and_rank, inputs=[file_input], outputs=[csv_output, stats_output, log_output, download_btn])
if __name__ == "__main__":
demo.launch(
server_name="0.0.0.0",
server_port=7860,
) |