diff --git a/yanfr-lab-open-hr/.env.example b/yanfr-lab-open-hr/.env.example
new file mode 100644
index 0000000..4d89586
--- /dev/null
+++ b/yanfr-lab-open-hr/.env.example
@@ -0,0 +1,4 @@
+HRFLOW_API_KEY=your_api_key_here
+HRFLOW_USER_EMAIL=your_email_here
+HRFLOW_SOURCE_KEY=your_source_key_here
+HRFLOW_BOARD_KEY=your_board_key_here
diff --git a/yanfr-lab-open-hr/README.md b/yanfr-lab-open-hr/README.md
new file mode 100644
index 0000000..a2b8b06
--- /dev/null
+++ b/yanfr-lab-open-hr/README.md
@@ -0,0 +1,59 @@
+# Open HR avec HrFlow.ai
+
+> Intelligent candidate sourcing with auto-generated questionnaires and HrFlow.ai-powered scoring.
+
+## What it does
+
+Open HR is an AI sourcing agent built for recruiters. Given a job title and requirements, it:
+
+1. **Auto-generates a weighted questionnaire** from the job description (skills, experience, education, stability)
+2. **Indexes the job** via HrFlow.ai Job Indexing API
+3. **Scores all candidates** in the source using HrFlow.ai Scoring API with a custom algorithm key
+4. **Combines AI scores with questionnaire weights** to produce precise decimal rankings
+5. **Allows refinement** via natural language feedback loop
+
+## HrFlow.ai APIs used
+
+- `POST /v1/job/indexing` — Index job description to enable scoring
+- `GET /v1/profiles/scoring` — Score all profiles against the job using algorithm key
+- `GET /v1/profiles/searching` — Fallback profile search
+- `POST /v1/profile/indexing` — Index candidate profiles
+
+## How to run
+
+### Prerequisites
+
+- Python 3.11+
+
+### Setup
+
+```bash
+# Install dependencies
+pip install -r requirements.txt
+
+# Copy environment variables
+cp .env.example .env
+# Fill in your HrFlow API keys in .env
+
+# Start the app
+uvicorn web_app_fr_v2:app --host 0.0.0.0 --port 8002
+```
+
+### Environment variables
+
+| Variable | Required | Description |
+|----------|----------|-------------|
+| `HRFLOW_API_KEY` | Yes | HrFlow.ai API secret key |
+| `HRFLOW_USER_EMAIL` | Yes | HrFlow.ai account email |
+| `HRFLOW_SOURCE_KEY` | Yes | HrFlow.ai source key |
+| `HRFLOW_BOARD_KEY` | Yes | HrFlow.ai board key |
+
+## Screenshots
+
+
+
+## Team
+
+- **Yan** — Full-stack development & HrFlow.ai integration
+- **Leo** — HrFlow.ai API integration & scoring algorithm
+- **Aike** — Development & HrFlow.ai integration
diff --git a/yanfr-lab-open-hr/app.json b/yanfr-lab-open-hr/app.json
new file mode 100644
index 0000000..171322d
--- /dev/null
+++ b/yanfr-lab-open-hr/app.json
@@ -0,0 +1,16 @@
+{
+ "$schema": "../../schemas/app.schema.json",
+ "name": "Open HR avec HrFlow.ai",
+ "description": "AI agent that auto-generates a weighted questionnaire from any job description, scores and ranks candidates using HrFlow.ai Scoring API, and refines results through a natural language feedback loop.",
+ "credentials": {
+ "source_keys": ["c1039b004a836e42427b9c89309ad99cf9b6a73c"],
+ "board_keys": ["c99d70c3a062c2f99ca78fce23b89fb4f53119ef"],
+ "algorithm_key": "b1ebac4c62fa96e06206f4433b95ae69674891ff"
+ },
+ "settings": {
+ "team_name": "yanfr-lab",
+ "theme_color": "#83D9DC",
+ "custom_filters": [],
+ "filters": []
+ }
+}
diff --git a/yanfr-lab-open-hr/assets/preview.png b/yanfr-lab-open-hr/assets/preview.png
new file mode 100755
index 0000000..dad7221
Binary files /dev/null and b/yanfr-lab-open-hr/assets/preview.png differ
diff --git a/yanfr-lab-open-hr/database.py b/yanfr-lab-open-hr/database.py
new file mode 100644
index 0000000..9dc9688
--- /dev/null
+++ b/yanfr-lab-open-hr/database.py
@@ -0,0 +1,74 @@
+import sqlite3
+import json
+from datetime import datetime
+
+DB_NAME = "sourcing.db"
+
+def init_db():
+ conn = sqlite3.connect(DB_NAME)
+ cursor = conn.cursor()
+ cursor.execute('''
+ CREATE TABLE IF NOT EXISTS searches (
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
+ query TEXT,
+ created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
+ )
+ ''')
+ cursor.execute('''
+ CREATE TABLE IF NOT EXISTS candidates (
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
+ search_id INTEGER,
+ name TEXT,
+ headline TEXT,
+ location TEXT,
+ profile_url TEXT,
+ source_platform TEXT,
+ hrflow_profile_key TEXT,
+ score REAL,
+ rank INTEGER,
+ raw_json TEXT,
+ created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
+ FOREIGN KEY(search_id) REFERENCES searches(id)
+ )
+ ''')
+ conn.commit()
+ conn.close()
+
+def save_search(query: str) -> int:
+ conn = sqlite3.connect(DB_NAME)
+ cursor = conn.cursor()
+ cursor.execute("INSERT INTO searches (query) VALUES (?)", (query,))
+ search_id = cursor.lastrowid
+ conn.commit()
+ conn.close()
+ return search_id
+
+def save_candidate(search_id: int, candidate: dict):
+ conn = sqlite3.connect(DB_NAME)
+ cursor = conn.cursor()
+ cursor.execute('''
+ INSERT INTO candidates (search_id, name, headline, location, profile_url,
+ source_platform, hrflow_profile_key, score, rank, raw_json)
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
+ ''', (
+ search_id, candidate['name'], candidate.get('headline'), candidate.get('location'),
+ candidate['profile_url'], candidate.get('source_platform', 'github'),
+ candidate.get('hrflow_profile_key'), candidate.get('score', 0),
+ candidate.get('rank'), json.dumps(candidate)
+ ))
+ conn.commit()
+ conn.close()
+
+def get_recent_searches(limit=20):
+ conn = sqlite3.connect(DB_NAME)
+ conn.row_factory = sqlite3.Row
+ cursor = conn.cursor()
+ cursor.execute('''
+ SELECT s.*, (SELECT COUNT(*) FROM candidates WHERE search_id = s.id) as candidate_count
+ FROM searches s ORDER BY created_at DESC LIMIT ?
+ ''', (limit,))
+ rows = [dict(row) for row in cursor.fetchall()]
+ conn.close()
+ return rows
+
+init_db()
\ No newline at end of file
diff --git a/yanfr-lab-open-hr/hrflow/__init__.py b/yanfr-lab-open-hr/hrflow/__init__.py
new file mode 100644
index 0000000..e69de29
diff --git a/yanfr-lab-open-hr/hrflow/client.py b/yanfr-lab-open-hr/hrflow/client.py
new file mode 100644
index 0000000..2eda25a
--- /dev/null
+++ b/yanfr-lab-open-hr/hrflow/client.py
@@ -0,0 +1,537 @@
+"""
+HrFlow AI Client — 12 APIs,使用已验证的正确端点。
+
+Base URL: https://api.hrflow.ai/v1
+
+APIs:
+ 1. Parsing API — POST /v1/text/parsing + POST /v1/job/indexing
+ 2. Tagging API — POST /v1/text/tagging
+ 3. Embedding API — POST /v1/profile/embedding
+ 4. Searching API — GET /v1/profiles/searching
+ 5. Matching API — GET /v1/profiles/scoring (无阈值宽召回)
+ 6. Scoring API — GET /v1/profiles/scoring (有阈值精排)
+ 7. Grading API — GET /v1/profile/grading
+ 8. Reasoning API — GET /v1/profile/revealing
+ 9. Signals API — POST /v1/profile/events
+ 10. Upskilling API — GET /v1/jobs/searching (skill gap)
+ 11. Data Studio — POST /v1/profile/indexing (batch sync)
+ 12. UI Studio — GET /v1/profile/indexing (card data)
+"""
+
+from __future__ import annotations
+
+import io
+import json
+import logging
+from typing import Any, Optional
+
+import httpx
+from tenacity import (
+ retry, retry_if_exception_type,
+ stop_after_attempt, wait_exponential,
+)
+
+from config import config
+
+logger = logging.getLogger(__name__)
+TIMEOUT = httpx.Timeout(30.0)
+BASE = "https://api.hrflow.ai/v1"
+
+
+def _auth_headers() -> dict:
+ return {
+ "X-API-KEY": config.HRFLOW_API_KEY,
+ "X-USER-EMAIL": config.HRFLOW_USER_EMAIL,
+ }
+
+
+def _check(resp: httpx.Response, label: str) -> dict:
+ if resp.status_code not in (200, 201):
+ logger.error("[HrFlow/%s] HTTP %s: %s", label, resp.status_code, resp.text[:300])
+ resp.raise_for_status()
+ body = resp.json()
+ if body.get("code") not in (200, 201, None):
+ raise RuntimeError(f"[HrFlow/{label}] error {body.get('code')}: {body.get('message')}")
+ return body
+
+
+def _get(path: str, params: dict) -> dict:
+ with httpx.Client(timeout=TIMEOUT) as c:
+ r = c.get(f"{BASE}{path}", headers={**_auth_headers(), "Content-Type": "application/json"}, params=params)
+ return _check(r, path)
+
+
+def _post_json(path: str, payload: dict) -> dict:
+ with httpx.Client(timeout=TIMEOUT) as c:
+ r = c.post(f"{BASE}{path}", headers={**_auth_headers(), "Content-Type": "application/json"}, json=payload)
+ return _check(r, path)
+
+
+def _post_multipart(path: str, data: dict, files: dict) -> dict:
+ with httpx.Client(timeout=TIMEOUT) as c:
+ r = c.post(f"{BASE}{path}", headers=_auth_headers(), data=data, files=files)
+ return _check(r, path)
+
+
+class HrFlowClient:
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 1. PARSING API
+ # ─────────────────────────────────────────────────────────────────────────
+
+ @retry(stop=stop_after_attempt(3), wait=wait_exponential(min=2, max=10),
+ retry=retry_if_exception_type((httpx.HTTPError, RuntimeError)))
+ def parse_job_from_text(self, board_key: str, job_text: str, job_title: str) -> dict:
+ """
+ Parsing API (Job) — POST /v1/job/indexing
+ 直接将 JD 写入 Board 获取 job_key(text_parsing 为可选增强)。
+ """
+ import uuid
+ job_key = str(uuid.uuid4())
+
+ # 尝试用 text/parsing 提取技能(若账号未开通则跳过)
+ skills = []
+ try:
+ parse_resp = _post_json("/text/parsing", {"text": f"{job_title}\n\n{job_text}"})
+ skills = parse_resp.get("data", {}).get("skills", [])
+ except Exception as e:
+ logger.info("[Parsing] text/parsing skipped (not available on this plan): %s", e)
+
+ job_payload = {
+ "board_key": board_key,
+ "job": {
+ "key": job_key,
+ "name": job_title,
+ "description": job_text,
+ "skills": skills,
+ "tags": [],
+ "location": {"text": ""},
+ "sections": [{"name": "description", "title": "Description", "description": job_text}],
+ },
+ }
+ index_resp = _post_json("/job/indexing", job_payload)
+ # 返回 job_key 合并到响应中
+ index_resp["_job_key"] = job_key
+ return index_resp
+
+ @retry(stop=stop_after_attempt(3), wait=wait_exponential(min=2, max=10),
+ retry=retry_if_exception_type((httpx.HTTPError, RuntimeError)))
+ def parse_profile_from_url(
+ self, resume_url: str, source_key: str, texts: Optional[list[str]] = None
+ ) -> dict:
+ """
+ Parsing API (Profile) — POST /v1/profile/parsing/file
+ 下载简历文件后以 multipart 上传。
+ 若 URL 不可访问,退回到直接索引一个最小 profile。
+ texts: GitHub README 等补充文本,拼接后附加到简历内容。
+ """
+ # 尝试下载简历
+ file_bytes: Optional[bytes] = None
+ filename = "resume.pdf"
+ try:
+ with httpx.Client(timeout=httpx.Timeout(15.0)) as c:
+ dl = c.get(resume_url, follow_redirects=True)
+ if dl.status_code == 200:
+ file_bytes = dl.content
+ # 从 URL 取文件名
+ filename = resume_url.split("/")[-1] or "resume.pdf"
+ except Exception as e:
+ logger.warning("[Parsing] Resume download failed (%s): %s", resume_url, e)
+
+ if file_bytes:
+ # 若有补充文本,拼成额外 txt 一起上传
+ extra = ("\n\n".join(texts) if texts else "").encode("utf-8")
+ if extra:
+ with httpx.Client(timeout=TIMEOUT) as c:
+ r = c.post(
+ f"{BASE}/profile/parsing/file",
+ headers=_auth_headers(),
+ data={"source_key": source_key, "sync_parsing": "1"},
+ files={
+ "file": (filename, file_bytes, "application/pdf"),
+ "file2": ("github.txt", extra, "text/plain"),
+ },
+ )
+ return _check(r, "/profile/parsing/file")
+ else:
+ return _post_multipart(
+ "/profile/parsing/file",
+ data={"source_key": source_key, "sync_parsing": "1"},
+ files={"file": (filename, file_bytes, "application/pdf")},
+ )
+ else:
+ # 无法下载 → 直接用文本索引一个 profile
+ combined_text = f"{resume_url}\n\n" + ("\n\n".join(texts) if texts else "")
+ return self._index_profile_from_text(source_key, combined_text)
+
+ def _index_profile_from_text(self, source_key: str, text: str) -> dict:
+ """兜底:当简历 URL 不可访问时,直接用文本解析后索引 profile。"""
+ import uuid
+ profile_key = str(uuid.uuid4())
+ # 先用 text/parsing 提取结构化信息
+ try:
+ parsed = _post_json("/text/parsing", {"text": text[:3000]})
+ skills = parsed.get("data", {}).get("skills", [])
+ except Exception:
+ skills = []
+
+ payload = {
+ "source_key": source_key,
+ "profile": {
+ "key": profile_key,
+ "skills": skills,
+ "experiences": [],
+ "educations": [],
+ "languages": [],
+ "certifications": [],
+ "info": {"full_name": "Unknown", "summary": text[:500]},
+ "sections": [{"name": "raw", "title": "Raw", "description": text[:2000]}],
+ },
+ }
+ resp = _post_json("/profile/indexing", payload)
+ resp["_profile_key"] = profile_key
+ return resp
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 2. TAGGING API — POST /v1/text/tagging
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def tag_profile(self, profile_key: str, source_key: str) -> dict:
+ """
+ Tagging API — 取 profile 文本后调用 POST /v1/text/tagging
+ 用 hrflow-skills tagger 打 HR 分类标签,统一技能词汇表。
+ """
+ # 先取 profile 内容
+ try:
+ p = _get("/profile/indexing", {"source_key": source_key, "profile_key": profile_key})
+ sections = p.get("data", {}).get("profile", {}).get("sections", [])
+ text = " ".join(s.get("description", "") for s in sections)[:2000] or profile_key
+ except Exception:
+ text = profile_key
+
+ try:
+ return _post_json("/text/tagging", {
+ "text": text,
+ "algorithm_key": "tagger-hrflow-skills",
+ "top_n": 10,
+ "output_lang": "en",
+ })
+ except Exception as e:
+ logger.info("[Tagging] text/tagging skipped (not available): %s", e)
+ return {"code": 200, "data": {"tags": []}}
+
+ def tag_job(self, board_key: str, job_key: str) -> dict:
+ """Tagging API — 对 Job 文本打 HR 技能标签。"""
+ try:
+ j = _get("/job/indexing", {"board_key": board_key, "job_key": job_key})
+ desc = j.get("data", {}).get("job", {}).get("description", "") or job_key
+ except Exception:
+ desc = job_key
+
+ try:
+ return _post_json("/text/tagging", {
+ "text": desc[:2000],
+ "algorithm_key": "tagger-hrflow-skills",
+ "top_n": 10,
+ "output_lang": "en",
+ })
+ except Exception as e:
+ logger.info("[Tagging] text/tagging skipped (not available): %s", e)
+ return {"code": 200, "data": {"tags": []}}
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 3. EMBEDDING API — POST /v1/profile/embedding
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def embed_profile(self, profile_key: str, source_key: str) -> dict:
+ """
+ Embedding API — POST /v1/profile/embedding
+ 生成语义向量,为 Searching / Scoring 提供语义相似度基础。
+ """
+ return _post_json("/profile/embedding", {
+ "source_key": source_key,
+ "profile_key": profile_key,
+ })
+
+ def embed_job(self, board_key: str, job_key: str) -> dict:
+ """Embedding API — POST /v1/job/embedding 生成 Job 语义向量。"""
+ return _post_json("/job/embedding", {
+ "board_key": board_key,
+ "job_key": job_key,
+ })
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 4. SEARCHING API — GET /v1/profiles/searching
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def search_profiles(
+ self,
+ source_keys: list[str],
+ query: str,
+ filters: Optional[dict] = None,
+ page: int = 1,
+ limit: int = 50,
+ ) -> dict:
+ """
+ Searching API — GET /v1/profiles/searching
+ 关键词+语义混合检索,作为 Scoring 前的轻量预过滤,降低成本。
+ """
+ params: dict[str, Any] = {
+ "source_keys": json.dumps(source_keys),
+ "text_keywords": query,
+ "page": page,
+ "limit": limit,
+ "sort_by": "searching",
+ "order_by": "desc",
+ }
+ if filters:
+ params.update(filters)
+ return _get("/profiles/searching", params)
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 5. MATCHING API — GET /v1/profiles/scoring (无阈值宽召回)
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def match_profiles_for_job(
+ self,
+ job_key: str,
+ board_key: str,
+ source_keys: list[str],
+ limit: int = 100,
+ ) -> dict:
+ """
+ Matching API — GET /v1/profiles/scoring (无 score_threshold)
+ 粗粒度宽匹配,扩大召回范围后再由 Scoring API 精排。
+ """
+ return _get("/profiles/scoring", {
+ "job_key": job_key,
+ "board_key": board_key,
+ "source_keys": json.dumps(source_keys),
+ "limit": limit,
+ "sort_by": "scoring",
+ "order_by": "desc",
+ })
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 6. SCORING API — GET /v1/profiles/scoring (核心)
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def score_profiles_for_job(
+ self,
+ job_key: str,
+ board_key: str,
+ source_keys: list[str],
+ score_threshold: float = 0.6,
+ limit: int = 20,
+ page: int = 1,
+ use_memory: bool = False,
+ ) -> dict:
+ """
+ Scoring API — GET /v1/profiles/scoring
+ 返回 predictions[1] 概率分(0–1),按 score desc 排序。
+ use_memory=True 激活在线微调(需先有 Signals)。
+ """
+ params: dict[str, Any] = {
+ "job_key": job_key,
+ "board_key": board_key,
+ "source_keys": json.dumps(source_keys),
+ "limit": limit,
+ "page": page,
+ "sort_by": "scoring",
+ "order_by": "desc",
+ }
+ if use_memory:
+ params["use_memory_context"] = 1
+ return _get("/profiles/scoring", params)
+
+ def index_job_sync(self, board_key: str, job_title: str) -> str:
+ """
+ Indexing API (Job) — POST /v1/job/indexing
+ Crée un job dans le Board et retourne son job_key.
+ Approche Leo : job minimal pour déclencher le scoring.
+ """
+ import time as _time
+ payload = {
+ "board_key": board_key,
+ "name": job_title[:50],
+ "reference": f"ref-{int(_time.time())}",
+ "summary": job_title,
+ "location": {"text": "Remote", "lat": None, "lng": None},
+ "sections": [{"name": "description", "title": "Job Description", "description": job_title}],
+ }
+ resp = _post_json("/job/indexing", payload)
+ return resp["data"]["key"]
+
+ def score_profiles_with_algo(
+ self,
+ job_key: str,
+ board_key: str,
+ source_keys: list[str],
+ limit: int = 100,
+ ) -> dict:
+ """
+ Scoring API — GET /v1/profiles/scoring avec algorithm_key (approche Leo).
+ Retourne profiles + predictions[][1] = score IA (0–1).
+ """
+ params: dict[str, Any] = {
+ "job_key": job_key,
+ "board_key": board_key,
+ "source_keys": json.dumps(source_keys),
+ "algorithm_key": "b1ebac4c62fa96e06206f4433b95ae69674891ff",
+ "limit": limit,
+ "sort_by": "scoring",
+ "order_by": "desc",
+ }
+ with httpx.Client(timeout=TIMEOUT) as c:
+ r = c.get(f"{BASE}/profiles/scoring",
+ headers={**_auth_headers(), "Content-Type": "application/json"},
+ params=params)
+ if r.status_code != 200:
+ logger.error("[HrFlow/scoring-algo] HTTP %s: %s", r.status_code, r.text[:300])
+ return {}
+ return r.json()
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 7. GRADING API — GET /v1/profile/grading
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def grade_profile(
+ self, profile_key: str, source_key: str, job_key: str, board_key: str
+ ) -> dict:
+ """
+ Grading API — GET /v1/profile/grading
+ 对单个 profile-job 对进行二次重排,输出等级(A/B/C)。
+ 在 Scoring Top-N 之后调用,进一步精细筛选。
+ """
+ return _get("/profile/grading", {
+ "source_key": source_key,
+ "profile_key": profile_key,
+ "board_key": board_key,
+ "job_key": job_key,
+ })
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 8. REASONING API — GET /v1/profile/revealing
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def get_reasoning(
+ self, profile_key: str, source_key: str, job_key: str, board_key: str
+ ) -> dict:
+ """
+ Reasoning API — GET /v1/profile/revealing
+ 返回评分依据的自然语言解释,满足 EU AI Act 可解释性要求。
+ """
+ return _get("/profile/revealing", {
+ "source_key": source_key,
+ "profile_key": profile_key,
+ "board_key": board_key,
+ "job_key": job_key,
+ })
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 9. SIGNALS API — POST /v1/profile/events
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def send_signal(
+ self,
+ profile_key: str,
+ source_key: str,
+ job_key: str,
+ board_key: str,
+ event_type: str,
+ rating: Optional[float] = None,
+ ) -> dict:
+ """
+ Signals API — POST /v1/profile/events
+ 将招聘官动作(面试/拒绝/offer)回写给 HrFlow,
+ 驱动在线微调,使后续评分越来越准确。
+ """
+ payload: dict[str, Any] = {
+ "source_key": source_key,
+ "profile_key": profile_key,
+ "board_key": board_key,
+ "job_key": job_key,
+ "type": event_type,
+ }
+ if rating is not None:
+ payload["rating"] = rating
+ return _post_json("/profile/events", payload)
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 10. UPSKILLING API — GET /v1/jobs/searching (skill gap)
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def get_upskilling(
+ self, profile_key: str, source_key: str, job_key: str, board_key: str
+ ) -> dict:
+ """
+ Upskilling API — GET /v1/profile/indexing (比对技能缺口)
+ 获取 profile 和 job 的技能列表,计算 gap(job 有而 profile 缺少的技能)。
+ 结果写入候选人输出卡片的"技能缺口"字段。
+ """
+ # 取 profile 技能
+ try:
+ p_resp = _get("/profile/indexing", {
+ "source_key": source_key,
+ "profile_key": profile_key,
+ })
+ p_skills = {s.get("name", "").lower()
+ for s in p_resp.get("data", {}).get("profile", {}).get("skills", [])}
+ except Exception:
+ p_skills = set()
+
+ # 取 job 技能
+ try:
+ j_resp = _get("/job/indexing", {
+ "board_key": board_key,
+ "job_key": job_key,
+ })
+ j_skills = [s for s in j_resp.get("data", {}).get("job", {}).get("skills", [])]
+ except Exception:
+ j_skills = []
+
+ # 计算缺口
+ gaps = [s for s in j_skills if s.get("name", "").lower() not in p_skills]
+ return {"code": 200, "data": {"upskilling": gaps}}
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 11. DATA STUDIO — POST /v1/profile/indexing (批量同步)
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def sync_profiles_to_source(self, source_key: str, profiles: list[dict]) -> dict:
+ """
+ Data Studio — 批量将候选人 profile 写入 HrFlow Source。
+ 生产环境中由 Data Studio 的 200+ 连接器(Greenhouse/Lever/Workday)替代。
+ """
+ results = []
+ for p in profiles:
+ try:
+ r = _post_json("/profile/indexing", {
+ "source_key": source_key,
+ "profile": p,
+ })
+ results.append({"key": p.get("key"), "status": "ok"})
+ except Exception as e:
+ results.append({"key": p.get("key"), "status": "error", "message": str(e)})
+ return {"code": 200, "data": {"synced": results}}
+
+ def list_sources(self) -> dict:
+ """Data Studio — GET /v1/sources 列出所有数据源。"""
+ return _get("/sources", {})
+
+ # ─────────────────────────────────────────────────────────────────────────
+ # 12. UI STUDIO — GET /v1/profile/indexing (recruiter card data)
+ # ─────────────────────────────────────────────────────────────────────────
+
+ def get_ui_profile_card(
+ self, profile_key: str, source_key: str, job_key: str, board_key: str
+ ) -> dict:
+ """
+ UI Studio — GET /v1/profile/indexing
+ 返回结构化 profile 数据,供前端招聘官卡片组件渲染。
+ 包含匿名化字段、技能高亮和匹配指示器。
+ """
+ return _get("/profile/indexing", {
+ "source_key": source_key,
+ "profile_key": profile_key,
+ })
diff --git a/yanfr-lab-open-hr/hrflow_client.py b/yanfr-lab-open-hr/hrflow_client.py
new file mode 100644
index 0000000..40002f8
--- /dev/null
+++ b/yanfr-lab-open-hr/hrflow_client.py
@@ -0,0 +1,93 @@
+import os
+import httpx
+import hashlib
+import time
+from dotenv import load_dotenv
+import asyncio
+
+load_dotenv()
+
+API_KEY = os.getenv("HRFLOW_API_KEY")
+USER_EMAIL = os.getenv("HRFLOW_USER_EMAIL")
+SOURCE_KEY = os.getenv("HRFLOW_SOURCE_KEY")
+BOARD_KEY = os.getenv("HRFLOW_BOARD_KEY")
+
+HEADERS = {"X-API-KEY": API_KEY, "X-USER-EMAIL": USER_EMAIL, "Content-Type": "application/json"}
+BASE_URL = "https://api.hrflow.ai/v1"
+
+
+async def index_job(query: str) -> str:
+ async with httpx.AsyncClient() as client:
+ import time
+ payload = {
+ "board_key": BOARD_KEY,
+ "name": query[:50],
+ "reference": f"ref-{int(time.time())}",
+ "summary": query,
+ "location": {
+ "text": "Remote", # 必须包含此字段
+ "lat": None,
+ "lng": None
+ },
+ "sections": [
+ {
+ "name": "description",
+ "title": "Job Description",
+ "description": query
+ }
+ ]
+ }
+
+ print(f"正在请求 HrFlow API... Board Key: {BOARD_KEY}")
+
+ resp = await client.post(
+ f"{BASE_URL}/job/indexing",
+ json=payload,
+ headers=HEADERS
+ )
+
+ if resp.status_code not in [200, 201]:
+ print("--- HrFlow API 返回了错误 ---")
+ print(f"状态码: {resp.status_code}")
+ print(f"详细错误内容: {resp.text}")
+ print("----------------------------")
+ raise Exception(f"HrFlow 报错: {resp.text}")
+
+ return resp.json()["data"]["key"]
+
+async def index_profile(candidate: dict) -> str:
+ ref = hashlib.md5(candidate['profile_url'].encode()).hexdigest()[:16]
+ payload = {"source_key": SOURCE_KEY, "reference": ref, "info": {"full_name": candidate['name'], "location": {"text": candidate['location']}, "urls": [{"type": "from_resume", "url": candidate['profile_url']}], "summary": candidate['headline'], "picture": candidate['avatar_url']}, "text": candidate['raw_text']}
+ async with httpx.AsyncClient() as client:
+ resp = await client.post(f"{BASE_URL}/profile/indexing", json=payload, headers=HEADERS)
+ return resp.json()["data"]["key"] if resp.status_code != 409 else resp.json()["data"]["key"]
+
+
+async def score_profiles(job_key: str, limit: int = 10):
+ params = {
+ "board_key": BOARD_KEY,
+ "job_key": job_key,
+ "source_keys": f'["{SOURCE_KEY}"]', # 必须是这种 ["key"] 的格式
+ "algorithm_key": 'b1ebac4c62fa96e06206f4433b95ae69674891ff',
+ "limit": limit,
+ "sort_by": "scoring",
+ "order_by": "desc"
+ }
+
+ async with httpx.AsyncClient(timeout=30.0) as client:
+ resp = await client.get(f"{BASE_URL}/profiles/scoring", params=params, headers=HEADERS)
+
+ if resp.status_code == 200:
+ # --- 关键:返回整个 JSON 字典 ---
+ return resp.json()
+ else:
+ print(f"❌ API报错: {resp.status_code}")
+ return {}
+
+async def explain_profile(job_key: str, profile_key: str) -> str:
+ params = {"board_key": BOARD_KEY, "job_key": job_key, "source_key": SOURCE_KEY, "profile_key": profile_key}
+ async with httpx.AsyncClient() as client:
+ try:
+ resp = await client.get(f"{BASE_URL}/profile/explaining", params=params, headers=HEADERS, timeout=2.0)
+ return resp.json()["data"]["explanation"]
+ except: return ""
\ No newline at end of file
diff --git a/yanfr-lab-open-hr/main.py b/yanfr-lab-open-hr/main.py
new file mode 100644
index 0000000..d0cf76e
--- /dev/null
+++ b/yanfr-lab-open-hr/main.py
@@ -0,0 +1,79 @@
+import asyncio
+from fastapi import FastAPI, Request
+from fastapi.responses import HTMLResponse
+from fastapi.staticfiles import StaticFiles
+import database as db
+import sourcing
+import hrflow_client as hrflow
+import uvicorn
+import os
+app = FastAPI()
+if not os.path.exists("static"): os.makedirs("static")
+app.mount("/static", StaticFiles(directory="static"), name="static")
+
+@app.get("/", response_class=HTMLResponse)
+async def read_index():
+ with open("static/index.html", encoding="utf-8") as f: return f.read()
+
+@app.post("/search")
+@app.post("/search")
+async def perform_search(request: Request):
+ try:
+ data = await request.json()
+ query = data.get("query")
+ if not query:
+ return {"status": "error", "message": "请输入搜索关键词"}
+
+ print(f"🔍 正在为岗位创建索引: {query}")
+ # 1. 创建 Job (这一步会让 Board Key 计数)
+ job_key = await hrflow.index_job(query)
+
+ # 2. 给 AI 一点反应时间 (因为没有了爬虫过程,建议保留几秒等待)
+ print("⏳ 正在请求 HrFlow AI 计算匹配得分...")
+ await asyncio.sleep(5)
+
+ # 3. 直接获取评分结果 (搜索 Source 中已有的所有人)
+ full_response = await hrflow.score_profiles(job_key)
+
+ # 解析数据
+ api_data = full_response.get('data', {})
+ profiles = api_data.get('profiles', [])
+ predictions = api_data.get('predictions', [])
+
+ results = []
+ for i, profile in enumerate(profiles):
+ info = profile.get('info', {})
+
+ # 提取姓名
+ display_name = info.get('full_name') or profile.get('reference') or "未知候选人"
+
+ # 提取分数
+ score_percent = "0%"
+ if i < len(predictions) and len(predictions[i]) > 1:
+ score_percent = f"{round(predictions[i][1] * 100, 2)}%"
+
+ # 提取链接
+ link = "#"
+ for u in info.get('urls', []):
+ if isinstance(u, dict) and u.get('url'):
+ link = u['url']
+ break
+
+ results.append({
+ "name": display_name,
+ "score": score_percent,
+ "url": link,
+ "summary": info.get('summary') or "暂无个人简介"
+ })
+
+ print(f"✅ 成功找到 {len(results)} 位匹配的人才")
+ return {"status": "success", "results": results}
+
+ except Exception as e:
+ print(f"❌ 搜索出错: {str(e)}")
+ return {"status": "error", "message": str(e)}
+@app.get("/history")
+async def get_history(): return db.get_recent_searches()
+
+if __name__ == "__main__":
+ uvicorn.run(app, host="0.0.0.0", port=8000)
\ No newline at end of file
diff --git a/yanfr-lab-open-hr/requirements.txt b/yanfr-lab-open-hr/requirements.txt
new file mode 100644
index 0000000..368eae4
--- /dev/null
+++ b/yanfr-lab-open-hr/requirements.txt
@@ -0,0 +1,4 @@
+fastapi==0.111.0
+uvicorn==0.29.0
+httpx==0.27.0
+python-dotenv==1.0.1
\ No newline at end of file
diff --git a/yanfr-lab-open-hr/sourcing.py b/yanfr-lab-open-hr/sourcing.py
new file mode 100644
index 0000000..8da3dfb
--- /dev/null
+++ b/yanfr-lab-open-hr/sourcing.py
@@ -0,0 +1,46 @@
+import httpx
+import asyncio
+import re
+
+
+async def search_candidates(query: str, n: int = 30) -> list[dict]:
+ query_lower = query.lower()
+ role = "developer"
+ skills = []
+
+ if "python" in query_lower: skills.append("python")
+ if "react" in query_lower or "frontend" in query_lower:
+ skills.append("react")
+ role = "frontend-developer"
+
+ search_queries = [f"{role} {' '.join(skills)}", f"{role} engineer"]
+ all_candidates = {}
+
+ async with httpx.AsyncClient(timeout=10.0) as client:
+ for sq in search_queries:
+ try:
+ url = f"https://api.github.com/search/users?q={sq}+in:bio+type:user&per_page=15"
+ headers = {"Accept": "application/vnd.github+json", "User-Agent": "SourcingAgent/1.0"}
+ resp = await client.get(url, headers=headers)
+ if resp.status_code != 200: continue
+
+ for item in resp.json().get("items", []):
+ login = item["login"]
+ if login in all_candidates: continue
+ await asyncio.sleep(0.1)
+ u_resp = await client.get(f"https://api.github.com/users/{login}", headers=headers)
+ if u_resp.status_code == 200:
+ u = u_resp.json()
+ all_candidates[login] = {
+ "name": u.get("name") or u.get("login"),
+ "headline": (u.get("bio") or "")[:200],
+ "location": u.get("location") or "Remote",
+ "profile_url": u["html_url"],
+ "source_platform": "github",
+ "avatar_url": u.get("avatar_url"),
+ "company": u.get("company") or "",
+ "raw_text": f"{u.get('name')} {u.get('bio')} {u.get('location')}"
+ }
+ except: pass
+ if len(all_candidates) >= n: break
+ return list(all_candidates.values())[:n]
\ No newline at end of file
diff --git a/yanfr-lab-open-hr/static/index_openhr.html b/yanfr-lab-open-hr/static/index_openhr.html
new file mode 100644
index 0000000..312272b
--- /dev/null
+++ b/yanfr-lab-open-hr/static/index_openhr.html
@@ -0,0 +1,691 @@
+
+
+
+
+
+Open HR
+
+
+
+
+
+
+
+
+
+
Chargement
+
+
+
+
+
+
+
+
+ Open HR
+
+
Sourcing intelligent · Questionnaire adaptatif · Pondération par critère · Boucle d'affinement IA
+
+
+
+
1
Poste
+
2
Questionnaire
+
3
Affinement
+
4
Résultats
+
+
+
+
+
+
+
+
+
📋 Décrivez le poste à pourvoir
+
+
+
+
+
+
+
+
+
+
+
+
🎯 Précisez vos critères de recrutement
+
+ Répondez aux questions et ajustez le curseur de pondération à droite de chaque question.
+ Plus le curseur est à droite, plus ce critère pèse dans le scoring final.
+
+ 🤖 Agent Open HR : Plus de 10 candidats correspondent à vos critères actuels.
+ Pour affiner la sélection, décrivez un critère supplémentaire ou une préférence spécifique.
+
+ Ex : « doit maîtriser Kubernetes », « minimum 7 ans d'expérience », « master requis », « profil stable »…
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
diff --git a/yanfr-lab-open-hr/web_app_fr_v2.py b/yanfr-lab-open-hr/web_app_fr_v2.py
new file mode 100644
index 0000000..d2dd40c
--- /dev/null
+++ b/yanfr-lab-open-hr/web_app_fr_v2.py
@@ -0,0 +1,962 @@
+"""
+AI Agent avec OpenClaw — Version 2
+Nouveautés v2 :
+ - Question de stabilité des employés dans le questionnaire général
+ - Poids dynamiques par question (slider côté frontend → envoyés au backend)
+ - Évaluation de la stabilité des candidats (durée moyenne de tenure)
+Port 8002.
+"""
+
+import json
+import os
+import re
+import sys
+import time
+import uuid
+from collections import Counter
+from datetime import datetime
+from typing import Optional
+
+from fastapi import FastAPI, HTTPException
+from fastapi.middleware.cors import CORSMiddleware
+from fastapi.responses import FileResponse
+from fastapi.staticfiles import StaticFiles
+from pydantic import BaseModel
+
+sys.path.insert(0, os.path.dirname(__file__))
+
+from config import config
+from hrflow.client import HrFlowClient
+
+app = FastAPI(title="AI Agent avec OpenClaw v2")
+app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"])
+app.mount("/static", StaticFiles(directory="static"), name="static")
+
+sessions: dict = {}
+hrflow = HrFlowClient()
+
+
+# ─── Request Models ───────────────────────────────────────────────────────────
+
+class StartRequest(BaseModel):
+ job_title: str
+ job_description: str
+
+class SubmitAnswersRequest(BaseModel):
+ session_id: str
+ answers: dict
+ weights: dict = {} # { question_id: slider_value (1–5) }
+
+class RefineRequest(BaseModel):
+ session_id: str
+ refinement: str
+
+
+# ─── Catégorie de poste ────────────────────────────────────────────────────────
+
+CATEGORY_SIGNALS = {
+ "frontend": ["frontend","front-end","前端","react","vue","angular","svelte","css","html",
+ "webpack","vite","nextjs","nuxt","ui/ux","界面","interface","intégration"],
+ "backend": ["backend","back-end","后端","服务端","api","server","django","flask","fastapi",
+ "spring","rails","express","nestjs","grpc","microservice","微服务","serveur"],
+ "fullstack": ["fullstack","full-stack","全栈","full stack","complet","polyvalent"],
+ "mobile": ["mobile","移动端","ios","android","swift","kotlin","react native","flutter","xamarin","application mobile"],
+ "devops": ["devops","sre","运维","infrastructure","kubernetes","k8s","docker","ci/cd","pipeline",
+ "terraform","ansible","aws","gcp","azure","cloud","nuage"],
+ "ml": ["machine learning","deep learning","机器学习","深度学习","algorithme","ai engineer",
+ "pytorch","tensorflow","nlp","cv","computer vision","llm","大模型","ia","intelligence artificielle"],
+ "data": ["data engineer","data analyst","数据工程","数据分析","etl","spark","flink",
+ "hadoop","kafka","数仓","数据仓库","bi","tableau","airflow","données","analyse"],
+ "security": ["security","安全","penetration","渗透","ctf","soc","siem","sécurité","cybersécurité"],
+ "embedded": ["embedded","嵌入式","firmware","rtos","stm32","fpga","驱动","内核","embarqué","temps réel"],
+}
+
+SPECIFIC_QUESTIONS: dict[str, list] = {
+ "frontend": [
+ {"id":"s1","question":"Frameworks front-end requis (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["React","Vue.js","Angular","Svelte","JS/TS natif"],"weight":5},
+ {"id":"s2","question":"Niveau TypeScript exigé",
+ "type":"single_choice","options":["Maîtrise obligatoire","Expérience souhaitée","Atout","Non requis"],"weight":4},
+ {"id":"s3","question":"Solutions CSS / styles (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Tailwind CSS","CSS Modules","Styled-components","SCSS/LESS","Libre"],"weight":3},
+ {"id":"s4","question":"Expérience en optimisation des performances (Core Web Vitals, etc.)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ {"id":"s5","question":"Expérience en outillage front-end (build tools, bundlers)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ {"id":"s6","question":"Participation à un Design System ou bibliothèque de composants",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ {"id":"s7","question":"Expérience SSR / SSG (Next.js, Nuxt.js, etc.)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ ],
+ "backend": [
+ {"id":"s1","question":"Langages de programmation principaux (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Python","Java","Go","Node.js","C++","Rust","PHP","Ruby"],"weight":5},
+ {"id":"s2","question":"Bases de données requises (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["PostgreSQL","MySQL","MongoDB","Redis","Elasticsearch","ClickHouse","Libre"],"weight":4},
+ {"id":"s3","question":"Expérience microservices / systèmes distribués",
+ "type":"single_choice","options":["Obligatoire (3+ ans)","Expérience souhaitée","Atout","Non requis"],"weight":4},
+ {"id":"s4","question":"Style d'API requis (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["RESTful","GraphQL","gRPC","WebSocket","Libre"],"weight":3},
+ {"id":"s5","question":"Expérience en messagerie (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Kafka","RabbitMQ","RocketMQ","Redis Pub/Sub","Non requis"],"weight":3},
+ {"id":"s6","question":"Expérience cloud",
+ "type":"single_choice","options":["Cloud-native obligatoire","Usage de base suffisant","Atout","Non requis"],"weight":3},
+ {"id":"s7","question":"Conception de systèmes haute disponibilité / haute charge",
+ "type":"single_choice","options":["Obligatoire (million RPS)","Échelle moyenne","Atout","Non requis"],"weight":5},
+ ],
+ "fullstack": [
+ {"id":"s1","question":"Préférence stack front-end (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["React","Vue.js","Angular","Next.js","Nuxt","Libre"],"weight":4},
+ {"id":"s2","question":"Préférence stack back-end (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Node.js","Python","Java","Go","PHP","Libre"],"weight":4},
+ {"id":"s3","question":"Répartition front / back du poste",
+ "type":"single_choice","options":["Front dominant (70/30)","Équilibré (50/50)","Back dominant (30/70)","Flexible"],"weight":3},
+ {"id":"s4","question":"Expérience de développement produit complet en autonomie",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ {"id":"s5","question":"Bases de données (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["PostgreSQL","MySQL","MongoDB","Redis","Libre"],"weight":3},
+ {"id":"s6","question":"Expérience DevOps / déploiement (Docker, CI/CD)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ ],
+ "mobile": [
+ {"id":"s1","question":"Plateformes cibles (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["iOS (Swift/ObjC)","Android (Kotlin/Java)","React Native","Flutter","HarmonyOS"],"weight":5},
+ {"id":"s2","question":"Expérience de publication App Store / Google Play",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ {"id":"s3","question":"Optimisation des performances (mémoire, rendu, batterie)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ {"id":"s4","question":"Développement de modules natifs / hybride",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ {"id":"s5","question":"CI/CD et automatisation des builds / releases",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ ],
+ "devops": [
+ {"id":"s1","question":"Plateformes cloud utilisées (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["AWS","GCP","Azure","Alibaba Cloud","Tencent Cloud","OVH"],"weight":5},
+ {"id":"s2","question":"Expérience Kubernetes",
+ "type":"single_choice","options":["Expert (production)","Expérience en utilisation","Notions suffisantes","Non requis"],"weight":5},
+ {"id":"s3","question":"Outils IaC (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Terraform","Ansible","Pulumi","CloudFormation","Non requis"],"weight":4},
+ {"id":"s4","question":"Monitoring / alerting (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Prometheus/Grafana","ELK Stack","Datadog","Libre"],"weight":3},
+ {"id":"s5","question":"Gestion de clusters à grande échelle (100+ nœuds)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ {"id":"s6","question":"Compétences en développement / scripting",
+ "type":"single_choice","options":["Dev requis (Python/Go)","Scripts shell suffisants","Non requis"],"weight":3},
+ ],
+ "ml": [
+ {"id":"s1","question":"Frameworks ML requis (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["PyTorch","TensorFlow","JAX","scikit-learn","XGBoost","Libre"],"weight":5},
+ {"id":"s2","question":"Profil recherche",
+ "type":"single_choice","options":["Publications (NeurIPS/ICML/ACL…)","Publications souhaitées","Ingénierie avant tout","Non requis"],"weight":4},
+ {"id":"s3","question":"Expérience en déploiement de modèles",
+ "type":"single_choice","options":["Obligatoire (TorchServe/Triton/ONNX)","Atout","Non requis"],"weight":4},
+ {"id":"s4","question":"Expérience LLM / grands modèles",
+ "type":"single_choice","options":["Obligatoire (Fine-tuning/RAG/Agent)","Atout","Non requis"],"weight":4},
+ {"id":"s5","question":"Volume de données traité",
+ "type":"single_choice","options":["Échelle TB+","Échelle GB suffisant","Non requis"],"weight":3},
+ {"id":"s6","question":"Entraînement sur clusters GPU",
+ "type":"single_choice","options":["Obligatoire (distribué)","Multi-GPU mono-machine","Atout","Non requis"],"weight":4},
+ ],
+ "data": [
+ {"id":"s1","question":"Stack data requise (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Spark","Flink","Hive","Kafka","Airflow","dbt","Libre"],"weight":5},
+ {"id":"s2","question":"Entrepôts de données (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["ClickHouse","BigQuery","Redshift","Snowflake","Hudi/Iceberg","Libre"],"weight":4},
+ {"id":"s3","question":"Volumétrie des données",
+ "type":"single_choice","options":["Échelle PB","Échelle TB","Échelle GB","Libre"],"weight":4},
+ {"id":"s4","question":"Modélisation data warehouse (ODS/DW/DM)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ {"id":"s5","question":"Outils BI (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Tableau","Power BI","Superset","Metabase","Non requis"],"weight":3},
+ {"id":"s6","question":"Langages requis (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Python","SQL (expert)","Scala","Java","Libre"],"weight":4},
+ ],
+ "security": [
+ {"id":"s1","question":"Domaines de sécurité (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Tests d'intrusion","Audit de code","Développement sécurisé (SDL)","SOC/Réponse incident","Sécurité cloud","Libre"],"weight":5},
+ {"id":"s2","question":"Découverte de vulnérabilités / CVE publiés",
+ "type":"single_choice","options":["Obligatoire (CVE publiés préférés)","Atout","Non requis"],"weight":4},
+ {"id":"s3","question":"Outils de sécurité (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Burp Suite","Metasploit","Nmap","Wireshark","Outils maison"],"weight":3},
+ {"id":"s4","question":"Conformité réglementaire (ISO 27001, RGPD, etc.)",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":3},
+ ],
+ "embedded": [
+ {"id":"s1","question":"Domaines embarqués (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Développement drivers","RTOS","BSP","FPGA","Firmware"],"weight":5},
+ {"id":"s2","question":"Plateformes matérielles (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["ARM Cortex-M","ARM Cortex-A","RISC-V","x86","DSP","Libre"],"weight":4},
+ {"id":"s3","question":"Protocoles de communication (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["SPI/I2C/UART","CAN","Ethernet","BLE/WiFi","Libre"],"weight":3},
+ {"id":"s4","question":"Développement noyau Linux",
+ "type":"single_choice","options":["Obligatoire","Atout","Non requis"],"weight":4},
+ ],
+ "general": [
+ {"id":"s1","question":"Niveau de spécialisation technique",
+ "type":"single_choice","options":["Expert (spécialité pointue)","Généraliste (large spectre)","Manager technique","Libre"],"weight":4},
+ {"id":"s2","question":"Taille d'équipe / projet préférée",
+ "type":"single_choice","options":["Grande entreprise (100+)","Équipe moyenne (20-100)","Startup / petite équipe (<20)","Libre"],"weight":3},
+ {"id":"s3","question":"Contributions open-source",
+ "type":"single_choice","options":["Obligatoire (100+ stars)","Atout","Non requis"],"weight":3},
+ {"id":"s4","question":"Expérience management d'équipe",
+ "type":"single_choice","options":["Obligatoire (5+ personnes)","Petite équipe suffisant","Non requis"],"weight":4},
+ {"id":"s5","question":"Documentation technique / conférences",
+ "type":"single_choice","options":["Obligatoire (blog/talks)","Atout","Non requis"],"weight":2},
+ {"id":"s6","question":"Collaboration cross-équipes",
+ "type":"single_choice","options":["Très important","Important","Neutre","Peu important"],"weight":3},
+ ],
+}
+
+# ── Nouvelle question stabilité (ajoutée aux questions générales) ──────────────
+STABILITY_QUESTION = {
+ "id": "g_stab",
+ "question": "Préférence en matière de stabilité des employés",
+ "type": "single_choice",
+ "options": [
+ "Stabilité forte (durée moyenne ≥ 3 ans par poste)",
+ "Stabilité modérée (1–3 ans par poste)",
+ "Mobilité acceptée (CDD, freelance, missions courtes)",
+ "Sans critère de stabilité",
+ ],
+ "weight": 3,
+}
+
+GENERAL_QUESTIONS = [
+ {"id":"g1","question":"Mode de travail souhaité",
+ "type":"single_choice","options":["100% télétravail","Hybride (partiel télétravail)","Présentiel","Flexible"],"weight":3},
+ {"id":"g2","question":"Niveau d'études minimum",
+ "type":"single_choice","options":["Sans critère","Bac+2","Licence (grandes écoles préférées)","Licence","Master","Doctorat"],"weight":3},
+ {"id":"g3","question":"Années d'expérience minimum",
+ "type":"single_choice","options":["Sans critère","1 an et +","3 ans et +","5 ans et +","8 ans et +","10 ans et +"],"weight":5},
+ {"id":"g4","question":"Langues requises (plusieurs choix possibles)",
+ "type":"multiple_choice","options":["Français (langue maternelle)","Anglais (lu/écrit)","Anglais (courant oral)","Espagnol","Autre"],"weight":3},
+ {"id":"g5","question":"Fourchette de salaire mensuel brut (indicatif)",
+ "type":"single_choice",
+ "options":["< 2 500 €","2 500–4 000 €","4 000–6 000 €","6 000–8 500 €","8 500 € +","À négocier"],"weight":1},
+ STABILITY_QUESTION, # ← nouvelle question v2
+]
+
+
+# ─── Générateur de questionnaire ──────────────────────────────────────────────
+
+def _detect_category(title: str, description: str) -> str:
+ text = (title + " " + description).lower()
+ scores: dict[str, int] = {}
+ for cat, keywords in CATEGORY_SIGNALS.items():
+ score = sum(1 for kw in keywords if kw in text)
+ if score:
+ scores[cat] = score
+ if not scores:
+ return "general"
+ return max(scores, key=scores.get)
+
+
+def _extract_mentioned_techs(description: str) -> list[str]:
+ ALL_TECHS = [
+ "React","Vue","Angular","Next.js","Nuxt","Svelte","TypeScript","JavaScript",
+ "Python","Java","Go","Golang","Rust","C++","C#","PHP","Ruby","Swift","Kotlin",
+ "Django","Flask","FastAPI","Spring","Rails","Express","NestJS","Laravel",
+ "PostgreSQL","MySQL","MongoDB","Redis","Elasticsearch","ClickHouse","SQLite",
+ "AWS","GCP","Azure","Kubernetes","Docker","Terraform","Ansible","CI/CD","GitHub Actions",
+ "Kafka","RabbitMQ","Spark","Flink","Airflow","dbt","Hadoop",
+ "PyTorch","TensorFlow","scikit-learn","LLM","RAG","GPT","BERT",
+ "GraphQL","gRPC","REST","WebSocket","Prometheus","Grafana","ELK",
+ "Flutter","React Native","iOS","Android","SwiftUI",
+ ]
+ found = []
+ desc_lower = description.lower()
+ for tech in ALL_TECHS:
+ if tech.lower() in desc_lower and tech not in found:
+ found.append(tech)
+ return found[:8]
+
+
+def _generate_questionnaire(job_title: str, job_description: str) -> dict:
+ category = _detect_category(job_title, job_description)
+ specific_qs = SPECIFIC_QUESTIONS.get(category, SPECIFIC_QUESTIONS["general"])
+
+ mentioned = _extract_mentioned_techs(job_description)
+ extra_qs = []
+ if mentioned:
+ extra_qs.append({
+ "id": "s_dyn",
+ "question": "Parmi les technologies mentionnées dans la fiche de poste, lesquelles sont indispensables ? (plusieurs choix possibles)",
+ "type": "multiple_choice",
+ "options": mentioned + ["Toutes sont des atouts"],
+ "weight": 5,
+ })
+
+ return {
+ "general": GENERAL_QUESTIONS,
+ "specific": extra_qs + specific_qs,
+ "_category": category,
+ "_mentioned_techs": mentioned,
+ }
+
+
+# ─── Stabilité des candidats ──────────────────────────────────────────────────
+
+def _average_tenure_years(profile: dict) -> float:
+ """Calcule la durée moyenne (en années) par poste dans le profil."""
+ total_years = float(profile.get("experiences_duration") or 0)
+ n_exp = len(profile.get("experiences", []))
+ if total_years <= 0 or n_exp == 0:
+ return 0.0
+ return round(total_years / n_exp, 1)
+
+
+# ─── Interprétation des réponses ──────────────────────────────────────────────
+
+def _interpret_answers(questionnaire: dict, answers: dict, user_weights: dict = None) -> dict:
+ """user_weights : { qid: int(1-5) } — overrides question's default weight."""
+ q_map: dict[str, dict] = {}
+ for q in questionnaire.get("general", []) + questionnaire.get("specific", []):
+ q_map[q["id"]] = q
+
+ required_skills: list[str] = []
+ preferred_skills: list[str] = []
+ min_years = 0
+ max_years = 0
+ education_level = "none"
+ work_mode = "any"
+ languages: list[str] = []
+ stability_pref = "none" # "high" | "moderate" | "flexible" | "none"
+
+ def effective_weight(qid: str, default: int) -> int:
+ if user_weights and qid in user_weights:
+ return int(user_weights[qid])
+ return default
+
+ for qid, answer in answers.items():
+ q = q_map.get(qid, {})
+ q_text = q.get("question", "").lower()
+ weight = effective_weight(qid, q.get("weight", 3))
+ ans_list = answer if isinstance(answer, list) else [answer] if answer else []
+
+ # Stabilité
+ if "stabilité" in q_text:
+ for a in ans_list:
+ a_lower = a.lower()
+ if "≥ 3" in a or "forte" in a_lower:
+ stability_pref = "high"
+ elif "1–3" in a or "modérée" in a_lower:
+ stability_pref = "moderate"
+ elif "mobilité" in a_lower or "acceptée" in a_lower:
+ stability_pref = "flexible"
+ continue
+
+ # Mode de travail
+ if "mode de travail" in q_text or "télétravail" in q_text:
+ mode_map = {"100% télétravail": "remote", "hybride": "hybrid", "présentiel": "onsite"}
+ for k, v in mode_map.items():
+ if any(k in a.lower() for a in ans_list):
+ work_mode = v
+ break
+
+ # Niveau d'études
+ elif "études" in q_text or "niveau" in q_text:
+ edu_map = {"doctorat": "phd", "master": "master", "licence": "bachelor", "bac+2": "none"}
+ for k, v in edu_map.items():
+ if any(k in a.lower() for a in ans_list):
+ education_level = v
+ break
+
+ # Années d'expérience
+ elif "expérience" in q_text and ("ans" in str(ans_list).lower() or "an" in str(ans_list).lower()):
+ for a in ans_list:
+ m = re.search(r"(\d+)", str(a))
+ if m:
+ min_years = max(min_years, int(m.group(1)))
+
+ # Langues
+ elif "langues" in q_text:
+ languages = [a for a in ans_list if "autre" not in a.lower() and "français" not in a.lower()]
+
+ # Technologies fiche de poste
+ elif "fiche de poste" in q_text or "mentionnées" in q_text:
+ for a in ans_list:
+ if "atout" in a.lower():
+ continue
+ if a and a not in required_skills:
+ required_skills.append(a)
+
+ # Compétences / outils — seuil weight dynamique
+ elif any(kw in q_text for kw in
+ ["framework","stack","langage","outil","plateforme","domaine","base de données",
+ "messagerie","protocole","matérielle","entrepôt"]):
+ clean = [a for a in ans_list if a and "libre" not in a.lower() and "non requis" not in a.lower()]
+ if weight >= 4:
+ required_skills.extend(c for c in clean if c not in required_skills)
+ elif weight >= 2:
+ preferred_skills.extend(c for c in clean if c not in preferred_skills)
+
+ # Questions binaires
+ elif any(kw in q_text for kw in ["obligatoire","expérience","participation","niveau","conception"]):
+ for a in ans_list:
+ if "obligatoire" in a.lower() or "maîtrise obligatoire" in a.lower():
+ concept = _extract_concept_fr(q_text)
+ if concept and concept not in required_skills:
+ required_skills.append(concept)
+ elif "atout" in a.lower():
+ concept = _extract_concept_fr(q_text)
+ if concept and concept not in preferred_skills:
+ preferred_skills.append(concept)
+
+ key_parts = []
+ if required_skills:
+ key_parts.append(", ".join(required_skills[:4]))
+ if min_years > 0:
+ key_parts.append(f"{min_years}+ ans d'expérience")
+ if education_level != "none":
+ key_parts.append({"bachelor":"Licence","master":"Master","phd":"Doctorat"}.get(education_level,"") + " requis")
+
+ return {
+ "required_skills": required_skills,
+ "preferred_skills": preferred_skills,
+ "min_experience_years": min_years,
+ "max_experience_years": max_years,
+ "education_level": education_level,
+ "work_mode": work_mode,
+ "languages": languages,
+ "stability_pref": stability_pref,
+ "key_requirements_summary": ", ".join(key_parts) if key_parts else "Profil polyvalent",
+ }
+
+
+def _extract_concept_fr(q_text: str) -> str:
+ CONCEPT_MAP = {
+ "typescript": "TypeScript", "performance": "Optimisation des performances",
+ "outillage": "Build tools", "ssr": "SSR/SSG", "design system": "Design System",
+ "microservice": "Microservices", "distribué": "Systèmes distribués", "haute disponibilité": "Haute disponibilité",
+ "déploiement de modèles": "Déploiement ML", "llm": "LLM/Grands modèles", "gpu": "Entraînement GPU",
+ "open-source": "Open-source", "management": "Management", "kubernetes": "Kubernetes",
+ "noyau linux": "Noyau Linux", "iac": "IaC",
+ }
+ for kw, label in CONCEPT_MAP.items():
+ if kw in q_text:
+ return label
+ return ""
+
+
+# ─── Scoring des candidats ────────────────────────────────────────────────────
+
+def _calculate_experience_years(profile: dict) -> float:
+ return round(float(profile.get("experiences_duration") or 0), 1)
+
+
+def _get_skills(profile: dict) -> set:
+ return {s.get("name", "").lower() for s in profile.get("skills", []) if s.get("name")}
+
+
+def _edu_level(profile: dict) -> str:
+ for edu in profile.get("educations", []):
+ t = (edu.get("title") or "").lower()
+ if any(k in t for k in ["phd","doctorate","doctorat"]): return "phd"
+ if any(k in t for k in ["master","msc","mba","mastère"]): return "master"
+ if any(k in t for k in ["bachelor","bsc","undergraduate","licence","ingénieur"]): return "bachelor"
+ return "none"
+
+
+_EDU_RANK = {"none": 0, "bachelor": 1, "master": 2, "phd": 3}
+_EDU_LABEL_FR = {"phd": "Doctorat", "master": "Master", "bachelor": "Licence", "none": "Non renseigné"}
+
+
+def _score_candidates(profiles: list, reqs: dict, user_weights: dict = None) -> list:
+ required = {s.lower() for s in reqs.get("required_skills", [])}
+ preferred = {s.lower() for s in reqs.get("preferred_skills", [])}
+ min_years = float(reqs.get("min_experience_years") or 0)
+ max_years = float(reqs.get("max_experience_years") or 99)
+ edu_req = _EDU_RANK.get(reqs.get("education_level", "none"), 0)
+ stability_pref = reqs.get("stability_pref", "none")
+
+ # Dimension multipliers from weight sliders (1–5 scale → 0.2–1.0)
+ uw = user_weights or {}
+ w_exp = uw.get("g3", 5) / 5.0
+ w_edu = uw.get("g2", 3) / 5.0
+ s_keys = [k for k in uw if k.startswith("s")]
+ w_skill = (sum(uw[k] for k in s_keys) / len(s_keys) / 5.0) if s_keys else 1.0
+ w_stab = uw.get("g_stab", 3) / 5.0
+
+ # Max achievable raw score (used for normalization)
+ max_possible = 50 * w_skill + 30 * w_exp + 20 * w_edu + 8 * w_stab
+
+ scored = []
+ for p in profiles:
+ skills = _get_skills(p)
+ years = _calculate_experience_years(p)
+ edu = _edu_level(p)
+ avg_tenure = _average_tenure_years(p)
+ info = p.get("info", {})
+ name = (info.get("full_name")
+ or f"{info.get('first_name','')} {info.get('last_name','')}".strip()
+ or p.get("key", "")[:12])
+
+ raw = 0.0
+
+ # Compétences — 50 pts × w_skill
+ skill_pts = 0.0
+ if required:
+ matched_req = required & skills
+ skill_pts += (len(matched_req) / len(required)) * 50
+ else:
+ skill_pts += 30
+ if preferred:
+ pref_hit = preferred & skills
+ skill_pts += (len(pref_hit) / len(preferred)) * 20
+ raw += skill_pts * w_skill
+
+ # Expérience — 30 pts × w_exp
+ exp_pts = 0.0
+ if min_years > 0:
+ if years >= min_years: exp_pts += 30
+ elif years >= min_years * 0.75: exp_pts += 15
+ elif years >= min_years * 0.5: exp_pts += 5
+ else:
+ exp_pts += 30
+ if max_years < 99 and years > max_years + 5:
+ exp_pts -= 8
+ raw += exp_pts * w_exp
+
+ # Formation — 20 pts × w_edu
+ edu_pts = 0.0
+ cand_edu_rank = _EDU_RANK.get(edu, 0)
+ if cand_edu_rank >= edu_req: edu_pts += 20
+ elif cand_edu_rank == edu_req - 1: edu_pts += 10
+ raw += edu_pts * w_edu
+
+ # Stabilité — ±8 pts × w_stab
+ stability_note = ""
+ stab_pts = 0.0
+ if stability_pref == "high":
+ if avg_tenure >= 3.0:
+ stab_pts = 8
+ stability_note = f"Bonne stabilité ({avg_tenure:.1f} ans/poste en moy.)"
+ elif avg_tenure > 0:
+ stab_pts = -5
+ stability_note = f"Mobilité fréquente ({avg_tenure:.1f} ans/poste en moy.)"
+ elif stability_pref == "moderate":
+ if 1.0 <= avg_tenure < 4.0:
+ stab_pts = 4
+ stability_note = f"Stabilité modérée ({avg_tenure:.1f} ans/poste en moy.)"
+ elif avg_tenure >= 4.0:
+ stability_note = f"Profil stable ({avg_tenure:.1f} ans/poste en moy.)"
+ elif stability_pref == "flexible":
+ if avg_tenure < 1.5:
+ stab_pts = 4
+ stability_note = f"Profil mobile ({avg_tenure:.1f} ans/poste en moy.)"
+ raw += stab_pts * w_stab
+
+ matched_req_list = sorted(required & skills)
+ missing_req_list = sorted(required - skills)
+ bonus_list = sorted((preferred & skills) - required)
+
+ # Normalize to 0–100 with decimal precision
+ if max_possible > 0:
+ final_score = round(min(raw / max_possible * 100, 100), 2)
+ else:
+ final_score = 0.0
+
+ # ── Mention ──────────────────────────────────────────────
+ if final_score >= 85: grade = "A"
+ elif final_score >= 70: grade = "B"
+ elif final_score >= 50: grade = "C"
+ else: grade = "D"
+
+ # ── Justification ─────────────────────────────────────────
+ parts = []
+ if required:
+ ratio = len(matched_req_list) / len(required)
+ if ratio == 1.0:
+ parts.append(f"Toutes les {len(required)} compétences requises ({', '.join(matched_req_list[:4])})")
+ elif ratio >= 0.6:
+ parts.append(f"{len(matched_req_list)}/{len(required)} compétences requises"
+ + (f" — manque : {', '.join(missing_req_list[:2])}" if missing_req_list else ""))
+ else:
+ parts.append(f"Seulement {len(matched_req_list)}/{len(required)} requises — manque {', '.join(missing_req_list[:3])}")
+ if min_years > 0:
+ label = f"{years:.1f} ans exp. (requis : {min_years:.0f}+)" if years >= min_years else f"{years:.1f} ans exp. (< {min_years:.0f} requis)"
+ parts.append(label)
+ elif years > 0:
+ parts.append(f"{years:.1f} ans d'expérience")
+ edu_label = _EDU_LABEL_FR.get(edu, "")
+ if edu_label and edu_label != "Non renseigné":
+ parts.append(edu_label)
+ if stability_note:
+ parts.append(stability_note)
+ if bonus_list:
+ parts.append(f"Bonus : {', '.join(bonus_list[:3])}")
+ reasoning = " ; ".join(parts) if parts else "Évaluation sur profil global"
+
+ scored.append({
+ "key": p.get("key", ""),
+ "name": name,
+ "email": info.get("email", ""),
+ "summary": (info.get("summary") or "")[:300],
+ "skills": sorted(skills),
+ "experience_years": years,
+ "avg_tenure": avg_tenure,
+ "education": edu,
+ "matched_skills": matched_req_list,
+ "missing_skills": missing_req_list,
+ "bonus_skills": bonus_list,
+ "score": final_score,
+ "grade": grade,
+ "reasoning": reasoning,
+ })
+
+ scored.sort(key=lambda x: x["score"], reverse=True)
+ return scored
+
+
+def _score_with_hrflow(profiles: list, predictions: list, reqs: dict, user_weights: dict = None) -> list:
+ """
+ Combine HrFlow AI predictions (Leo's algorithm_key) avec les ajustements
+ du questionnaire pondérés par les sliders utilisateur.
+ Score final = score_IA_HrFlow (0-100) + delta questionnaire (±30 pts max).
+ """
+ required = {s.lower() for s in reqs.get("required_skills", [])}
+ preferred = {s.lower() for s in reqs.get("preferred_skills", [])}
+ min_years = float(reqs.get("min_experience_years") or 0)
+ max_years = float(reqs.get("max_experience_years") or 99)
+ edu_req = _EDU_RANK.get(reqs.get("education_level", "none"), 0)
+ stability_pref = reqs.get("stability_pref", "none")
+
+ uw = user_weights or {}
+ w_exp = uw.get("g3", 5) / 5.0
+ w_edu = uw.get("g2", 3) / 5.0
+ s_keys = [k for k in uw if k.startswith("s")]
+ w_skill = (sum(uw[k] for k in s_keys) / len(s_keys) / 5.0) if s_keys else 1.0
+ w_stab = uw.get("g_stab", 3) / 5.0
+
+ scored = []
+ for i, p in enumerate(profiles):
+ # Score de base HrFlow IA (predictions[i][1] = probabilité 0-1)
+ ai_score = 0.0
+ if i < len(predictions) and len(predictions[i]) > 1:
+ ai_score = round(predictions[i][1] * 100, 2)
+
+ skills = _get_skills(p)
+ years = _calculate_experience_years(p)
+ edu = _edu_level(p)
+ avg_tenure = _average_tenure_years(p)
+ info = p.get("info", {})
+ name = (info.get("full_name")
+ or f"{info.get('first_name','')} {info.get('last_name','')}".strip()
+ or p.get("key", "")[:12])
+
+ delta = 0.0
+
+ # Compétences : ±15 pts pondérés
+ matched_req_list = sorted(required & skills) if required else []
+ missing_req_list = sorted(required - skills) if required else []
+ bonus_list = sorted((preferred & skills) - required) if preferred else []
+ if required:
+ skill_ratio = len(matched_req_list) / len(required)
+ delta += (skill_ratio - 0.5) * 20 * w_skill
+
+ # Expérience : ±10 pts pondérés
+ if min_years > 0:
+ if years >= min_years: delta += 8 * w_exp
+ elif years >= min_years * 0.75: delta += 2 * w_exp
+ else: delta -= 8 * w_exp
+ if max_years < 99 and years > max_years + 5:
+ delta -= 5 * w_exp
+
+ # Formation : ±8 pts pondérés
+ cand_edu_rank = _EDU_RANK.get(edu, 0)
+ if cand_edu_rank >= edu_req: delta += 5 * w_edu
+ elif cand_edu_rank < edu_req - 1: delta -= 8 * w_edu
+
+ # Stabilité : ±5 pts pondérés
+ stability_note = ""
+ if stability_pref == "high":
+ if avg_tenure >= 3.0:
+ delta += 5 * w_stab
+ stability_note = f"Bonne stabilité ({avg_tenure:.1f} ans/poste)"
+ elif avg_tenure > 0:
+ delta -= 4 * w_stab
+ stability_note = f"Mobilité fréquente ({avg_tenure:.1f} ans/poste)"
+ elif stability_pref == "moderate":
+ if 1.0 <= avg_tenure < 4.0:
+ delta += 2 * w_stab
+ stability_note = f"Stabilité modérée ({avg_tenure:.1f} ans/poste)"
+ elif stability_pref == "flexible":
+ if avg_tenure < 1.5:
+ delta += 2 * w_stab
+ stability_note = f"Profil mobile ({avg_tenure:.1f} ans/poste)"
+
+ final_score = round(min(max(ai_score + delta, 0), 100), 2)
+
+ if final_score >= 85: grade = "A"
+ elif final_score >= 70: grade = "B"
+ elif final_score >= 50: grade = "C"
+ else: grade = "D"
+
+ parts = [f"Score HrFlow IA : {ai_score:.1f}/100"]
+ if required:
+ ratio = len(matched_req_list) / len(required)
+ if ratio == 1.0:
+ parts.append(f"Toutes les {len(required)} compétences requises ({', '.join(matched_req_list[:3])})")
+ elif ratio >= 0.5:
+ parts.append(f"{len(matched_req_list)}/{len(required)} compétences" +
+ (f" — manque : {', '.join(missing_req_list[:2])}" if missing_req_list else ""))
+ else:
+ parts.append(f"Seulement {len(matched_req_list)}/{len(required)} requises — manque {', '.join(missing_req_list[:2])}")
+ if min_years > 0:
+ parts.append(f"{years:.1f} ans exp. (requis : {min_years:.0f}+)" if years >= min_years
+ else f"{years:.1f} ans exp. (< {min_years:.0f} requis)")
+ elif years > 0:
+ parts.append(f"{years:.1f} ans d'expérience")
+ edu_label = _EDU_LABEL_FR.get(edu, "")
+ if edu_label and edu_label != "Non renseigné":
+ parts.append(edu_label)
+ if stability_note:
+ parts.append(stability_note)
+ if bonus_list:
+ parts.append(f"Bonus : {', '.join(bonus_list[:3])}")
+ reasoning = " ; ".join(parts) if parts else "Évaluation HrFlow IA"
+
+ scored.append({
+ "key": p.get("key", ""),
+ "name": name,
+ "email": info.get("email", ""),
+ "summary": (info.get("summary") or "")[:300],
+ "skills": sorted(skills),
+ "experience_years": years,
+ "avg_tenure": avg_tenure,
+ "education": edu,
+ "matched_skills": matched_req_list,
+ "missing_skills": missing_req_list,
+ "bonus_skills": bonus_list,
+ "score": final_score,
+ "ai_base_score": ai_score,
+ "grade": grade,
+ "reasoning": reasoning,
+ })
+
+ scored.sort(key=lambda x: x["score"], reverse=True)
+ return scored
+
+
+# ─── Analyse du vivier ────────────────────────────────────────────────────────
+
+def _analyse_candidates(candidates: list) -> dict:
+ if len(candidates) < 3:
+ return {"common_skills": [], "differentiators": [], "experience_range": "", "score_range": "", "candidate_count": len(candidates)}
+
+ top = candidates[:min(20, len(candidates))]
+ n = len(top)
+ threshold = 0.65
+
+ all_skills_flat = [s for c in top for s in c["skills"]]
+ skill_counts = Counter(all_skills_flat)
+
+ common_skills = [s for s, cnt in skill_counts.most_common(30) if cnt >= n * threshold]
+ differentiators = [s for s, cnt in skill_counts.most_common(50) if n * 0.15 <= cnt < n * threshold]
+
+ years_list = [c["experience_years"] for c in top]
+ scores = [c["score"] for c in top]
+
+ return {
+ "candidate_count": n,
+ "common_skills": common_skills[:12],
+ "differentiators": differentiators[:12],
+ "experience_range": f"{min(years_list):.0f}–{max(years_list):.0f} ans" if years_list else "",
+ "score_range": f"{min(scores):.0f}–{max(scores):.0f}" if scores else "",
+ "education_distribution": dict(Counter(c["education"] for c in top)),
+ }
+
+
+# ─── Affinement ───────────────────────────────────────────────────────────────
+
+def _apply_refinement(candidates: list, refinement_text: str, original_reqs: dict, user_weights: dict = None) -> list:
+ text_lower = refinement_text.lower()
+ updated = dict(original_reqs)
+
+ ALL_TECHS = [
+ "react","vue","angular","typescript","javascript","python","java","go","rust",
+ "kubernetes","docker","aws","gcp","azure","kafka","spark","pytorch","tensorflow",
+ "llm","graphql","grpc","redis","mongodb","postgresql","mysql",
+ "flutter","swift","kotlin","next.js","nuxt","terraform","ansible",
+ "elasticsearch","clickhouse","airflow","dbt","scikit-learn","rag",
+ ]
+ extra_required = [t for t in ALL_TECHS if t in text_lower]
+ if extra_required:
+ existing = [s.lower() for s in updated.get("required_skills", [])]
+ for t in extra_required:
+ if t not in existing:
+ updated.setdefault("required_skills", []).append(t)
+
+ m = re.search(r"(\d+)\s*(?:ans?|années?)", refinement_text, re.IGNORECASE)
+ if m:
+ updated["min_experience_years"] = max(updated.get("min_experience_years", 0), int(m.group(1)))
+
+ if any(k in text_lower for k in ["doctorat","phd"]):
+ updated["education_level"] = "phd"
+ elif any(k in text_lower for k in ["master","mastère","m2"]):
+ if _EDU_RANK.get(updated.get("education_level","none"),0) < 2:
+ updated["education_level"] = "master"
+
+ if "stable" in text_lower or "stabilité" in text_lower:
+ updated["stability_pref"] = "high"
+ elif "mobile" in text_lower or "freelance" in text_lower:
+ updated["stability_pref"] = "flexible"
+
+ fake_profiles = []
+ for c in candidates:
+ fake_profiles.append({
+ "key": c["key"],
+ "info": {"full_name": c["name"], "email": c["email"], "summary": c["summary"]},
+ "skills": [{"name": s} for s in c["skills"]],
+ "experiences": [{}] * max(1, round(c["experience_years"] / max(c.get("avg_tenure", 1), 0.1))),
+ "experiences_duration": c["experience_years"],
+ "educations": _fake_edu(c["education"]),
+ })
+ return _score_candidates(fake_profiles, updated, user_weights)
+
+
+def _fake_exp_tenure(total_years: float, avg_tenure: float) -> list:
+ if total_years <= 0:
+ return []
+ from datetime import timedelta
+ tenure = avg_tenure if avg_tenure > 0 else total_years
+ n_jobs = max(1, round(total_years / tenure))
+ exps = []
+ end = datetime.now()
+ for _ in range(n_jobs):
+ start = end - timedelta(days=int(tenure * 365))
+ exps.append({"title":"Engineer","company":"Company",
+ "date_start": start.isoformat(),"date_end": end.isoformat()})
+ end = start
+ return exps
+
+
+def _fake_edu(level: str) -> list:
+ titles = {"bachelor":"Licence","master":"Master","phd":"Doctorat","none":""}
+ t = titles.get(level, "")
+ return [{"title": t}] if t else []
+
+
+# ─── API Endpoints ────────────────────────────────────────────────────────────
+
+@app.get("/")
+def root():
+ return FileResponse("static/index_openhr.html")
+
+
+@app.post("/api/start")
+def api_start(req: StartRequest):
+ if not req.job_title.strip():
+ raise HTTPException(400, "L'intitulé du poste est requis")
+ questionnaire = _generate_questionnaire(req.job_title, req.job_description)
+ session_id = str(uuid.uuid4())
+ sessions[session_id] = {
+ "job_title": req.job_title,
+ "job_description": req.job_description,
+ "questionnaire": questionnaire,
+ "candidates": [],
+ "requirements": {},
+ }
+ return {"session_id": session_id, "questionnaire": questionnaire}
+
+
+@app.post("/api/submit")
+def api_submit(req: SubmitAnswersRequest):
+ session = sessions.get(req.session_id)
+ if not session:
+ raise HTTPException(404, "Session introuvable, veuillez actualiser")
+
+ user_weights = req.weights or {}
+ reqs = _interpret_answers(session["questionnaire"], req.answers, user_weights)
+ session["requirements"] = reqs
+ session["user_weights"] = user_weights
+
+ try:
+ # Étape 1 : indexer le poste sur le board pour obtenir un job_key
+ job_key = hrflow.index_job_sync(config.HRFLOW_BOARD_KEY, session["job_title"])
+ session["job_key"] = job_key
+
+ # Étape 2 : attendre que HrFlow calcule les scores (approche Leo)
+ time.sleep(5)
+
+ # Étape 3 : récupérer les profils scorés par l'IA HrFlow (algorithm_key)
+ scoring_resp = hrflow.score_profiles_with_algo(
+ job_key=job_key,
+ board_key=config.HRFLOW_BOARD_KEY,
+ source_keys=[config.HRFLOW_SOURCE_KEY],
+ limit=100,
+ )
+ raw_profiles = scoring_resp.get("data", {}).get("profiles", [])
+ predictions = scoring_resp.get("data", {}).get("predictions", [])
+
+ # Fallback : si le scoring API échoue, revenir à la recherche
+ if not raw_profiles:
+ search_resp = hrflow.search_profiles(
+ source_keys=[config.HRFLOW_SOURCE_KEY],
+ query=session["job_title"],
+ page=1, limit=100,
+ )
+ raw_profiles = search_resp.get("data", {}).get("profiles", [])
+ predictions = []
+ except Exception as e:
+ raise HTTPException(500, f"Erreur lors de la récupération des candidats : {e}")
+
+ if predictions:
+ scored = _score_with_hrflow(raw_profiles, predictions, reqs, user_weights)
+ else:
+ scored = _score_candidates(raw_profiles, reqs, user_weights)
+ session["candidates"] = scored
+
+ qualified = [c for c in scored if c["score"] >= 35]
+ if not qualified:
+ qualified = scored[:10]
+
+ analysis = _analyse_candidates(qualified)
+
+ if len(qualified) > 10:
+ return {
+ "status": "refine",
+ "analysis": analysis,
+ "requirements_summary": reqs.get("key_requirements_summary", ""),
+ "sample_candidates": qualified[:6],
+ "total_qualified": len(qualified),
+ }
+
+ return {
+ "status": "done",
+ "candidates": qualified[:15],
+ "analysis": analysis,
+ "requirements_summary": reqs.get("key_requirements_summary", ""),
+ "total_qualified": len(qualified),
+ }
+
+
+@app.post("/api/refine")
+def api_refine(req: RefineRequest):
+ session = sessions.get(req.session_id)
+ if not session:
+ raise HTTPException(404, "Session introuvable")
+
+ candidates = session.get("candidates", [])
+ if not candidates:
+ raise HTTPException(400, "Aucun candidat disponible")
+
+ refined = _apply_refinement(candidates, req.refinement, session["requirements"], session.get("user_weights", {}))
+ qualified = [c for c in refined if c["score"] >= 35]
+ if not qualified:
+ qualified = refined[:10]
+
+ analysis = _analyse_candidates(qualified)
+ return {
+ "status": "done",
+ "candidates": qualified[:15],
+ "analysis": analysis,
+ "requirements_summary": req.refinement[:80],
+ "total_qualified": len(qualified),
+ }
+
+
+@app.get("/api/health")
+def health():
+ return {"status": "ok", "service": "AI Agent avec OpenClaw v2", "port": 8002}