Compare commits
No commits in common. "master" and "feat/W3-web-v1b" have entirely different histories.
master
...
feat/W3-we
57 changed files with 143 additions and 6215 deletions
|
|
@ -1,7 +1,6 @@
|
|||
# ---- Database ----
|
||||
# Used by apps/api to connect to the postgres service defined in docker-compose.yml.
|
||||
# Postgres is not published to the host; all services run inside the compose network.
|
||||
DATABASE_URL=postgresql://jobhunt:***@postgres:5432/jobhunt
|
||||
DATABASE_URL=postgresql://jobhunt:***@localhost:5433/jobhunt
|
||||
|
||||
# ---- LLM Gateway ----
|
||||
# Primary provider (default: GLM-5.2 via ollama-cloud).
|
||||
|
|
|
|||
|
|
@ -11,12 +11,7 @@ jobs:
|
|||
api-tests:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
run: |
|
||||
git init
|
||||
git remote add origin https://x-access-token:${FORGEJO_TOKEN}@srvr.nu/git/hermes/jobhunt-platform.git
|
||||
git fetch --depth 1 origin ${GITHUB_SHA}
|
||||
git checkout FETCH_HEAD
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up Docker
|
||||
run: |
|
||||
docker --version
|
||||
|
|
@ -29,12 +24,7 @@ jobs:
|
|||
package-tests:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
run: |
|
||||
git init
|
||||
git remote add origin https://x-access-token:${FORGEJO_TOKEN}@srvr.nu/git/hermes/jobhunt-platform.git
|
||||
git fetch --depth 1 origin ${GITHUB_SHA}
|
||||
git checkout FETCH_HEAD
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
|
|
@ -63,27 +53,17 @@ jobs:
|
|||
. .venv/bin/activate
|
||||
uv pip install -e ".[dev]"
|
||||
pytest -q
|
||||
- name: Run matching tests
|
||||
run: |
|
||||
cd packages/matching
|
||||
uv venv
|
||||
. .venv/bin/activate
|
||||
uv pip install -e ".[dev]"
|
||||
pytest -q
|
||||
|
||||
web-tests:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
run: |
|
||||
git init
|
||||
git remote add origin https://x-access-token:${FORGEJO_TOKEN}@srvr.nu/git/hermes/jobhunt-platform.git
|
||||
git fetch --depth 1 origin ${GITHUB_SHA}
|
||||
git checkout FETCH_HEAD
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up Node
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: "22"
|
||||
cache: npm
|
||||
cache-dependency-path: apps/web/package-lock.json
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
cd apps/web
|
||||
|
|
|
|||
|
|
@ -1,145 +0,0 @@
|
|||
name: Deploy to Production
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: 'Leave as "auto" to bump from latest git tag, or enter a specific version (e.g. v0.1.2)'
|
||||
required: false
|
||||
default: 'auto'
|
||||
type: string
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
name: Build and deploy
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
run: |
|
||||
git init
|
||||
git remote add origin https://x-access-token:${FORGEJO_TOKEN}@srvr.nu/git/hermes/jobhunt-platform.git
|
||||
git fetch --depth 1 origin ${GITHUB_SHA}
|
||||
git checkout FETCH_HEAD
|
||||
|
||||
- name: Resolve version
|
||||
run: |
|
||||
INPUT_VERSION="${{ github.event.inputs.version }}"
|
||||
if [ -z "$INPUT_VERSION" ] || [ "$INPUT_VERSION" = "auto" ]; then
|
||||
git fetch --tags origin
|
||||
LATEST=$(git tag --list 'v*' --sort=-v:refname | head -1)
|
||||
if [ -z "$LATEST" ]; then LATEST="v0.0.0"; fi
|
||||
BASE="${LATEST#v}"
|
||||
MAJOR=$(echo "$BASE" | cut -d. -f1)
|
||||
MINOR=$(echo "$BASE" | cut -d. -f2)
|
||||
PATCH=$(echo "$BASE" | cut -d. -f3)
|
||||
PATCH=$(( ${PATCH:-0} + 1 ))
|
||||
VERSION="v${MAJOR:-0}.${MINOR:-0}.${PATCH}"
|
||||
echo "Latest tag: $LATEST → auto-bumped to $VERSION"
|
||||
else
|
||||
VERSION="$INPUT_VERSION"
|
||||
echo "Using manual version: $VERSION"
|
||||
fi
|
||||
if ! echo "$VERSION" | grep -qE '^v[0-9]+\.[0-9]+\.[0-9]+$'; then
|
||||
echo "ERROR: resolved version '$VERSION' is not valid semver (expected vX.Y.Z)"
|
||||
exit 1
|
||||
fi
|
||||
echo "VERSION=$VERSION" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Tag version
|
||||
run: |
|
||||
git tag -d ${{ env.VERSION }} 2>/dev/null || true
|
||||
git push origin --delete ${{ env.VERSION }} 2>/dev/null || true
|
||||
git tag ${{ env.VERSION }}
|
||||
git push origin ${{ env.VERSION }}
|
||||
|
||||
- name: Write production .env
|
||||
env:
|
||||
LLM_PRIMARY_KEY: ${{ secrets.LLM_PRIMARY_KEY }}
|
||||
run: |
|
||||
{
|
||||
printf 'DATABASE_URL=%s\n' 'postgresql://jobhunt:jobhunt@postgres:5432/jobhunt'
|
||||
printf 'LLM_PRIMARY_BASE_URL=%s\n' 'https://ollama.com/v1'
|
||||
printf 'LLM_PRIMARY_KEY=%s\n' "$LLM_PRIMARY_KEY"
|
||||
printf 'LLM_PRIMARY_MODEL=%s\n' 'glm-5.2'
|
||||
printf 'LLM_CHEAP_MODEL=%s\n' 'glm-5.2'
|
||||
printf 'LLM_STRONG_MODEL=%s\n' 'glm-5.2'
|
||||
printf 'VITE_API_BASE=%s\n' '/api'
|
||||
} > .env
|
||||
|
||||
- name: Build and start production stack
|
||||
run: |
|
||||
docker compose -p jobhunt -f docker-compose.prod.yml down
|
||||
docker compose -p jobhunt -f docker-compose.prod.yml up --build -d
|
||||
|
||||
- name: Health checks with rollback
|
||||
run: |
|
||||
echo "Waiting for services to start..."
|
||||
sleep 15
|
||||
|
||||
API_OK=false
|
||||
for i in 1 2 3 4 5 6 7 8 9 10; do
|
||||
if docker run --rm --network jobhunt_default curlimages/curl:8.5.0 \
|
||||
-sf http://jobhunt-api:8000/api/health > /dev/null; then
|
||||
echo "API is healthy"
|
||||
API_OK=true
|
||||
break
|
||||
fi
|
||||
echo "API check attempt $i failed, retrying in 5s..."
|
||||
sleep 5
|
||||
done
|
||||
|
||||
WEB_OK=false
|
||||
for i in 1 2 3 4 5; do
|
||||
if docker run --rm --network jobhunt_default curlimages/curl:8.5.0 \
|
||||
-sf http://jobhunt-web/ > /dev/null; then
|
||||
echo "Frontend is serving"
|
||||
WEB_OK=true
|
||||
break
|
||||
fi
|
||||
echo "Frontend check attempt $i failed, retrying in 5s..."
|
||||
sleep 5
|
||||
done
|
||||
|
||||
if [ "$API_OK" != "true" ] || [ "$WEB_OK" != "true" ]; then
|
||||
echo ""
|
||||
echo "═══════════════════════════════════════════════════"
|
||||
echo " HEALTH CHECK FAILED — DIAGNOSTICS"
|
||||
echo "═══════════════════════════════════════════════════"
|
||||
echo ""
|
||||
docker compose -p jobhunt -f docker-compose.prod.yml ps
|
||||
echo ""
|
||||
echo "--- API logs ---"
|
||||
docker logs jobhunt-api 2>&1 | tail -80 || true
|
||||
echo ""
|
||||
echo "--- Postgres logs ---"
|
||||
docker logs jobhunt-postgres 2>&1 | tail -30 || true
|
||||
echo ""
|
||||
echo "═══════════════════════════════════════════════════"
|
||||
echo " ROLLING BACK DEPLOYMENT"
|
||||
echo "═══════════════════════════════════════════════════"
|
||||
echo ""
|
||||
docker compose -p jobhunt -f docker-compose.prod.yml down
|
||||
echo ""
|
||||
echo "Rolled back. Containers stopped. DB volume preserved."
|
||||
echo "Read API logs above to find the root cause before redeploying."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Seed demo data (idempotent)
|
||||
run: |
|
||||
docker run --rm --network jobhunt_default curlimages/curl:8.5.0 \
|
||||
-sf -X POST http://jobhunt-api:8000/api/concierge/seed-demo || \
|
||||
echo "WARN: demo seed failed (non-fatal)"
|
||||
|
||||
- name: Print deploy status
|
||||
run: |
|
||||
echo ""
|
||||
echo "═══════════════════════════════════════════════════"
|
||||
echo " Deployed ${{ env.VERSION }} to production"
|
||||
echo "═══════════════════════════════════════════════════"
|
||||
echo ""
|
||||
docker compose -p jobhunt -f docker-compose.prod.yml ps
|
||||
echo ""
|
||||
echo "Web UI: http://tocke:8085"
|
||||
echo "API: http://tocke:8000/api/health"
|
||||
echo ""
|
||||
|
|
@ -20,7 +20,6 @@ WORKDIR /app/apps/api
|
|||
RUN pip install --no-cache-dir -e ".[dev]" \
|
||||
&& pip install --no-cache-dir -e /app/packages/llm-gateway \
|
||||
&& pip install --no-cache-dir -e /app/packages/artifacts \
|
||||
&& pip install --no-cache-dir -e /app/packages/matching \
|
||||
&& pip install --no-cache-dir pypdf python-docx apscheduler
|
||||
|
||||
CMD ["pytest", "-q"]
|
||||
|
|
@ -73,7 +73,6 @@ def reset_database(database_url: str | None = None) -> None:
|
|||
conn.execute(
|
||||
"""
|
||||
DROP TABLE IF EXISTS task_run, outbox, approval, artifact,
|
||||
email_suggestion, notification_log,
|
||||
application, job_posting, cv_section, profile, schema_migrations
|
||||
CASCADE
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -55,28 +55,6 @@ def get_job_posting(posting_id: str) -> dict[str, Any] | None:
|
|||
return _normalize_posting(row)
|
||||
|
||||
|
||||
def update_posting_cluster_id(posting_id: str, cluster_id: str) -> dict[str, Any] | None:
|
||||
"""Set the cluster_id on a job posting."""
|
||||
row = execute(
|
||||
"UPDATE job_posting SET cluster_id = %s WHERE id = %s RETURNING *",
|
||||
(cluster_id, posting_id),
|
||||
)
|
||||
if row is None:
|
||||
return None
|
||||
return _normalize_posting(row)
|
||||
|
||||
|
||||
def update_posting_apply_by(posting_id: str, apply_by: Any) -> dict[str, Any] | None:
|
||||
"""Set the apply_by date on a job posting."""
|
||||
row = execute(
|
||||
"UPDATE job_posting SET apply_by = %s WHERE id = %s RETURNING *",
|
||||
(apply_by, posting_id),
|
||||
)
|
||||
if row is None:
|
||||
return None
|
||||
return _normalize_posting(row)
|
||||
|
||||
|
||||
def list_postings() -> list[dict[str, Any]]:
|
||||
rows = fetch_all("SELECT * FROM job_posting ORDER BY fetched_at DESC")
|
||||
return [_normalize_posting(r) for r in rows]
|
||||
|
|
@ -93,8 +71,6 @@ def _normalize_posting(row: dict[str, Any]) -> dict[str, Any]:
|
|||
"location": row.get("location"),
|
||||
"description": row.get("description", ""),
|
||||
"fetched_at": row["fetched_at"].isoformat() if row.get("fetched_at") else None,
|
||||
"cluster_id": row.get("cluster_id"),
|
||||
"apply_by": row.get("apply_by").isoformat() if row.get("apply_by") else None,
|
||||
}
|
||||
|
||||
|
||||
|
|
@ -507,31 +483,3 @@ def get_digest(limit: int = 20) -> list[dict[str, Any]]:
|
|||
(limit,),
|
||||
)
|
||||
return [_normalize_application(r) for r in rows]
|
||||
|
||||
|
||||
def get_upcoming_deadlines(days: int = 7) -> list[dict[str, Any]]:
|
||||
"""Return applications whose job_posting has apply_by within the next *days* days.
|
||||
|
||||
Returns list of dicts: {application_id, title, company, apply_by}.
|
||||
"""
|
||||
rows = fetch_all(
|
||||
"""
|
||||
SELECT a.id AS application_id, j.title, j.company, j.apply_by
|
||||
FROM application a
|
||||
JOIN job_posting j ON a.job_posting_id = j.id
|
||||
WHERE j.apply_by IS NOT NULL
|
||||
AND j.apply_by >= CURRENT_DATE
|
||||
AND j.apply_by <= CURRENT_DATE + %s * INTERVAL '1 day'
|
||||
ORDER BY j.apply_by ASC
|
||||
""",
|
||||
(days,),
|
||||
)
|
||||
result: list[dict[str, Any]] = []
|
||||
for row in rows:
|
||||
result.append({
|
||||
"application_id": str(row["application_id"]),
|
||||
"title": row["title"],
|
||||
"company": row["company"],
|
||||
"apply_by": row["apply_by"].isoformat() if row.get("apply_by") else None,
|
||||
})
|
||||
return result
|
||||
|
|
@ -1,468 +0,0 @@
|
|||
"""IMAP email watch: polls UNSEEN messages and classifies them.
|
||||
|
||||
Uses stdlib imaplib (SSL). Enabled only when EMAIL_WATCH_ENABLED=true.
|
||||
Config: IMAP_HOST, IMAP_PORT, IMAP_USER, IMAP_PASS.
|
||||
|
||||
Flow:
|
||||
1. Connect via IMAP SSL.
|
||||
2. Fetch UNSEEN messages since last poll.
|
||||
3. For each message, match sender domain + subject/body keywords to
|
||||
open applications (status in sent/interviewing).
|
||||
4. Classify via LLM task 'email_classify' -> {classification, state_proposal, reason}.
|
||||
5. Insert email_suggestion rows (skip noise and dedupe by from+subject+day).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import email
|
||||
import email.utils
|
||||
import hashlib
|
||||
import imaplib
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Sequence
|
||||
|
||||
from app.db import execute, fetch_all, fetch_one
|
||||
from app import llm
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
VALID_CLASSIFICATIONS = frozenset({
|
||||
"interview_invite",
|
||||
"rejection",
|
||||
"question",
|
||||
"noise",
|
||||
})
|
||||
|
||||
# Keywords for cheap pre-matching before LLM classify
|
||||
INTERVIEW_KEYWORDS = ("interview", "invite", "meeting", "schedule", "call")
|
||||
REJECTION_KEYWORDS = ("regret", "unfortunately", "not moving", "rejection", "position has been filled")
|
||||
QUESTION_KEYWORDS = ("question", "clarif", "additional", "could you", "please provide")
|
||||
NOISE_KEYWORDS = ("newsletter", "unsubscribe", "promotion", "advert", "offer")
|
||||
|
||||
|
||||
def is_email_watch_enabled() -> bool:
|
||||
"""Check if email watch is enabled."""
|
||||
return os.environ.get("EMAIL_WATCH_ENABLED", "false").lower() in (
|
||||
"true",
|
||||
"1",
|
||||
"yes",
|
||||
)
|
||||
|
||||
|
||||
def _extract_sender_domain(from_addr: str) -> str:
|
||||
"""Extract the domain from an email From header."""
|
||||
parsed = email.utils.parseaddr(from_addr)
|
||||
addr = parsed[1] or from_addr
|
||||
parts = addr.split("@")
|
||||
if len(parts) >= 2:
|
||||
return parts[-1].lower().strip()
|
||||
return ""
|
||||
|
||||
|
||||
def _build_snippet(body: str, max_len: int = 300) -> str:
|
||||
"""Truncate body to a snippet."""
|
||||
body = body.replace("\r", " ").replace("\n", " ").strip()
|
||||
if len(body) > max_len:
|
||||
return body[:max_len] + "..."
|
||||
return body
|
||||
|
||||
|
||||
def _parse_email_message(raw_bytes: bytes) -> dict[str, str]:
|
||||
"""Parse raw email bytes into a dict with from, subject, body."""
|
||||
msg = email.message_from_bytes(raw_bytes)
|
||||
from_addr = msg.get("From", "")
|
||||
subject = msg.get("Subject", "")
|
||||
date_str = msg.get("Date", "")
|
||||
|
||||
# Extract body (prefer plain text)
|
||||
body = ""
|
||||
if msg.is_multipart():
|
||||
for part in msg.walk():
|
||||
ct = part.get_content_type()
|
||||
if ct == "text/plain":
|
||||
payload = part.get_payload(decode=True)
|
||||
if payload:
|
||||
body = payload.decode("utf-8", errors="replace")
|
||||
break
|
||||
if not body:
|
||||
for part in msg.walk():
|
||||
if part.get_content_type().startswith("text/"):
|
||||
payload = part.get_payload(decode=True)
|
||||
if payload:
|
||||
body = payload.decode("utf-8", errors="replace")
|
||||
break
|
||||
else:
|
||||
payload = msg.get_payload(decode=True)
|
||||
if payload:
|
||||
body = payload.decode("utf-8", errors="replace")
|
||||
|
||||
return {
|
||||
"from": from_addr,
|
||||
"subject": subject,
|
||||
"body": body,
|
||||
"date": date_str,
|
||||
}
|
||||
|
||||
|
||||
def _parse_date(date_str: str) -> datetime:
|
||||
"""Parse RFC 2822 date string to UTC datetime. Falls back to now()."""
|
||||
if date_str:
|
||||
try:
|
||||
parsed = email.utils.parsedate_to_datetime(date_str)
|
||||
if parsed is not None:
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def match_application(
|
||||
from_addr: str,
|
||||
subject: str,
|
||||
body: str,
|
||||
applications: Sequence[dict[str, Any]],
|
||||
) -> dict[str, Any] | None:
|
||||
"""Match an email to an open application.
|
||||
|
||||
Match by:
|
||||
1. Company name in subject/body
|
||||
2. Sender domain in posting URL
|
||||
|
||||
Only matches applications with state in ('sent', 'interviewing').
|
||||
"""
|
||||
sender_domain = _extract_sender_domain(from_addr)
|
||||
text_lower = (subject + " " + body).lower()
|
||||
|
||||
for app_row in applications:
|
||||
state = app_row.get("state", "")
|
||||
if state not in ("sent", "interviewing"):
|
||||
continue
|
||||
|
||||
company = (app_row.get("company") or "").lower()
|
||||
title = (app_row.get("title") or "").lower()
|
||||
posting_url = app_row.get("url") or ""
|
||||
|
||||
# Match 1: company name in subject/body
|
||||
if company and len(company) > 2 and company in text_lower:
|
||||
return app_row
|
||||
|
||||
# Match 2: sender domain in posting URL
|
||||
if sender_domain and sender_domain in (posting_url or "").lower():
|
||||
return app_row
|
||||
|
||||
# Match 3: title keywords in subject (looser)
|
||||
if title and len(title) > 3:
|
||||
title_words = [w for w in title.split() if len(w) > 3]
|
||||
matches = sum(1 for w in title_words if w in text_lower)
|
||||
if matches >= 2 and len(title_words) >= 2:
|
||||
return app_row
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def classify_email(subject: str, body: str) -> dict[str, Any]:
|
||||
"""Classify an email via LLM task 'email_classify'.
|
||||
|
||||
Returns {classification, state_proposal, reason}.
|
||||
Falls back to inline keyword heuristics if LLM fails.
|
||||
"""
|
||||
text_lower = (subject + " " + body).lower()
|
||||
|
||||
result = llm.run_task(
|
||||
"email_classify",
|
||||
f"Subject: {subject}\nBody: {body[:1000]}",
|
||||
)
|
||||
|
||||
classification = result.get("classification", "noise")
|
||||
state_proposal = result.get("state_proposal")
|
||||
reason = result.get("reason", "")
|
||||
|
||||
# Validate
|
||||
if classification not in VALID_CLASSIFICATIONS:
|
||||
classification = "noise"
|
||||
|
||||
return {
|
||||
"classification": classification,
|
||||
"state_proposal": state_proposal,
|
||||
"reason": reason,
|
||||
}
|
||||
|
||||
|
||||
def is_duplicate(from_addr: str, subject: str, received_at: datetime) -> bool:
|
||||
"""Check if a similar email_suggestion already exists (same from+subject+day)."""
|
||||
day_start = received_at.replace(hour=0, minute=0, second=0, microsecond=0)
|
||||
row = fetch_one(
|
||||
"""
|
||||
SELECT id FROM email_suggestion
|
||||
WHERE mailbox_from = %s AND subject = %s
|
||||
AND received_at >= %s AND received_at < %s + interval '1 day'
|
||||
LIMIT 1
|
||||
""",
|
||||
(from_addr, subject, day_start, day_start),
|
||||
)
|
||||
return row is not None
|
||||
|
||||
|
||||
def create_email_suggestion(
|
||||
application_id: str | None,
|
||||
mailbox_from: str,
|
||||
subject: str,
|
||||
snippet: str,
|
||||
classification: str,
|
||||
state_proposal: str | None,
|
||||
received_at: datetime,
|
||||
) -> dict[str, Any]:
|
||||
"""Insert an email_suggestion row."""
|
||||
row = execute(
|
||||
"""
|
||||
INSERT INTO email_suggestion
|
||||
(application_id, mailbox_from, subject, snippet, classification, state_proposal, status, received_at)
|
||||
VALUES
|
||||
(%s, %s, %s, %s, %s, %s, 'pending', %s)
|
||||
RETURNING *
|
||||
""",
|
||||
(application_id, mailbox_from, subject, snippet, classification, state_proposal, received_at),
|
||||
)
|
||||
if row is None:
|
||||
raise RuntimeError("insert email_suggestion failed")
|
||||
return _normalize_suggestion(row)
|
||||
|
||||
|
||||
def _normalize_suggestion(row: dict[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
"id": str(row["id"]),
|
||||
"application_id": str(row["application_id"]) if row.get("application_id") else None,
|
||||
"mailbox_from": row["mailbox_from"],
|
||||
"subject": row["subject"],
|
||||
"snippet": row["snippet"],
|
||||
"classification": row["classification"],
|
||||
"state_proposal": row.get("state_proposal"),
|
||||
"status": row["status"],
|
||||
"received_at": row["received_at"].isoformat() if row.get("received_at") else None,
|
||||
"created_at": row["created_at"].isoformat() if row.get("created_at") else None,
|
||||
}
|
||||
|
||||
|
||||
def list_pending_suggestions() -> list[dict[str, Any]]:
|
||||
"""Return all pending email_suggestions, newest first."""
|
||||
rows = fetch_all(
|
||||
"""
|
||||
SELECT * FROM email_suggestion
|
||||
WHERE status = 'pending'
|
||||
ORDER BY received_at DESC
|
||||
"""
|
||||
)
|
||||
return [_normalize_suggestion(r) for r in rows]
|
||||
|
||||
|
||||
def get_suggestion(suggestion_id: str) -> dict[str, Any] | None:
|
||||
row = fetch_one("SELECT * FROM email_suggestion WHERE id = %s", (suggestion_id,))
|
||||
if row is None:
|
||||
return None
|
||||
return _normalize_suggestion(row)
|
||||
|
||||
|
||||
def update_suggestion_status(
|
||||
suggestion_id: str,
|
||||
status: str,
|
||||
) -> dict[str, Any] | None:
|
||||
row = execute(
|
||||
"UPDATE email_suggestion SET status = %s WHERE id = %s RETURNING *",
|
||||
(status, suggestion_id),
|
||||
)
|
||||
if row is None:
|
||||
return None
|
||||
return _normalize_suggestion(row)
|
||||
|
||||
|
||||
def get_open_applications() -> list[dict[str, Any]]:
|
||||
"""Return applications in sent/interviewing state for matching."""
|
||||
rows = fetch_all(
|
||||
"""
|
||||
SELECT a.*, j.company, j.title, j.url
|
||||
FROM application a
|
||||
JOIN job_posting j ON a.job_posting_id = j.id
|
||||
WHERE a.state IN ('sent', 'interviewing')
|
||||
"""
|
||||
)
|
||||
result: list[dict[str, Any]] = []
|
||||
for row in rows:
|
||||
result.append({
|
||||
"id": str(row["id"]),
|
||||
"state": row["state"],
|
||||
"company": row.get("company", ""),
|
||||
"title": row.get("title", ""),
|
||||
"url": row.get("url", ""),
|
||||
})
|
||||
return result
|
||||
|
||||
|
||||
# --- IMAP poll ---
|
||||
|
||||
class FakeImap:
|
||||
"""Test double for imaplib IMAP4_SSL. No network.
|
||||
|
||||
Usage:
|
||||
fake = FakeImap(messages=[(uid1, raw1), (uid2, raw2)])
|
||||
poll_inbox(fake) # uses fake instead of real connection
|
||||
"""
|
||||
|
||||
def __init__(self, messages: list[tuple[bytes, bytes]] | None = None) -> None:
|
||||
# messages: list of (uid, raw_email_bytes)
|
||||
self._messages = messages or []
|
||||
self._seen_uids: set[bytes] = set()
|
||||
self.selected = False
|
||||
|
||||
def select(self, mailbox: str = "INBOX") -> tuple[str, list[bytes]]:
|
||||
self.selected = True
|
||||
count = len(self._messages)
|
||||
return ("OK", [str(count).encode()])
|
||||
|
||||
def search(self, charset: str | None, *criteria: str) -> tuple[str, list[bytes]]:
|
||||
# Return uids of messages matching criteria (we keep it simple)
|
||||
uids = [uid for uid, _ in self._messages]
|
||||
return ("OK", [b" ".join(uids)])
|
||||
|
||||
def fetch(self, uid: bytes, parts: str) -> tuple[str, list[tuple[bytes, bytes]]]:
|
||||
for msg_uid, raw in self._messages:
|
||||
if msg_uid == uid:
|
||||
return ("OK", [(uid, raw)])
|
||||
return ("OK", [])
|
||||
|
||||
def store(self, uid: bytes, flags: str, flag_set: str) -> tuple[str, list[bytes]]:
|
||||
self._seen_uids.add(uid)
|
||||
return ("OK", [uid])
|
||||
|
||||
def close(self) -> tuple[str, list[bytes]]:
|
||||
self.selected = False
|
||||
return ("OK", [b""])
|
||||
|
||||
def logout(self) -> tuple[str, list[bytes]]:
|
||||
return ("OK", [b"BYE"])
|
||||
|
||||
|
||||
def _connect_imap() -> Any:
|
||||
"""Connect to IMAP server using env config."""
|
||||
host = os.environ.get("IMAP_HOST", "")
|
||||
port = int(os.environ.get("IMAP_PORT", "993"))
|
||||
user = os.environ.get("IMAP_USER", "")
|
||||
password = os.environ.get("IMAP_PASS", "")
|
||||
|
||||
conn = imaplib.IMAP4_SSL(host, port)
|
||||
conn.login(user, password)
|
||||
return conn
|
||||
|
||||
|
||||
def poll_inbox(conn: Any = None) -> list[dict[str, Any]]:
|
||||
"""Poll the inbox for UNSEEN messages, classify, and create suggestions.
|
||||
|
||||
If conn is provided (e.g. FakeImap for tests), uses it instead of connecting.
|
||||
Returns list of created suggestions.
|
||||
|
||||
No network is used when conn is a FakeImap.
|
||||
"""
|
||||
created: list[dict[str, Any]] = []
|
||||
own_conn = False
|
||||
|
||||
if conn is None:
|
||||
conn = _connect_imap()
|
||||
own_conn = True
|
||||
|
||||
try:
|
||||
conn.select("INBOX")
|
||||
status, data = conn.search(None, "UNSEEN")
|
||||
if status != "OK":
|
||||
logger.warning("imap_watch: search failed: %s", status)
|
||||
return created
|
||||
|
||||
uids = []
|
||||
if data and data[0]:
|
||||
uids = data[0].split()
|
||||
|
||||
if not uids:
|
||||
return created
|
||||
|
||||
# Get open applications for matching
|
||||
applications = get_open_applications()
|
||||
|
||||
for uid in uids:
|
||||
status, fetch_data = conn.fetch(uid, "(RFC822)")
|
||||
if status != "OK" or not fetch_data:
|
||||
continue
|
||||
|
||||
raw_bytes = b""
|
||||
for item in fetch_data:
|
||||
if isinstance(item, tuple) and len(item) >= 2:
|
||||
raw_bytes = item[1]
|
||||
break
|
||||
|
||||
if not raw_bytes:
|
||||
continue
|
||||
|
||||
parsed = _parse_email_message(raw_bytes)
|
||||
from_addr = parsed["from"]
|
||||
subject = parsed["subject"]
|
||||
body = parsed["body"]
|
||||
received_at = _parse_date(parsed["date"])
|
||||
snippet = _build_snippet(body)
|
||||
|
||||
# Dedupe: same from+subject+day
|
||||
if is_duplicate(from_addr, subject, received_at):
|
||||
logger.debug("imap_watch: dedupe skip: %s / %s", from_addr, subject)
|
||||
continue
|
||||
|
||||
# Match to application
|
||||
app_row = match_application(from_addr, subject, body, applications)
|
||||
|
||||
# Classify
|
||||
classification_result = classify_email(subject, body)
|
||||
classification = classification_result["classification"]
|
||||
state_proposal = classification_result["state_proposal"]
|
||||
|
||||
# Skip pure noise (don't create suggestion rows)
|
||||
if classification == "noise":
|
||||
continue
|
||||
|
||||
suggestion = create_email_suggestion(
|
||||
application_id=app_row["id"] if app_row else None,
|
||||
mailbox_from=from_addr,
|
||||
subject=subject,
|
||||
snippet=snippet,
|
||||
classification=classification,
|
||||
state_proposal=state_proposal,
|
||||
received_at=received_at,
|
||||
)
|
||||
created.append(suggestion)
|
||||
|
||||
# Send notification for interview invites
|
||||
if classification == "interview_invite" and app_row:
|
||||
from app.notify import send_notification
|
||||
send_notification(
|
||||
"email_suggestion",
|
||||
f"Interview invite from {app_row.get('company', 'unknown')}",
|
||||
{
|
||||
"suggestion_id": suggestion["id"],
|
||||
"application_id": app_row["id"],
|
||||
"classification": classification,
|
||||
},
|
||||
)
|
||||
|
||||
# Mark as seen
|
||||
try:
|
||||
conn.store(uid, "+FLAGS", "\\Seen")
|
||||
except Exception:
|
||||
logger.debug("imap_watch: store failed for uid %s", uid)
|
||||
|
||||
finally:
|
||||
if own_conn:
|
||||
try:
|
||||
conn.close()
|
||||
conn.logout()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return created
|
||||
|
|
@ -123,68 +123,9 @@ MOCK_OUTPUTS: dict[str, dict[str, Any]] = {
|
|||
"posting requirements. Be specific."
|
||||
),
|
||||
},
|
||||
"email_classify": {
|
||||
"classification": "interview_invite",
|
||||
"state_proposal": "interviewing",
|
||||
"reason": "The email mentions an interview invitation.",
|
||||
},
|
||||
"cv_tailor": {
|
||||
"tailored_cv": {
|
||||
"summary": "Senior Python Developer with 6+ years building scalable backend systems.",
|
||||
"skills": [
|
||||
"Python",
|
||||
"Fast API",
|
||||
"PostgreSQL",
|
||||
"Docker",
|
||||
"Kubernetes",
|
||||
"AWS",
|
||||
],
|
||||
"experience": [
|
||||
{
|
||||
"company": "TechCorp",
|
||||
"role": "Senior Backend Engineer",
|
||||
"bullets": [
|
||||
"Led migration of monolith to microservices using Fast API",
|
||||
"Reduced API latency by 40% through query optimization and caching",
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
"change_log": [
|
||||
{"action": "reordered", "detail": "Moved Python and Fast API to top of skills"},
|
||||
{"action": "rephrased", "detail": "Rewrote first experience bullet to emphasize Fast API"},
|
||||
],
|
||||
},
|
||||
"deadline_extract": {
|
||||
"apply_by": None,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _mock_cv_tailor(prompt: str) -> dict[str, Any]:
|
||||
"""Prompt-aware mock tailor: extracts source bullets from the prompt and
|
||||
rephrases them deterministically, so the result always passes the
|
||||
hallucination guard (which requires traceable source overlap)."""
|
||||
import re
|
||||
|
||||
bullets = re.findall(r'"([A-ZÅÄÖ][^"]{20,300})"', prompt)
|
||||
bullets = [b for b in bullets if "{" not in b and ":" not in b][:4]
|
||||
if not bullets:
|
||||
bullets = ["Experienced backend developer focused on reliability"]
|
||||
tailored = []
|
||||
change_log = []
|
||||
for b in bullets[:2]:
|
||||
tailored.append(f"{b} (tailored for this posting)")
|
||||
change_log.append({"action": "rephrased", "detail": f"Emphasized relevance of {b[:60]}"})
|
||||
for b in bullets[2:]:
|
||||
tailored.append(b)
|
||||
change_log.append({"action": "kept", "detail": f"Retained as-is {b[:60]}"})
|
||||
return {
|
||||
"tailored_cv": {"summary": bullets[0][:160], "bullets": tailored},
|
||||
"change_log": change_log,
|
||||
}
|
||||
|
||||
|
||||
def run_task(
|
||||
task: str,
|
||||
prompt: str,
|
||||
|
|
@ -202,8 +143,6 @@ def run_task(
|
|||
# Mock mode
|
||||
time.sleep(0.01) # simulate latency
|
||||
result = MOCK_OUTPUTS.get(task, {"result": "mock"})
|
||||
if task == "cv_tailor":
|
||||
result = _mock_cv_tailor(prompt)
|
||||
|
||||
# Validate against schema if provided (basic check)
|
||||
# In real gateway this would be jsonschema validation
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -1,183 +0,0 @@
|
|||
"""Notification channels: LogChannel + WebhookChannel.
|
||||
|
||||
Protocol-based: NotificationChannel defines send(kind, text, data) -> bool.
|
||||
LogChannel writes to notification_log (delivered=true always).
|
||||
WebhookChannel POSTs to NOTIFY_WEBHOOK_URL (2xx=delivered, else error row).
|
||||
|
||||
send_notification(kind, text, data) is the public entry point used by
|
||||
the scheduler and email watch modules.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from typing import Any, Protocol
|
||||
|
||||
from app.db import execute, fetch_all
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class NotificationChannel(Protocol):
|
||||
"""Protocol for notification delivery channels."""
|
||||
|
||||
def send(self, kind: str, text: str, data: dict[str, Any]) -> bool:
|
||||
"""Send a notification. Returns True if delivered, False otherwise."""
|
||||
...
|
||||
|
||||
|
||||
def _normalize_log_row(row: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = row.get("payload")
|
||||
if payload is not None and not isinstance(payload, dict):
|
||||
payload = json.loads(payload) if isinstance(payload, str) else payload
|
||||
return {
|
||||
"id": str(row["id"]),
|
||||
"channel": row["channel"],
|
||||
"kind": row["kind"],
|
||||
"payload": payload or {},
|
||||
"delivered": row["delivered"],
|
||||
"error": row.get("error"),
|
||||
"created_at": row["created_at"].isoformat() if row.get("created_at") else None,
|
||||
}
|
||||
|
||||
|
||||
class LogChannel:
|
||||
"""Writes notification entries to notification_log with delivered=true."""
|
||||
|
||||
channel_name: str = "log"
|
||||
|
||||
def send(self, kind: str, text: str, data: dict[str, Any]) -> bool:
|
||||
row = execute(
|
||||
"""
|
||||
INSERT INTO notification_log (channel, kind, payload, delivered, error)
|
||||
VALUES (%s, %s, %s, true, NULL)
|
||||
RETURNING *
|
||||
""",
|
||||
(self.channel_name, kind, json.dumps({"text": text, **data})),
|
||||
)
|
||||
if row is None:
|
||||
logger.error("LogChannel: insert notification_log failed")
|
||||
return False
|
||||
logger.info("LogChannel: delivered kind=%s", kind)
|
||||
return True
|
||||
|
||||
|
||||
class WebhookChannel:
|
||||
"""POSTs notification payload to NOTIFY_WEBHOOK_URL.
|
||||
|
||||
On 2xx response: writes notification_log with delivered=true.
|
||||
On non-2xx or exception: writes notification_log with delivered=false and error.
|
||||
"""
|
||||
|
||||
channel_name: str = "webhook"
|
||||
|
||||
def __init__(self, url: str | None = None) -> None:
|
||||
self.url = url or os.environ.get("NOTIFY_WEBHOOK_URL", "")
|
||||
|
||||
def send(self, kind: str, text: str, data: dict[str, Any]) -> bool:
|
||||
payload: dict[str, Any] = {"kind": kind, "text": text, "data": data}
|
||||
error: str | None = None
|
||||
delivered = False
|
||||
|
||||
if not self.url:
|
||||
error = "NOTIFY_WEBHOOK_URL not configured"
|
||||
logger.warning("WebhookChannel: %s", error)
|
||||
else:
|
||||
try:
|
||||
import httpx
|
||||
|
||||
resp = httpx.post(self.url, json=payload, timeout=10)
|
||||
if 200 <= resp.status_code < 300:
|
||||
delivered = True
|
||||
else:
|
||||
error = f"HTTP {resp.status_code}: {resp.text[:200]}"
|
||||
logger.warning("WebhookChannel: %s", error)
|
||||
except Exception as exc:
|
||||
error = str(exc)
|
||||
logger.warning("WebhookChannel: exception: %s", error)
|
||||
|
||||
row = execute(
|
||||
"""
|
||||
INSERT INTO notification_log (channel, kind, payload, delivered, error)
|
||||
VALUES (%s, %s, %s, %s, %s)
|
||||
RETURNING *
|
||||
""",
|
||||
(
|
||||
self.channel_name,
|
||||
kind,
|
||||
json.dumps(payload),
|
||||
delivered,
|
||||
error,
|
||||
),
|
||||
)
|
||||
if row is None:
|
||||
logger.error("WebhookChannel: insert notification_log failed")
|
||||
return delivered
|
||||
|
||||
|
||||
# --- Channel registry ---
|
||||
|
||||
_channels: list[NotificationChannel] | None = None
|
||||
|
||||
|
||||
def get_channels() -> list[NotificationChannel]:
|
||||
"""Return the list of active notification channels.
|
||||
|
||||
LogChannel is always included.
|
||||
WebhookChannel is included when NOTIFY_WEBHOOK_URL is set.
|
||||
"""
|
||||
global _channels
|
||||
if _channels is not None:
|
||||
return _channels
|
||||
|
||||
channels: list[NotificationChannel] = [LogChannel()]
|
||||
webhook_url = os.environ.get("NOTIFY_WEBHOOK_URL", "").strip()
|
||||
if webhook_url:
|
||||
channels.append(WebhookChannel())
|
||||
_channels = channels
|
||||
return channels
|
||||
|
||||
|
||||
def set_channels(channels: list[NotificationChannel] | None) -> None:
|
||||
"""Override channel list (for testing)."""
|
||||
global _channels
|
||||
_channels = channels
|
||||
|
||||
|
||||
def reset_channels() -> None:
|
||||
"""Reset to default (for testing)."""
|
||||
global _channels
|
||||
_channels = None
|
||||
|
||||
|
||||
def send_notification(kind: str, text: str, data: dict[str, Any] | None = None) -> None:
|
||||
"""Send a notification via all active channels.
|
||||
|
||||
kind: e.g. 'daily_digest', 'email_suggestion'
|
||||
text: human-readable notification text
|
||||
data: structured payload dict
|
||||
"""
|
||||
chs = get_channels()
|
||||
payload = data or {}
|
||||
for ch in chs:
|
||||
try:
|
||||
ch.send(kind, text, payload)
|
||||
except Exception:
|
||||
logger.exception("send_notification: channel %s failed", type(ch).__name__)
|
||||
|
||||
|
||||
# --- Repository helpers ---
|
||||
|
||||
def list_notification_log(limit: int = 50) -> list[dict[str, Any]]:
|
||||
"""Return recent notification_log rows, newest first."""
|
||||
rows = fetch_all(
|
||||
"""
|
||||
SELECT * FROM notification_log
|
||||
ORDER BY created_at DESC
|
||||
LIMIT %s
|
||||
""",
|
||||
(limit,),
|
||||
)
|
||||
return [_normalize_log_row(r) for r in rows]
|
||||
|
|
@ -1,10 +1,7 @@
|
|||
"""APScheduler integration: daily fetch + batch score + imap poll + digest.
|
||||
"""APScheduler integration: daily fetch + batch score job.
|
||||
|
||||
Starts during app lifespan when SCHEDULER_ENABLED=true (default false).
|
||||
Jobs:
|
||||
- daily_fetch_score at 07:00: fetch postings + batch score
|
||||
- daily_digest at 07:30: send daily digest notification
|
||||
- imap_poll every 15 min: poll inbox for new emails (gated by EMAIL_WATCH_ENABLED)
|
||||
Runs a daily job at 07:00 that fetches postings and batch-scores pending applications.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -26,15 +23,6 @@ def is_scheduler_enabled() -> bool:
|
|||
)
|
||||
|
||||
|
||||
def is_email_watch_enabled() -> bool:
|
||||
"""Check if email watch is enabled via env."""
|
||||
return os.environ.get("EMAIL_WATCH_ENABLED", "false").lower() in (
|
||||
"true",
|
||||
"1",
|
||||
"yes",
|
||||
)
|
||||
|
||||
|
||||
async def _daily_fetch_and_score() -> None:
|
||||
"""Daily job: fetch postings and batch-score pending applications."""
|
||||
logger.info("Scheduler: running daily fetch + batch score")
|
||||
|
|
@ -68,67 +56,6 @@ async def _daily_fetch_and_score() -> None:
|
|||
logger.exception("Scheduler: daily job failed")
|
||||
|
||||
|
||||
async def _daily_digest() -> None:
|
||||
"""Daily digest: build /today payload text and send notification."""
|
||||
logger.info("Scheduler: running daily digest")
|
||||
try:
|
||||
from app.notify import send_notification
|
||||
from app.db import repo_app
|
||||
|
||||
digest_apps = repo_app.get_digest(limit=20)
|
||||
nudge_apps = repo_app.get_nudge_applications()
|
||||
pending = repo_app.count_pending_approvals()
|
||||
|
||||
lines: list[str] = []
|
||||
lines.append("Daily Digest")
|
||||
lines.append(f"Scored applications: {len(digest_apps)}")
|
||||
if digest_apps:
|
||||
lines.append("")
|
||||
lines.append("Top opportunities:")
|
||||
for item in digest_apps[:5]:
|
||||
score = item.get("score")
|
||||
score_str = f" (score: {int(score)})" if score else ""
|
||||
lines.append(
|
||||
f" - {item.get('title', '?')} at {item.get('company', '?')}{score_str}"
|
||||
)
|
||||
|
||||
if nudge_apps:
|
||||
lines.append("")
|
||||
lines.append(f"Follow-up nudges: {len(nudge_apps)}")
|
||||
for n in nudge_apps[:5]:
|
||||
lines.append(
|
||||
f" - {n.get('title', '?')} at {n.get('company', '?')}"
|
||||
f" ({n.get('days_since_sent', 0)} days since sent)"
|
||||
)
|
||||
|
||||
if pending:
|
||||
lines.append("")
|
||||
lines.append(f"Pending approvals: {pending}")
|
||||
|
||||
text = "\n".join(lines)
|
||||
send_notification("daily_digest", text, {
|
||||
"digest_count": len(digest_apps),
|
||||
"nudge_count": len(nudge_apps),
|
||||
"pending_approvals": pending,
|
||||
})
|
||||
logger.info("Scheduler: daily digest sent")
|
||||
except Exception:
|
||||
logger.exception("Scheduler: daily digest failed")
|
||||
|
||||
|
||||
async def _imap_poll() -> None:
|
||||
"""IMAP poll job: check for new emails and create suggestions."""
|
||||
if not is_email_watch_enabled():
|
||||
return
|
||||
logger.info("Scheduler: running imap poll")
|
||||
try:
|
||||
from app.imap_watch import poll_inbox
|
||||
created = poll_inbox()
|
||||
logger.info("Scheduler: imap poll created %s suggestions", len(created))
|
||||
except Exception:
|
||||
logger.exception("Scheduler: imap poll failed")
|
||||
|
||||
|
||||
def start_scheduler() -> None:
|
||||
"""Start the APScheduler if enabled."""
|
||||
global _scheduler
|
||||
|
|
@ -139,7 +66,6 @@ def start_scheduler() -> None:
|
|||
try:
|
||||
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||
from apscheduler.triggers.cron import CronTrigger
|
||||
from apscheduler.triggers.interval import IntervalTrigger
|
||||
except ImportError:
|
||||
logger.warning(
|
||||
"APScheduler not installed; scheduler will not start."
|
||||
|
|
@ -153,20 +79,8 @@ def start_scheduler() -> None:
|
|||
id="daily_fetch_score",
|
||||
replace_existing=True,
|
||||
)
|
||||
_scheduler.add_job(
|
||||
_daily_digest,
|
||||
CronTrigger(hour=7, minute=30),
|
||||
id="daily_digest",
|
||||
replace_existing=True,
|
||||
)
|
||||
_scheduler.add_job(
|
||||
_imap_poll,
|
||||
IntervalTrigger(minutes=15),
|
||||
id="imap_poll",
|
||||
replace_existing=True,
|
||||
)
|
||||
_scheduler.start()
|
||||
logger.info("Scheduler started: daily fetch+score at 07:00, digest at 07:30, imap poll every 15 min")
|
||||
logger.info("Scheduler started: daily fetch+score at 07:00")
|
||||
|
||||
|
||||
def stop_scheduler() -> None:
|
||||
|
|
|
|||
|
|
@ -108,8 +108,6 @@ class JobPostingOut(BaseModel):
|
|||
location: str | None = None
|
||||
description: str = ""
|
||||
fetched_at: str | None = None
|
||||
cluster_id: str | None = None
|
||||
apply_by: str | None = None
|
||||
|
||||
|
||||
class ScoreResponse(BaseModel):
|
||||
|
|
@ -306,7 +304,6 @@ class TodayResponse(BaseModel):
|
|||
digest: list[DigestItem]
|
||||
nudges: list[NudgeItem]
|
||||
pending_approvals: int
|
||||
deadlines: list[DeadlineItem] = []
|
||||
|
||||
|
||||
# --- v1: Interview Prep ---
|
||||
|
|
@ -323,69 +320,3 @@ class SeedDemoResponse(BaseModel):
|
|||
postings: int
|
||||
applications: int
|
||||
sections: int
|
||||
clusters: int = 0
|
||||
deadlines: int = 0
|
||||
suggestions: int = 0
|
||||
notifications: int = 0
|
||||
task_runs: int = 0
|
||||
cv_artifacts: int = 0
|
||||
|
||||
|
||||
# --- v1.1: Email Suggestions ---
|
||||
|
||||
class EmailSuggestionOut(BaseModel):
|
||||
id: str
|
||||
application_id: str | None = None
|
||||
mailbox_from: str
|
||||
subject: str
|
||||
snippet: str
|
||||
classification: str
|
||||
state_proposal: str | None = None
|
||||
status: str
|
||||
received_at: str | None = None
|
||||
created_at: str | None = None
|
||||
|
||||
|
||||
# --- v1.1: Notification Log ---
|
||||
|
||||
class NotificationLogOut(BaseModel):
|
||||
id: str
|
||||
channel: str
|
||||
kind: str
|
||||
payload: dict[str, Any] = {}
|
||||
delivered: bool
|
||||
error: str | None = None
|
||||
created_at: str | None = None
|
||||
|
||||
|
||||
# --- v1.1: Clusters ---
|
||||
|
||||
class ClusterPostingOut(BaseModel):
|
||||
id: str
|
||||
title: str
|
||||
company: str
|
||||
source: str
|
||||
url: str
|
||||
score: float | None = None
|
||||
|
||||
|
||||
class ClusterOut(BaseModel):
|
||||
cluster_id: str
|
||||
postings: list[ClusterPostingOut] = []
|
||||
|
||||
|
||||
# --- v1.1: Tailor CV ---
|
||||
|
||||
class TailorCvResponse(BaseModel):
|
||||
artifact_id: str
|
||||
change_log: list[dict[str, Any]]
|
||||
keyword_coverage: dict[str, Any]
|
||||
|
||||
|
||||
# --- v1.1: Deadlines ---
|
||||
|
||||
class DeadlineItem(BaseModel):
|
||||
application_id: str
|
||||
title: str
|
||||
company: str
|
||||
apply_by: str | None = None
|
||||
|
|
@ -1,24 +0,0 @@
|
|||
-- 003_email_notify.sql -- email suggestions + notification log
|
||||
|
||||
CREATE TABLE IF NOT EXISTS email_suggestion (
|
||||
id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
application_id uuid REFERENCES application(id) ON DELETE SET NULL,
|
||||
mailbox_from text NOT NULL,
|
||||
subject text NOT NULL,
|
||||
snippet text NOT NULL,
|
||||
classification text NOT NULL CHECK (classification IN ('interview_invite','rejection','question','noise')),
|
||||
state_proposal text,
|
||||
status text NOT NULL DEFAULT 'pending' CHECK (status IN ('pending','accepted','dismissed')),
|
||||
received_at timestamptz NOT NULL,
|
||||
created_at timestamptz DEFAULT now()
|
||||
);
|
||||
|
||||
CREATE TABLE IF NOT EXISTS notification_log (
|
||||
id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
channel text NOT NULL,
|
||||
kind text NOT NULL,
|
||||
payload jsonb NOT NULL,
|
||||
delivered boolean NOT NULL,
|
||||
error text,
|
||||
created_at timestamptz DEFAULT now()
|
||||
);
|
||||
|
|
@ -1,4 +0,0 @@
|
|||
-- 004_dedupe_deadline.sql -- cluster_id for dedupe + apply_by deadline
|
||||
|
||||
ALTER TABLE job_posting ADD COLUMN IF NOT EXISTS cluster_id text;
|
||||
ALTER TABLE job_posting ADD COLUMN IF NOT EXISTS apply_by date;
|
||||
|
|
@ -38,7 +38,7 @@ def _truncate_tables():
|
|||
with psycopg.connect(DATABASE_URL) as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
TRUNCATE TABLE notification_log, email_suggestion, task_run, outbox, approval, artifact,
|
||||
TRUNCATE TABLE task_run, outbox, approval, artifact,
|
||||
application, job_posting, cv_section, profile
|
||||
RESTART IDENTITY CASCADE
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -1,682 +0,0 @@
|
|||
"""Tests for v1.1: email watch, notifications, suggestions endpoints.
|
||||
|
||||
Covers:
|
||||
- IMAP matching logic (company in subject, sender domain in URL, title keywords)
|
||||
- Classifier -> suggestion row
|
||||
- Noise dedupe (same from+subject+day)
|
||||
- Accept applies transition through guard
|
||||
- Dismiss marks suggestion
|
||||
- Webhook success/failure notification_log rows
|
||||
- LogChannel writes delivered=true
|
||||
- Daily digest payload shape
|
||||
- GET /suggestions, POST accept, POST dismiss, GET /notifications/log
|
||||
- FakeImap end-to-end poll
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import email as email_mod
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from app.config import DATABASE_URL
|
||||
from app.db import repo_app
|
||||
from app.imap_watch import (
|
||||
FakeImap,
|
||||
_build_snippet,
|
||||
_extract_sender_domain,
|
||||
_parse_email_message,
|
||||
classify_email,
|
||||
create_email_suggestion,
|
||||
is_duplicate,
|
||||
match_application,
|
||||
poll_inbox,
|
||||
)
|
||||
from app.notify import (
|
||||
LogChannel,
|
||||
WebhookChannel,
|
||||
get_channels,
|
||||
list_notification_log,
|
||||
reset_channels,
|
||||
send_notification,
|
||||
set_channels,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client():
|
||||
from app.main import app
|
||||
return TestClient(app)
|
||||
|
||||
|
||||
# --- Helpers ---
|
||||
|
||||
def _make_email(raw_from: str, subject: str, body: str, date: str = "") -> bytes:
|
||||
"""Build raw email bytes for FakeImap."""
|
||||
msg = email_mod.message_from_string(
|
||||
f"From: {raw_from}\r\n"
|
||||
f"Subject: {subject}\r\n"
|
||||
f"Date: {date or 'Mon, 01 Jul 2026 10:00:00 +0000'}\r\n"
|
||||
f"\r\n"
|
||||
f"{body}"
|
||||
)
|
||||
return msg.as_bytes()
|
||||
|
||||
|
||||
def _create_app_in_state(state: str = "sent", company: str = "TechCorp", url: str = "https://techcorp.com/jobs/1") -> dict:
|
||||
"""Create a posting + application, force state via SQL."""
|
||||
posting = repo_app.create_job_posting(
|
||||
source="manual_url",
|
||||
url=url,
|
||||
company=company,
|
||||
title="Senior Python Developer",
|
||||
location="Malmo",
|
||||
description="",
|
||||
raw={},
|
||||
)
|
||||
app_row = repo_app.create_application(posting["id"])
|
||||
if state != "discovered":
|
||||
repo_app.update_application_score(app_row["id"], 80, {"factors": {}})
|
||||
if state in ("approved", "rejected"):
|
||||
repo_app.update_application_state(app_row["id"], "scored")
|
||||
repo_app.update_application_state(app_row["id"], state)
|
||||
elif state == "sent":
|
||||
repo_app.update_application_state(app_row["id"], "scored")
|
||||
repo_app.update_application_state(app_row["id"], "approved")
|
||||
repo_app.update_application_state(app_row["id"], "drafting")
|
||||
# Bypass guard for test
|
||||
from app.db import execute
|
||||
execute(
|
||||
"UPDATE application SET state = 'sent', last_activity_at = now() WHERE id = %s",
|
||||
(app_row["id"],),
|
||||
)
|
||||
elif state == "interviewing":
|
||||
repo_app.update_application_state(app_row["id"], "scored")
|
||||
repo_app.update_application_state(app_row["id"], "approved")
|
||||
repo_app.update_application_state(app_row["id"], "drafting")
|
||||
from app.db import execute
|
||||
execute(
|
||||
"UPDATE application SET state = 'sent' WHERE id = %s",
|
||||
(app_row["id"],),
|
||||
)
|
||||
execute(
|
||||
"UPDATE application SET state = 'interviewing' WHERE id = %s",
|
||||
(app_row["id"],),
|
||||
)
|
||||
return app_row
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# IMAP matching logic (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestImapMatching:
|
||||
def test_match_by_company_name_in_subject(self):
|
||||
"""Email subject contains company name -> match."""
|
||||
apps = [
|
||||
{"id": "app1", "state": "sent", "company": "TechCorp", "title": "Python Dev", "url": "https://example.com"},
|
||||
]
|
||||
result = match_application(
|
||||
"recruiter@gmail.com",
|
||||
"Interview at TechCorp",
|
||||
"Please come for an interview.",
|
||||
apps,
|
||||
)
|
||||
assert result is not None
|
||||
assert result["id"] == "app1"
|
||||
|
||||
def test_match_by_sender_domain_in_url(self):
|
||||
"""Sender domain matches the posting URL -> match."""
|
||||
apps = [
|
||||
{"id": "app2", "state": "sent", "company": "Unknown", "title": "Dev", "url": "https://techcorp.com/careers/1"},
|
||||
]
|
||||
result = match_application(
|
||||
"hr@techcorp.com",
|
||||
"Your application",
|
||||
"We reviewed your application.",
|
||||
apps,
|
||||
)
|
||||
assert result is not None
|
||||
assert result["id"] == "app2"
|
||||
|
||||
def test_match_by_title_keywords(self):
|
||||
"""Email subject contains 2+ title words -> match."""
|
||||
apps = [
|
||||
{"id": "app3", "state": "interviewing", "company": "SomeCompany", "title": "Senior Python Developer", "url": "https://other.com"},
|
||||
]
|
||||
result = match_application(
|
||||
"someone@other.com",
|
||||
"Senior Python position update",
|
||||
"Regarding the developer role.",
|
||||
apps,
|
||||
)
|
||||
assert result is not None
|
||||
assert result["id"] == "app3"
|
||||
|
||||
def test_no_match_wrong_state(self):
|
||||
"""Applications in discovered state are not matched."""
|
||||
apps = [
|
||||
{"id": "app4", "state": "discovered", "company": "TechCorp", "title": "Dev", "url": "https://techcorp.com"},
|
||||
]
|
||||
result = match_application(
|
||||
"hr@techcorp.com",
|
||||
"Interview at TechCorp",
|
||||
"Come for an interview.",
|
||||
apps,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
def test_no_match_unrelated_email(self):
|
||||
"""Email unrelated to any application -> no match."""
|
||||
apps = [
|
||||
{"id": "app5", "state": "sent", "company": "TechCorp", "title": "Dev", "url": "https://techcorp.com"},
|
||||
]
|
||||
result = match_application(
|
||||
"newsletter@spam.com",
|
||||
"Buy now!",
|
||||
"Special offer for you.",
|
||||
apps,
|
||||
)
|
||||
assert result is None
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Classifier -> row (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestClassifierToRow:
|
||||
def test_classify_returns_interview_invite(self):
|
||||
"""classify_email returns interview_invite from mock."""
|
||||
result = classify_email("Interview invitation", "Please come for an interview next week.")
|
||||
assert result["classification"] == "interview_invite"
|
||||
assert result["state_proposal"] == "interviewing"
|
||||
assert "reason" in result
|
||||
|
||||
def test_classify_falls_back_on_invalid_classification(self):
|
||||
"""Invalid classification from LLM falls back to noise."""
|
||||
import app.llm as llm_mod
|
||||
original = llm_mod.MOCK_OUTPUTS.get("email_classify", {}).copy()
|
||||
try:
|
||||
llm_mod.MOCK_OUTPUTS["email_classify"] = {"classification": "bogus", "state_proposal": None, "reason": "test"}
|
||||
result = classify_email("test", "test")
|
||||
assert result["classification"] == "noise"
|
||||
finally:
|
||||
llm_mod.MOCK_OUTPUTS["email_classify"] = original
|
||||
|
||||
def test_create_email_suggestion_row(self):
|
||||
"""create_email_suggestion inserts a row correctly."""
|
||||
now = datetime.now(timezone.utc)
|
||||
suggestion = create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Interview invite",
|
||||
snippet="Please come for an interview.",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=now,
|
||||
)
|
||||
assert suggestion["id"] is not None
|
||||
assert suggestion["mailbox_from"] == "hr@example.com"
|
||||
assert suggestion["classification"] == "interview_invite"
|
||||
assert suggestion["status"] == "pending"
|
||||
assert suggestion["state_proposal"] == "interviewing"
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Noise dedupe (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestNoiseDedupe:
|
||||
def test_duplicate_detected_same_day(self):
|
||||
"""Same from+subject+day is flagged as duplicate."""
|
||||
now = datetime.now(timezone.utc)
|
||||
create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Interview",
|
||||
snippet="Come for an interview.",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=now,
|
||||
)
|
||||
assert is_duplicate("hr@example.com", "Interview", now)
|
||||
|
||||
def test_different_subject_not_duplicate(self):
|
||||
"""Different subject -> not duplicate."""
|
||||
now = datetime.now(timezone.utc)
|
||||
create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Interview",
|
||||
snippet="Come.",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=now,
|
||||
)
|
||||
assert not is_duplicate("hr@example.com", "Different Subject", now)
|
||||
|
||||
def test_different_sender_not_duplicate(self):
|
||||
"""Different sender -> not duplicate."""
|
||||
now = datetime.now(timezone.utc)
|
||||
create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Interview",
|
||||
snippet="Come.",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=now,
|
||||
)
|
||||
assert not is_duplicate("other@example.com", "Interview", now)
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Accept applies transition through guard (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestAcceptSuggestion:
|
||||
def test_accept_applies_transition(self, client):
|
||||
"""POST /suggestions/{id}/accept transitions app from sent to interviewing."""
|
||||
app_row = _create_app_in_state("sent", company="TechCorp")
|
||||
suggestion = create_email_suggestion(
|
||||
application_id=app_row["id"],
|
||||
mailbox_from="hr@techcorp.com",
|
||||
subject="Interview at TechCorp",
|
||||
snippet="Please come in for an interview.",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=datetime.now(timezone.utc),
|
||||
)
|
||||
|
||||
resp = client.post(f"/api/suggestions/{suggestion['id']}/accept")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["status"] == "accepted"
|
||||
|
||||
# Verify application state changed
|
||||
updated_app = repo_app.get_application(app_row["id"])
|
||||
assert updated_app["state"] == "interviewing"
|
||||
|
||||
def test_accept_invalid_transition_409(self, client):
|
||||
"""Accept with invalid transition (e.g. discovered -> interviewing) returns 409."""
|
||||
posting = repo_app.create_job_posting(
|
||||
source="manual_url", url="https://example.com/bad/1",
|
||||
company="X", title="X", location=None, description="", raw={},
|
||||
)
|
||||
app_row = repo_app.create_application(posting["id"])
|
||||
|
||||
suggestion = create_email_suggestion(
|
||||
application_id=app_row["id"],
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Test",
|
||||
snippet="Test",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=datetime.now(timezone.utc),
|
||||
)
|
||||
|
||||
resp = client.post(f"/api/suggestions/{suggestion['id']}/accept")
|
||||
assert resp.status_code == 409
|
||||
assert "invalid_transition" in str(resp.json()["detail"])
|
||||
|
||||
def test_accept_404_nonexistent(self, client):
|
||||
"""Accept on nonexistent suggestion -> 404."""
|
||||
resp = client.post("/api/suggestions/00000000-0000-0000-0000-000000000000/accept")
|
||||
assert resp.status_code == 404
|
||||
|
||||
def test_accept_already_accepted_409(self, client):
|
||||
"""Accept on already accepted suggestion -> 409."""
|
||||
app_row = _create_app_in_state("sent")
|
||||
suggestion = create_email_suggestion(
|
||||
application_id=app_row["id"],
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Test",
|
||||
snippet="Test",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=datetime.now(timezone.utc),
|
||||
)
|
||||
client.post(f"/api/suggestions/{suggestion['id']}/accept")
|
||||
resp = client.post(f"/api/suggestions/{suggestion['id']}/accept")
|
||||
assert resp.status_code == 409
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Dismiss suggestion (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestDismissSuggestion:
|
||||
def test_dismiss_marks_as_dismissed(self, client):
|
||||
"""POST /suggestions/{id}/dismiss marks as dismissed."""
|
||||
suggestion = create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="hr@example.com",
|
||||
subject="Test",
|
||||
snippet="Test",
|
||||
classification="question",
|
||||
state_proposal=None,
|
||||
received_at=datetime.now(timezone.utc),
|
||||
)
|
||||
|
||||
resp = client.post(f"/api/suggestions/{suggestion['id']}/dismiss")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["status"] == "dismissed"
|
||||
|
||||
def test_dismiss_404_nonexistent(self, client):
|
||||
"""Dismiss nonexistent -> 404."""
|
||||
resp = client.post("/api/suggestions/00000000-0000-0000-0000-000000000000/dismiss")
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# GET /suggestions (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestGetSuggestions:
|
||||
def test_get_suggestions_returns_pending(self, client):
|
||||
"""GET /suggestions returns only pending suggestions."""
|
||||
create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="a@example.com",
|
||||
subject="Subject A",
|
||||
snippet="Snippet A",
|
||||
classification="interview_invite",
|
||||
state_proposal="interviewing",
|
||||
received_at=datetime.now(timezone.utc),
|
||||
)
|
||||
create_email_suggestion(
|
||||
application_id=None,
|
||||
mailbox_from="b@example.com",
|
||||
subject="Subject B",
|
||||
snippet="Snippet B",
|
||||
classification="rejection",
|
||||
state_proposal=None,
|
||||
received_at=datetime.now(timezone.utc),
|
||||
)
|
||||
|
||||
resp = client.get("/api/suggestions")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert len(data) == 2
|
||||
assert all(s["status"] == "pending" for s in data)
|
||||
|
||||
def test_get_suggestions_empty(self, client):
|
||||
"""GET /suggestions returns empty list when no suggestions."""
|
||||
resp = client.get("/api/suggestions")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json() == []
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Notification channels (6 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestLogChannel:
|
||||
def test_log_channel_writes_delivered_true(self):
|
||||
"""LogChannel writes notification_log with delivered=true."""
|
||||
reset_channels()
|
||||
ch = LogChannel()
|
||||
result = ch.send("daily_digest", "Test digest", {"count": 5})
|
||||
assert result is True
|
||||
|
||||
logs = list_notification_log(limit=10)
|
||||
assert len(logs) >= 1
|
||||
latest = logs[0]
|
||||
assert latest["channel"] == "log"
|
||||
assert latest["kind"] == "daily_digest"
|
||||
assert latest["delivered"] is True
|
||||
assert latest["error"] is None
|
||||
|
||||
def test_send_notification_log_channel(self):
|
||||
"""send_notification via LogChannel creates a log entry."""
|
||||
reset_channels()
|
||||
set_channels([LogChannel()])
|
||||
send_notification("email_suggestion", "Interview invite from TechCorp", {"id": "test"})
|
||||
logs = list_notification_log(limit=10)
|
||||
assert len(logs) >= 1
|
||||
assert logs[0]["kind"] == "email_suggestion"
|
||||
assert logs[0]["delivered"] is True
|
||||
reset_channels()
|
||||
|
||||
|
||||
class TestWebhookChannel:
|
||||
def test_webhook_success_2xx(self, monkeypatch):
|
||||
"""WebhookChannel with 2xx response writes delivered=true."""
|
||||
class FakeResponse:
|
||||
status_code = 200
|
||||
text = "OK"
|
||||
|
||||
class FakeClient:
|
||||
@staticmethod
|
||||
def post(url, json=None, timeout=None):
|
||||
assert url == "https://hook.example.com/notify"
|
||||
return FakeResponse()
|
||||
|
||||
import httpx
|
||||
monkeypatch.setattr(httpx, "post", FakeClient.post)
|
||||
|
||||
reset_channels()
|
||||
ch = WebhookChannel(url="https://hook.example.com/notify")
|
||||
result = ch.send("daily_digest", "Digest", {"count": 3})
|
||||
assert result is True
|
||||
|
||||
logs = list_notification_log(limit=10)
|
||||
latest = logs[0]
|
||||
assert latest["channel"] == "webhook"
|
||||
assert latest["delivered"] is True
|
||||
assert latest["error"] is None
|
||||
reset_channels()
|
||||
|
||||
def test_webhook_failure_non_2xx(self, monkeypatch):
|
||||
"""WebhookChannel with non-2xx writes delivered=false with error."""
|
||||
class FakeResponse:
|
||||
status_code = 500
|
||||
text = "Internal Server Error"
|
||||
|
||||
class FakeClient:
|
||||
@staticmethod
|
||||
def post(url, json=None, timeout=None):
|
||||
return FakeResponse()
|
||||
|
||||
import httpx
|
||||
monkeypatch.setattr(httpx, "post", FakeClient.post)
|
||||
|
||||
reset_channels()
|
||||
ch = WebhookChannel(url="https://hook.example.com/notify")
|
||||
result = ch.send("daily_digest", "Digest", {"count": 3})
|
||||
assert result is False
|
||||
|
||||
logs = list_notification_log(limit=10)
|
||||
latest = logs[0]
|
||||
assert latest["channel"] == "webhook"
|
||||
assert latest["delivered"] is False
|
||||
assert latest["error"] is not None
|
||||
assert "500" in latest["error"]
|
||||
reset_channels()
|
||||
|
||||
def test_webhook_exception_writes_error(self, monkeypatch):
|
||||
"""WebhookChannel with connection exception writes delivered=false."""
|
||||
def raise_exc(url, json=None, timeout=None):
|
||||
raise ConnectionError("Connection refused")
|
||||
|
||||
import httpx
|
||||
monkeypatch.setattr(httpx, "post", raise_exc)
|
||||
|
||||
reset_channels()
|
||||
ch = WebhookChannel(url="https://hook.example.com/notify")
|
||||
result = ch.send("daily_digest", "Digest", {"count": 1})
|
||||
assert result is False
|
||||
|
||||
logs = list_notification_log(limit=10)
|
||||
latest = logs[0]
|
||||
assert latest["delivered"] is False
|
||||
assert "Connection refused" in (latest["error"] or "")
|
||||
reset_channels()
|
||||
|
||||
def test_webhook_no_url_writes_error(self):
|
||||
"""WebhookChannel with no URL writes delivered=false with config error."""
|
||||
reset_channels()
|
||||
ch = WebhookChannel(url="")
|
||||
result = ch.send("test", "test", {})
|
||||
assert result is False
|
||||
|
||||
logs = list_notification_log(limit=10)
|
||||
latest = logs[0]
|
||||
assert latest["delivered"] is False
|
||||
assert "not configured" in (latest["error"] or "")
|
||||
reset_channels()
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Notification log endpoint (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestNotificationLogEndpoint:
|
||||
def test_get_notifications_log(self, client):
|
||||
"""GET /notifications/log returns entries."""
|
||||
reset_channels()
|
||||
set_channels([LogChannel()])
|
||||
send_notification("daily_digest", "Test", {"count": 1})
|
||||
reset_channels()
|
||||
|
||||
resp = client.get("/api/notifications/log")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert len(data) >= 1
|
||||
assert "channel" in data[0]
|
||||
assert "kind" in data[0]
|
||||
assert "delivered" in data[0]
|
||||
|
||||
def test_get_notifications_log_empty(self, client):
|
||||
"""GET /notifications/log returns empty when no entries."""
|
||||
resp = client.get("/api/notifications/log")
|
||||
assert resp.status_code == 200
|
||||
assert resp.json() == []
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Digest payload shape (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestDigestPayload:
|
||||
def test_daily_digest_notification_text(self):
|
||||
"""Daily digest notification contains expected text fields."""
|
||||
reset_channels()
|
||||
set_channels([LogChannel()])
|
||||
send_notification("daily_digest", "Daily Digest\nScored applications: 5", {
|
||||
"digest_count": 5,
|
||||
"nudge_count": 2,
|
||||
"pending_approvals": 1,
|
||||
})
|
||||
logs = list_notification_log(limit=10)
|
||||
entry = logs[0]
|
||||
assert entry["kind"] == "daily_digest"
|
||||
payload = entry["payload"]
|
||||
assert "text" in payload
|
||||
assert "Daily Digest" in payload["text"]
|
||||
assert payload.get("digest_count") == 5
|
||||
reset_channels()
|
||||
|
||||
def test_daily_digest_payload_has_counts(self):
|
||||
"""Digest payload includes digest_count, nudge_count, pending_approvals."""
|
||||
reset_channels()
|
||||
set_channels([LogChannel()])
|
||||
send_notification("daily_digest", "text", {
|
||||
"digest_count": 3,
|
||||
"nudge_count": 1,
|
||||
"pending_approvals": 0,
|
||||
})
|
||||
logs = list_notification_log(limit=10)
|
||||
payload = logs[0]["payload"]
|
||||
assert payload["digest_count"] == 3
|
||||
assert payload["nudge_count"] == 1
|
||||
assert payload["pending_approvals"] == 0
|
||||
reset_channels()
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# FakeImap end-to-end poll (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestFakeImapPoll:
|
||||
def test_poll_creates_suggestion_for_matching_email(self):
|
||||
"""FakeImap poll creates a suggestion when email matches an application."""
|
||||
app_row = _create_app_in_state("sent", company="TechCorp", url="https://techcorp.com/jobs/1")
|
||||
|
||||
raw_email = _make_email(
|
||||
"hr@techcorp.com",
|
||||
"Interview at TechCorp",
|
||||
"Please come for an interview next Tuesday.",
|
||||
)
|
||||
fake = FakeImap(messages=[(b"1", raw_email)])
|
||||
created = poll_inbox(fake)
|
||||
assert len(created) == 1
|
||||
assert created[0]["classification"] == "interview_invite"
|
||||
assert created[0]["mailbox_from"] == "hr@techcorp.com"
|
||||
assert created[0]["application_id"] == app_row["id"]
|
||||
|
||||
def test_poll_skips_noise_emails(self):
|
||||
"""FakeImap poll skips noise classification (no suggestion created)."""
|
||||
_create_app_in_state("sent", company="TechCorp")
|
||||
|
||||
# Mock email_classify to return noise
|
||||
import app.llm as llm_mod
|
||||
original = llm_mod.MOCK_OUTPUTS.get("email_classify", {}).copy()
|
||||
try:
|
||||
llm_mod.MOCK_OUTPUTS["email_classify"] = {
|
||||
"classification": "noise",
|
||||
"state_proposal": None,
|
||||
"reason": "spam",
|
||||
}
|
||||
raw_email = _make_email(
|
||||
"newsletter@spam.com",
|
||||
"Buy our product",
|
||||
"Special offer just for you!",
|
||||
)
|
||||
fake = FakeImap(messages=[(b"1", raw_email)])
|
||||
created = poll_inbox(fake)
|
||||
assert len(created) == 0
|
||||
finally:
|
||||
llm_mod.MOCK_OUTPUTS["email_classify"] = original
|
||||
|
||||
def test_poll_dedupe_skips_same_from_subject_day(self):
|
||||
"""FakeImap poll dedupes same from+subject+day."""
|
||||
raw_email = _make_email(
|
||||
"hr@example.com",
|
||||
"Same Subject",
|
||||
"Same body content.",
|
||||
)
|
||||
# First poll
|
||||
fake1 = FakeImap(messages=[(b"1", raw_email)])
|
||||
created1 = poll_inbox(fake1)
|
||||
assert len(created1) == 1
|
||||
|
||||
# Second poll with same message -> dedupe
|
||||
fake2 = FakeImap(messages=[(b"1", raw_email)])
|
||||
created2 = poll_inbox(fake2)
|
||||
assert len(created2) == 0 # deduped
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Email parsing helpers (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestEmailParsing:
|
||||
def test_extract_sender_domain(self):
|
||||
"""Extract domain from From header."""
|
||||
assert _extract_sender_domain("John Doe <hr@techcorp.com>") == "techcorp.com"
|
||||
assert _extract_sender_domain("noreply@example.org") == "example.org"
|
||||
assert _extract_sender_domain("") == ""
|
||||
|
||||
def test_build_snippet_truncates(self):
|
||||
"""Snippet is truncated to max_len."""
|
||||
long_body = "A" * 500
|
||||
snippet = _build_snippet(long_body, max_len=50)
|
||||
assert len(snippet) <= 53 # 50 + "..."
|
||||
assert snippet.endswith("...")
|
||||
|
||||
def test_build_snippet_short_body(self):
|
||||
"""Short body is not truncated."""
|
||||
snippet = _build_snippet("Hello", max_len=300)
|
||||
assert snippet == "Hello"
|
||||
|
|
@ -1,737 +0,0 @@
|
|||
"""Tests for v1.1 wave B WB1: dedupe + tailor + deadline integration.
|
||||
|
||||
Covers:
|
||||
- Cluster assignment on posting creation (manual POST /postings)
|
||||
- Cluster stability across re-imports (same posting URL -> same cluster_id)
|
||||
- GET /clusters endpoint shape (cluster_id, postings with id/title/company/source/url/score)
|
||||
- Tailor CV happy path (artifact created, change_log, keyword_coverage)
|
||||
- Tailor CV hallucination rejection (fabricated mock returning unmapped bullet -> 502)
|
||||
- Keyword coverage numbers vs fixture
|
||||
- Deadline persisted on scoring (single + batch)
|
||||
- /today deadlines filter window (next 7 days)
|
||||
|
||||
Total: >= 20 new tests.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from datetime import date, datetime, timedelta, timezone
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from app.db import execute, fetch_one, repo_app, repo_profile
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def client():
|
||||
from app.main import app
|
||||
return TestClient(app)
|
||||
|
||||
|
||||
# --- Helpers ---
|
||||
|
||||
def _create_posting_direct(
|
||||
company: str = "TechCorp",
|
||||
title: str = "Senior Python Developer",
|
||||
url: str = "https://example.com/1",
|
||||
description: str = "We need a Python developer with FastAPI experience.",
|
||||
source: str = "manual_url",
|
||||
) -> dict:
|
||||
"""Create a posting directly via repo."""
|
||||
posting = repo_app.create_job_posting(
|
||||
source=source,
|
||||
url=url,
|
||||
company=company,
|
||||
title=title,
|
||||
location="Malmo",
|
||||
description=description,
|
||||
raw={},
|
||||
)
|
||||
repo_app.create_application(posting["id"])
|
||||
return posting
|
||||
|
||||
|
||||
def _create_app_with_profile_and_sections(
|
||||
client,
|
||||
company: str = "TechCorp",
|
||||
title: str = "Senior Python Developer",
|
||||
url: str = "https://example.com/tc1",
|
||||
description: str = "Python FastAPI PostgreSQL Docker Kubernetes AWS",
|
||||
) -> str:
|
||||
"""Create a profile with sections + posting + application. Returns app_id."""
|
||||
# Create profile
|
||||
client.get("/api/profile")
|
||||
client.put("/api/profile", json={
|
||||
"full_name": "Test User",
|
||||
"email": "test@test.com",
|
||||
"headline": "Backend Developer",
|
||||
"summary": "Experienced backend developer.",
|
||||
})
|
||||
# Create sections
|
||||
client.post("/api/profile/sections", json={
|
||||
"kind": "experience",
|
||||
"title": "Backend Developer",
|
||||
"org": "TechCorp",
|
||||
"bullets": [
|
||||
"Led migration of monolith to microservices using Fast API",
|
||||
"Reduced API latency by 40% through query optimization and caching",
|
||||
],
|
||||
"tags": ["python", "fastapi"],
|
||||
})
|
||||
client.post("/api/profile/sections", json={
|
||||
"kind": "skills",
|
||||
"title": "Technical Skills",
|
||||
"bullets": ["Python", "PostgreSQL", "Docker", "FastAPI"],
|
||||
"tags": ["python", "docker"],
|
||||
})
|
||||
# Create posting + application
|
||||
posting = _create_posting_direct(company=company, title=title, url=url, description=description)
|
||||
# Re-fetch to get the app
|
||||
apps = repo_app.list_applications()
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == posting["id"]:
|
||||
return a["id"]
|
||||
raise RuntimeError("Application not found")
|
||||
|
||||
|
||||
def _make_posting_dict(posting_id: str, company: str, title: str, description: str) -> dict:
|
||||
"""Build a posting dict suitable for cluster()."""
|
||||
return {
|
||||
"id": posting_id,
|
||||
"employer": company,
|
||||
"title": title,
|
||||
"description": description,
|
||||
}
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Cluster assignment on create (4 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestClusterAssignmentOnCreate:
|
||||
def test_single_posting_gets_cluster_id(self, client):
|
||||
"""A posting created via POST /postings gets a cluster_id assigned."""
|
||||
resp = client.post("/api/postings", json={"url": "https://example.com/cluster/1"})
|
||||
assert resp.status_code == 201
|
||||
|
||||
postings = client.get("/api/postings").json()
|
||||
assert len(postings) >= 1
|
||||
# The first posting might get c1 or no cluster (if it's the only one, cluster() gives it c1)
|
||||
# With matching available, even 1 posting gets cluster c1
|
||||
if len(postings) == 1:
|
||||
# Single posting: cluster() returns {"c1": [id]}
|
||||
assert postings[0]["cluster_id"] is not None
|
||||
|
||||
def test_duplicate_postings_same_cluster(self, client):
|
||||
"""Two identical postings (same company, title, description) get same cluster_id."""
|
||||
_create_posting_direct(
|
||||
company="Acme Corp",
|
||||
title="Software Engineer",
|
||||
url="https://example.com/dup/1",
|
||||
description="We need a Python developer with Docker experience.",
|
||||
)
|
||||
_create_posting_direct(
|
||||
company="Acme Corp",
|
||||
title="Software Engineer",
|
||||
url="https://example.com/dup/2",
|
||||
description="We need a Python developer with Docker experience.",
|
||||
)
|
||||
|
||||
# Run cluster assignment manually
|
||||
from app.main import _assign_cluster_id
|
||||
from app.db import repo_app
|
||||
all_postings = repo_app.list_postings()
|
||||
for p in all_postings:
|
||||
_assign_cluster_id(p["id"])
|
||||
|
||||
postings = repo_app.list_postings()
|
||||
cluster_ids = [p["cluster_id"] for p in postings if p["cluster_id"]]
|
||||
# Both should have the same cluster_id
|
||||
assert len(cluster_ids) >= 2
|
||||
assert len(set(cluster_ids)) == 1
|
||||
|
||||
def test_different_postings_different_clusters(self, client):
|
||||
"""Completely different postings get different cluster_ids."""
|
||||
_create_posting_direct(
|
||||
company="CompanyA",
|
||||
title="Chef",
|
||||
url="https://example.com/diff/1",
|
||||
description="Looking for an experienced chef.",
|
||||
)
|
||||
_create_posting_direct(
|
||||
company="CompanyB",
|
||||
title="Pilot",
|
||||
url="https://example.com/diff/2",
|
||||
description="Commercial airline pilot needed.",
|
||||
)
|
||||
|
||||
from app.main import _assign_cluster_id
|
||||
from app.db import repo_app
|
||||
all_postings = repo_app.list_postings()
|
||||
for p in all_postings:
|
||||
_assign_cluster_id(p["id"])
|
||||
|
||||
postings = repo_app.list_postings()
|
||||
cluster_ids = [p["cluster_id"] for p in postings if p["cluster_id"]]
|
||||
if len(cluster_ids) >= 2:
|
||||
assert len(set(cluster_ids)) >= 2
|
||||
|
||||
def test_cluster_id_in_get_postings(self, client):
|
||||
"""GET /postings returns cluster_id field."""
|
||||
_create_posting_direct(
|
||||
company="TestCo",
|
||||
title="Dev",
|
||||
url="https://example.com/field/1",
|
||||
description="Test description",
|
||||
)
|
||||
|
||||
resp = client.get("/api/postings")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert len(data) >= 1
|
||||
assert "cluster_id" in data[0]
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Cluster stability across re-imports (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestClusterStability:
|
||||
def test_reimport_preserves_cluster_id(self, client):
|
||||
"""Re-importing a posting (same URL) keeps the cluster_id stable."""
|
||||
# First import
|
||||
posting1 = _create_posting_direct(
|
||||
company="StableCorp",
|
||||
title="Engineer",
|
||||
url="https://example.com/stable/1",
|
||||
description="Stable description for engineer role.",
|
||||
)
|
||||
|
||||
from app.main import _assign_cluster_id
|
||||
_assign_cluster_id(posting1["id"])
|
||||
|
||||
p1 = repo_app.get_job_posting(posting1["id"])
|
||||
original_cluster_id = p1.get("cluster_id")
|
||||
|
||||
# Re-import: same URL -> ON CONFLICT DO UPDATE, returns same row
|
||||
posting2 = repo_app.create_job_posting(
|
||||
source="manual_url",
|
||||
url="https://example.com/stable/1",
|
||||
company="StableCorp",
|
||||
title="Engineer",
|
||||
description="Stable description for engineer role.",
|
||||
raw={},
|
||||
)
|
||||
assert posting2["id"] == posting1["id"]
|
||||
|
||||
p2 = repo_app.get_job_posting(posting2["id"])
|
||||
assert p2.get("cluster_id") == original_cluster_id
|
||||
|
||||
def test_new_duplicate_joins_existing_cluster(self, client):
|
||||
"""A new posting that's a duplicate of an existing one joins its cluster_id."""
|
||||
# First posting
|
||||
p1 = _create_posting_direct(
|
||||
company="JoinCorp",
|
||||
title="Backend Developer",
|
||||
url="https://example.com/join/1",
|
||||
description="Python developer with PostgreSQL and Docker.",
|
||||
)
|
||||
from app.main import _assign_cluster_id
|
||||
_assign_cluster_id(p1["id"])
|
||||
|
||||
p1_row = repo_app.get_job_posting(p1["id"])
|
||||
original_cluster = p1_row.get("cluster_id")
|
||||
|
||||
# Second posting (duplicate)
|
||||
p2 = _create_posting_direct(
|
||||
company="JoinCorp",
|
||||
title="Backend Developer",
|
||||
url="https://example.com/join/2",
|
||||
description="Python developer with PostgreSQL and Docker.",
|
||||
)
|
||||
_assign_cluster_id(p2["id"])
|
||||
|
||||
p2_row = repo_app.get_job_posting(p2["id"])
|
||||
assert p2_row.get("cluster_id") == original_cluster
|
||||
|
||||
def test_third_duplicate_extends_cluster(self, client):
|
||||
"""Third duplicate posting joins the same cluster as the first two."""
|
||||
desc = "Senior Python developer with FastAPI and PostgreSQL experience."
|
||||
p1 = _create_posting_direct(
|
||||
company="ExtCorp", title="Senior Python Developer",
|
||||
url="https://example.com/ext/1", description=desc,
|
||||
)
|
||||
from app.main import _assign_cluster_id
|
||||
_assign_cluster_id(p1["id"])
|
||||
c1 = repo_app.get_job_posting(p1["id"]).get("cluster_id")
|
||||
|
||||
p2 = _create_posting_direct(
|
||||
company="ExtCorp", title="Senior Python Developer",
|
||||
url="https://example.com/ext/2", description=desc,
|
||||
)
|
||||
_assign_cluster_id(p2["id"])
|
||||
|
||||
p3 = _create_posting_direct(
|
||||
company="ExtCorp", title="Senior Python Developer",
|
||||
url="https://example.com/ext/3", description=desc,
|
||||
)
|
||||
_assign_cluster_id(p3["id"])
|
||||
|
||||
c2 = repo_app.get_job_posting(p2["id"]).get("cluster_id")
|
||||
c3 = repo_app.get_job_posting(p3["id"]).get("cluster_id")
|
||||
assert c1 == c2 == c3
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# GET /clusters endpoint shape (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestClustersEndpoint:
|
||||
def test_clusters_returns_list(self, client):
|
||||
"""GET /clusters returns a list of cluster objects."""
|
||||
resp = client.get("/api/clusters")
|
||||
assert resp.status_code == 200
|
||||
assert isinstance(resp.json(), list)
|
||||
|
||||
def test_clusters_shape(self, client):
|
||||
"""Each cluster has cluster_id and postings with required fields."""
|
||||
# Create two duplicate postings
|
||||
desc = "Full stack developer with React and Node.js experience needed."
|
||||
p1 = _create_posting_direct(
|
||||
company="ShapeCorp", title="Full Stack Developer",
|
||||
url="https://example.com/shape/1", description=desc,
|
||||
)
|
||||
from app.main import _assign_cluster_id
|
||||
_assign_cluster_id(p1["id"])
|
||||
|
||||
p2 = _create_posting_direct(
|
||||
company="ShapeCorp", title="Full Stack Developer",
|
||||
url="https://example.com/shape/2", description=desc,
|
||||
)
|
||||
_assign_cluster_id(p2["id"])
|
||||
|
||||
resp = client.get("/api/clusters")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert len(data) >= 1
|
||||
|
||||
cluster = data[0]
|
||||
assert "cluster_id" in cluster
|
||||
assert "postings" in cluster
|
||||
assert isinstance(cluster["postings"], list)
|
||||
assert len(cluster["postings"]) >= 2
|
||||
|
||||
p = cluster["postings"][0]
|
||||
assert "id" in p
|
||||
assert "title" in p
|
||||
assert "company" in p
|
||||
assert "source" in p
|
||||
assert "url" in p
|
||||
assert "score" in p
|
||||
|
||||
def test_clusters_empty_when_no_cluster_ids(self, client):
|
||||
"""GET /clusters returns empty list when no postings have cluster_ids."""
|
||||
# Create a posting but don't assign cluster_id
|
||||
_create_posting_direct(
|
||||
company="NoCluster", title="Dev",
|
||||
url="https://example.com/nocluster/1", description="Something unique.",
|
||||
)
|
||||
# Don't call _assign_cluster_id
|
||||
|
||||
resp = client.get("/api/clusters")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
# Should be empty since no cluster_ids assigned
|
||||
assert len(data) == 0
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Tailor CV happy path (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestTailorCvHappyPath:
|
||||
def test_tailor_cv_returns_artifact_and_coverage(self, client):
|
||||
"""POST /applications/{id}/tailor-cv returns artifact_id, change_log, keyword_coverage."""
|
||||
app_id = _create_app_with_profile_and_sections(client)
|
||||
|
||||
resp = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
|
||||
assert "artifact_id" in data
|
||||
assert data["artifact_id"] is not None
|
||||
assert "change_log" in data
|
||||
assert isinstance(data["change_log"], list)
|
||||
assert len(data["change_log"]) >= 1
|
||||
assert "keyword_coverage" in data
|
||||
kc = data["keyword_coverage"]
|
||||
assert "matched" in kc
|
||||
assert "missing" in kc
|
||||
assert "ratio" in kc
|
||||
|
||||
def test_tailor_cv_artifact_stored(self, client):
|
||||
"""The tailored CV artifact appears in the application's artifacts list."""
|
||||
app_id = _create_app_with_profile_and_sections(client)
|
||||
|
||||
resp = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp.status_code == 200
|
||||
artifact_id = resp.json()["artifact_id"]
|
||||
|
||||
artifacts = client.get(f"/api/applications/{app_id}/artifacts").json()
|
||||
cv_artifacts = [a for a in artifacts if a["kind"] == "cv"]
|
||||
assert len(cv_artifacts) >= 1
|
||||
assert any(a["id"] == artifact_id for a in cv_artifacts)
|
||||
ai_artifact = [a for a in cv_artifacts if a["id"] == artifact_id][0]
|
||||
assert ai_artifact["origin"] == "ai_drafted"
|
||||
|
||||
def test_tailor_cv_404_nonexistent(self, client):
|
||||
"""Tailor CV on nonexistent application returns 404."""
|
||||
resp = client.post("/api/applications/00000000-0000-0000-0000-000000000000/tailor-cv")
|
||||
assert resp.status_code == 404
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Tailor CV hallucination rejection (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestTailorCvHallucinationGuard:
|
||||
def test_hallucination_rejection_502(self, client, monkeypatch):
|
||||
"""When the tailor output has bullets with no source mapping, return 502."""
|
||||
import app.llm as llm_mod
|
||||
|
||||
app_id = _create_app_with_profile_and_sections(client)
|
||||
fabricated = {
|
||||
"tailored_cv": {
|
||||
"summary": "Developer",
|
||||
"skills": ["Python"],
|
||||
"experience": [
|
||||
{
|
||||
"company": "FakeCorp",
|
||||
"role": "Fake Role",
|
||||
"bullets": [
|
||||
"Completely fabricated achievement that has no overlap with any source bullet xyzqwerty",
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
"change_log": [{"action": "invented", "detail": "Made up a bullet"}],
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
llm_mod, "run_task",
|
||||
lambda task, prompt, *a, **k: fabricated,
|
||||
)
|
||||
resp = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp.status_code == 502
|
||||
detail = resp.json()["detail"]
|
||||
assert "hallucination_guard" in str(detail)
|
||||
|
||||
def test_hallucination_rejection_with_empty_source_bullets(self, client):
|
||||
"""When there are no source bullets, hallucination guard is not triggered (no source to map to)."""
|
||||
# Create profile with no sections
|
||||
client.get("/api/profile")
|
||||
client.put("/api/profile", json={
|
||||
"full_name": "Test User",
|
||||
"email": "test@test.com",
|
||||
})
|
||||
# Create posting + app
|
||||
posting = _create_posting_direct(
|
||||
company="NoSourceCo",
|
||||
title="Dev",
|
||||
url="https://example.com/nosource/1",
|
||||
description="Python developer",
|
||||
)
|
||||
apps = repo_app.list_applications()
|
||||
app_id = None
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == posting["id"]:
|
||||
app_id = a["id"]
|
||||
break
|
||||
|
||||
# When no source bullets exist, the guard doesn't trigger (source_bullets is empty)
|
||||
resp = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
# Should succeed since source_bullets is empty -> guard not triggered
|
||||
assert resp.status_code == 200
|
||||
|
||||
def test_hallucination_rejection_preserves_existing_output(self, client, monkeypatch):
|
||||
"""After a 502 hallucination rejection, a subsequent valid call works."""
|
||||
import app.llm as llm_mod
|
||||
|
||||
app_id = _create_app_with_profile_and_sections(client)
|
||||
fabricated = {
|
||||
"tailored_cv": {
|
||||
"summary": "Dev",
|
||||
"skills": ["Python"],
|
||||
"experience": [
|
||||
{
|
||||
"company": "X",
|
||||
"role": "X",
|
||||
"bullets": ["Fabricated xyzqwerty zzz new content"],
|
||||
},
|
||||
],
|
||||
},
|
||||
"change_log": [],
|
||||
}
|
||||
with monkeypatch.context() as mp:
|
||||
mp.setattr(
|
||||
llm_mod, "run_task",
|
||||
lambda task, prompt, *a, **k: fabricated,
|
||||
)
|
||||
resp1 = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp1.status_code == 502
|
||||
|
||||
# Default prompt-aware mock is guard-safe -> succeeds
|
||||
resp2 = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp2.status_code == 200
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Keyword coverage numbers vs fixture (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestKeywordCoverage:
|
||||
def test_keyword_coverage_has_matched_and_missing(self, client):
|
||||
"""Keyword coverage from tailor-cv contains matched and missing keywords."""
|
||||
app_id = _create_app_with_profile_and_sections(
|
||||
client,
|
||||
description="Python FastAPI PostgreSQL Docker Kubernetes AWS Java Spring",
|
||||
)
|
||||
|
||||
resp = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp.status_code == 200
|
||||
kc = resp.json()["keyword_coverage"]
|
||||
|
||||
assert "matched" in kc
|
||||
assert "missing" in kc
|
||||
assert "ratio" in kc
|
||||
assert isinstance(kc["matched"], list)
|
||||
assert isinstance(kc["missing"], list)
|
||||
assert isinstance(kc["ratio"], (int, float))
|
||||
assert 0.0 <= kc["ratio"] <= 1.0
|
||||
|
||||
def test_keyword_coverage_ratio_is_reasonable(self, client):
|
||||
"""With matching CV keywords, coverage ratio should be > 0."""
|
||||
app_id = _create_app_with_profile_and_sections(
|
||||
client,
|
||||
description="Python FastAPI PostgreSQL Docker Kubernetes AWS",
|
||||
)
|
||||
|
||||
resp = client.post(f"/api/applications/{app_id}/tailor-cv")
|
||||
assert resp.status_code == 200
|
||||
kc = resp.json()["keyword_coverage"]
|
||||
|
||||
# The mock CV has Python, Fast API, PostgreSQL, Docker, Kubernetes, AWS
|
||||
# which should match most posting keywords
|
||||
assert kc["ratio"] > 0.0
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Deadline persisted on scoring (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestDeadlinePersisted:
|
||||
def test_deadline_extracted_on_single_score(self, client):
|
||||
"""Scoring a posting also runs deadline_extract and persists apply_by."""
|
||||
posting = _create_posting_direct(
|
||||
company="DeadlineCo",
|
||||
title="Dev",
|
||||
url="https://example.com/deadline/1",
|
||||
description="Apply by 2026-12-31.",
|
||||
)
|
||||
apps = repo_app.list_applications()
|
||||
app_id = None
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == posting["id"]:
|
||||
app_id = a["id"]
|
||||
break
|
||||
|
||||
# Mock deadline_extract to return a date
|
||||
import app.llm as llm_mod
|
||||
original = llm_mod.MOCK_OUTPUTS.get("deadline_extract", {}).copy()
|
||||
try:
|
||||
llm_mod.MOCK_OUTPUTS["deadline_extract"] = {"apply_by": "2026-12-31"}
|
||||
resp = client.post(f"/api/postings/{posting['id']}/score")
|
||||
assert resp.status_code == 200
|
||||
|
||||
# Verify apply_by was persisted
|
||||
p = repo_app.get_job_posting(posting["id"])
|
||||
assert p["apply_by"] == "2026-12-31"
|
||||
finally:
|
||||
llm_mod.MOCK_OUTPUTS["deadline_extract"] = original
|
||||
|
||||
def test_deadline_null_does_not_persist(self, client):
|
||||
"""When deadline_extract returns null apply_by, nothing is persisted."""
|
||||
posting = _create_posting_direct(
|
||||
company="NoDeadlineCo",
|
||||
title="Dev",
|
||||
url="https://example.com/nodeadline/1",
|
||||
description="No deadline mentioned.",
|
||||
)
|
||||
|
||||
# Default mock returns apply_by=None
|
||||
resp = client.post(f"/api/postings/{posting['id']}/score")
|
||||
assert resp.status_code == 200
|
||||
|
||||
p = repo_app.get_job_posting(posting["id"])
|
||||
assert p["apply_by"] is None
|
||||
|
||||
def test_deadline_extracted_on_batch_score(self, client):
|
||||
"""Batch scoring also runs deadline_extract and persists apply_by."""
|
||||
posting = _create_posting_direct(
|
||||
company="BatchDeadlineCo",
|
||||
title="Dev",
|
||||
url="https://example.com/batchdeadline/1",
|
||||
description="Apply by 2026-11-15.",
|
||||
)
|
||||
apps = repo_app.list_applications()
|
||||
app_id = None
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == posting["id"]:
|
||||
app_id = a["id"]
|
||||
break
|
||||
|
||||
import app.llm as llm_mod
|
||||
original = llm_mod.MOCK_OUTPUTS.get("deadline_extract", {}).copy()
|
||||
try:
|
||||
llm_mod.MOCK_OUTPUTS["deadline_extract"] = {"apply_by": "2026-11-15"}
|
||||
resp = client.post("/api/scoring/batch", json={"application_ids": [app_id]})
|
||||
assert resp.status_code == 200
|
||||
|
||||
p = repo_app.get_job_posting(posting["id"])
|
||||
assert p["apply_by"] == "2026-11-15"
|
||||
finally:
|
||||
llm_mod.MOCK_OUTPUTS["deadline_extract"] = original
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# /today deadlines filter window (3 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestTodayDeadlines:
|
||||
def test_today_returns_deadlines_field(self, client):
|
||||
"""/today response includes deadlines field."""
|
||||
resp = client.get("/api/today")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert "deadlines" in data
|
||||
assert isinstance(data["deadlines"], list)
|
||||
|
||||
def test_today_deadlines_within_7_days(self, client):
|
||||
"""Deadlines within next 7 days appear in /today."""
|
||||
# Create posting with apply_by in 3 days
|
||||
posting = _create_posting_direct(
|
||||
company="WeekCo",
|
||||
title="Dev",
|
||||
url="https://example.com/week/1",
|
||||
description="Dev role.",
|
||||
)
|
||||
apps = repo_app.list_applications()
|
||||
app_id = None
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == posting["id"]:
|
||||
app_id = a["id"]
|
||||
break
|
||||
|
||||
future_date = date.today() + timedelta(days=3)
|
||||
repo_app.update_posting_apply_by(posting["id"], future_date)
|
||||
|
||||
resp = client.get("/api/today")
|
||||
assert resp.status_code == 200
|
||||
deadlines = resp.json()["deadlines"]
|
||||
assert len(deadlines) >= 1
|
||||
|
||||
dl = [d for d in deadlines if d["application_id"] == app_id]
|
||||
assert len(dl) == 1
|
||||
assert dl[0]["title"] == "Dev"
|
||||
assert dl[0]["company"] == "WeekCo"
|
||||
assert dl[0]["apply_by"] == future_date.isoformat()
|
||||
|
||||
def test_today_deadlines_excludes_beyond_7_days(self, client):
|
||||
"""Deadlines beyond 7 days do NOT appear in /today."""
|
||||
posting = _create_posting_direct(
|
||||
company="FarCo",
|
||||
title="Dev",
|
||||
url="https://example.com/far/1",
|
||||
description="Dev role.",
|
||||
)
|
||||
apps = repo_app.list_applications()
|
||||
app_id = None
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == posting["id"]:
|
||||
app_id = a["id"]
|
||||
break
|
||||
|
||||
far_date = date.today() + timedelta(days=30)
|
||||
repo_app.update_posting_apply_by(posting["id"], far_date)
|
||||
|
||||
resp = client.get("/api/today")
|
||||
assert resp.status_code == 200
|
||||
deadlines = resp.json()["deadlines"]
|
||||
far_deadlines = [d for d in deadlines if d["application_id"] == app_id]
|
||||
assert len(far_deadlines) == 0
|
||||
|
||||
|
||||
# ========================================================================
|
||||
# Extra integration tests (2 tests)
|
||||
# ========================================================================
|
||||
|
||||
class TestExtraIntegration:
|
||||
def test_get_postings_has_apply_by_field(self, client):
|
||||
"""GET /postings includes apply_by field."""
|
||||
posting = _create_posting_direct(
|
||||
company="ApplyByCo",
|
||||
title="Dev",
|
||||
url="https://example.com/applyby/1",
|
||||
description="Dev role.",
|
||||
)
|
||||
repo_app.update_posting_apply_by(posting["id"], date.today() + timedelta(days=5))
|
||||
|
||||
resp = client.get("/api/postings")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
p = [x for x in data if x["id"] == posting["id"]][0]
|
||||
assert p["apply_by"] is not None
|
||||
|
||||
def test_clusters_sorted_by_best_score_desc(self, client):
|
||||
"""Clusters are sorted by best score descending."""
|
||||
# Create cluster 1 with a high-score posting
|
||||
desc1 = "Python developer with PostgreSQL and Docker experience needed."
|
||||
p1 = _create_posting_direct(
|
||||
company="HighScoreCo", title="Python Developer",
|
||||
url="https://example.com/sort/1", description=desc1,
|
||||
)
|
||||
from app.main import _assign_cluster_id
|
||||
_assign_cluster_id(p1["id"])
|
||||
|
||||
# Score p1
|
||||
apps = repo_app.list_applications()
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == p1["id"]:
|
||||
repo_app.update_application_score(a["id"], 90, {"factors": {}})
|
||||
break
|
||||
|
||||
# Create cluster 2 with a low-score posting
|
||||
desc2 = "Marketing specialist for social media campaigns and content creation."
|
||||
p2 = _create_posting_direct(
|
||||
company="LowScoreCo", title="Marketing Specialist",
|
||||
url="https://example.com/sort/2", description=desc2,
|
||||
)
|
||||
_assign_cluster_id(p2["id"])
|
||||
|
||||
for a in apps:
|
||||
if a["job_posting_id"] == p2["id"]:
|
||||
repo_app.update_application_score(a["id"], 30, {"factors": {}})
|
||||
break
|
||||
|
||||
resp = client.get("/api/clusters")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
if len(data) >= 2:
|
||||
# Best scores should be descending
|
||||
best_scores = []
|
||||
for c in data:
|
||||
scores = [p.get("score") or 0 for p in c["postings"]]
|
||||
best_scores.append(max(scores) if scores else 0)
|
||||
assert best_scores[0] >= best_scores[1]
|
||||
|
|
@ -279,22 +279,14 @@ class TestInterviewPrep:
|
|||
|
||||
class TestSeedDemo:
|
||||
def test_seed_demo_creates_data(self, client):
|
||||
"""POST /concierge/seed-demo creates profile, postings, applications, and extended data."""
|
||||
"""POST /concierge/seed-demo creates profile, postings, applications."""
|
||||
resp = client.post("/api/concierge/seed-demo")
|
||||
assert resp.status_code == 200
|
||||
data = resp.json()
|
||||
assert data["profile"] == "Demo Demosson"
|
||||
assert data["postings"] == 6
|
||||
assert data["applications"] == 6
|
||||
assert data["sections"] == 4
|
||||
# 6 standalone + 3 agency + 2 deadline + 1 redflag + 1 interviewing = 13 postings
|
||||
assert data["postings"] == 13
|
||||
# 6 standalone + 3 agency + 2 deadline + 1 redflag + 1 interviewing = 13 applications
|
||||
assert data["applications"] == 13
|
||||
assert data["clusters"] >= 1
|
||||
assert data["deadlines"] >= 2
|
||||
assert data["suggestions"] == 2
|
||||
assert data["notifications"] == 3
|
||||
assert data["task_runs"] == 6
|
||||
assert data["cv_artifacts"] >= 1
|
||||
|
||||
def test_seed_demo_idempotent(self, client):
|
||||
"""Running seed-demo twice returns the same counts."""
|
||||
|
|
@ -306,12 +298,6 @@ class TestSeedDemo:
|
|||
assert resp2.json()["postings"] == resp1.json()["postings"]
|
||||
assert resp2.json()["applications"] == resp1.json()["applications"]
|
||||
assert resp2.json()["sections"] == resp1.json()["sections"]
|
||||
assert resp2.json()["clusters"] == resp1.json()["clusters"]
|
||||
assert resp2.json()["deadlines"] == resp1.json()["deadlines"]
|
||||
assert resp2.json()["suggestions"] == resp1.json()["suggestions"]
|
||||
assert resp2.json()["notifications"] == resp1.json()["notifications"]
|
||||
assert resp2.json()["task_runs"] == resp1.json()["task_runs"]
|
||||
assert resp2.json()["cv_artifacts"] == resp1.json()["cv_artifacts"]
|
||||
|
||||
def test_seed_demo_has_nudge_candidate(self, client):
|
||||
"""After seeding, /today should show a nudge for the backdated sent app."""
|
||||
|
|
@ -327,104 +313,6 @@ class TestSeedDemo:
|
|||
digest = resp.json()["digest"]
|
||||
assert len(digest) >= 1
|
||||
|
||||
# --- WS1: Extended seed demo tests (8 new) ---
|
||||
|
||||
def test_seed_demo_agency_cluster_present(self, client):
|
||||
"""Seed creates a 3-posting agency cluster with the same cluster_id."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
postings = client.get("/api/postings").json()
|
||||
agency_names = {"Aderanto AB", "Wise IT", "TechTalent Nord"}
|
||||
agency_postings = [p for p in postings if p.get("company") in agency_names]
|
||||
assert len(agency_postings) == 3
|
||||
cluster_ids = {p["cluster_id"] for p in agency_postings if p.get("cluster_id")}
|
||||
assert len(cluster_ids) == 1, f"Expected 1 cluster_id, got {cluster_ids}"
|
||||
|
||||
def test_seed_demo_deadlines_populated(self, client):
|
||||
"""Seed creates at least 2 postings with apply_by in the next 7 days."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
resp = client.get("/api/today")
|
||||
assert resp.status_code == 200
|
||||
deadlines = resp.json().get("deadlines", [])
|
||||
assert len(deadlines) >= 2
|
||||
for d in deadlines:
|
||||
assert d["apply_by"] is not None
|
||||
|
||||
def test_seed_demo_red_flag_rationale(self, client):
|
||||
"""Seed creates an application with red_flags in its score_rationale."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
apps = client.get("/api/applications").json()
|
||||
red_flag_apps = [
|
||||
a for a in apps
|
||||
if a.get("score_rationale") and isinstance(a["score_rationale"], dict)
|
||||
and "red_flags" in a["score_rationale"]
|
||||
]
|
||||
assert len(red_flag_apps) >= 1
|
||||
red_flags = red_flag_apps[0]["score_rationale"]["red_flags"]
|
||||
assert isinstance(red_flags, list)
|
||||
assert any("unpaid trial" in str(rf).lower() for rf in red_flags)
|
||||
|
||||
def test_seed_demo_has_interviewing_application(self, client):
|
||||
"""Seed creates at least one application in interviewing state."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
apps = client.get("/api/applications").json()
|
||||
interviewing = [a for a in apps if a["state"] == "interviewing"]
|
||||
assert len(interviewing) >= 1
|
||||
|
||||
def test_seed_demo_has_cover_letter_artifact(self, client):
|
||||
"""Seed creates a cover_letter artifact (origin user_drafted) with Swedish text on the approved app."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
apps = client.get("/api/applications").json()
|
||||
for a in apps:
|
||||
artifacts = client.get(f"/api/applications/{a['id']}/artifacts").json()
|
||||
for art in artifacts:
|
||||
if art["kind"] == "cover_letter" and art["origin"] == "user_drafted":
|
||||
return
|
||||
assert False, "No user_drafted cover_letter artifact found"
|
||||
|
||||
def test_seed_demo_suggestions_present(self, client):
|
||||
"""Seed creates 2 pending email_suggestion rows with expected classifications."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
suggestions = client.get("/api/suggestions").json()
|
||||
assert len(suggestions) == 2
|
||||
classifications = {s["classification"] for s in suggestions}
|
||||
assert "interview_invite" in classifications
|
||||
assert "question" in classifications
|
||||
# Verify the interview_invite comes from recruiter@festina-demo.se
|
||||
interview_suggestion = [s for s in suggestions if s["classification"] == "interview_invite"][0]
|
||||
assert interview_suggestion["mailbox_from"] == "recruiter@festina-demo.se"
|
||||
|
||||
def test_seed_demo_notification_log_present(self, client):
|
||||
"""Seed creates 3 notification_log rows: 2 delivered, 1 webhook failed."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
resp = client.get("/api/notifications/log")
|
||||
assert resp.status_code == 200
|
||||
logs = resp.json()
|
||||
assert len(logs) == 3
|
||||
# At least one delivered (daily_digest or email_suggestion)
|
||||
delivered = [l for l in logs if l["delivered"] is True]
|
||||
assert len(delivered) >= 2
|
||||
# At least one webhook failed with error text
|
||||
failed = [l for l in logs if l["delivered"] is False]
|
||||
assert len(failed) >= 1
|
||||
assert failed[0]["error"] is not None
|
||||
assert len(failed[0]["error"]) > 0
|
||||
|
||||
def test_seed_demo_task_run_telemetry_variance(self, client):
|
||||
"""Seed creates 6 task_run rows across multiple providers and models."""
|
||||
client.post("/api/concierge/seed-demo")
|
||||
resp = client.get("/api/telemetry/tasks")
|
||||
assert resp.status_code == 200
|
||||
tasks = resp.json()
|
||||
assert len(tasks) == 6
|
||||
providers = {t["provider"] for t in tasks}
|
||||
models = {t["model"] for t in tasks}
|
||||
assert len(providers) >= 3, f"Expected >= 3 providers, got {providers}"
|
||||
assert len(models) >= 4, f"Expected >= 4 models, got {models}"
|
||||
# Verify cost variance for CostDisplay
|
||||
costs = [t["cost_usd"] for t in tasks if t["cost_usd"] is not None]
|
||||
assert len(costs) >= 2
|
||||
assert max(costs) > min(costs)
|
||||
|
||||
|
||||
# --- SMTP Transport ---
|
||||
|
||||
|
|
|
|||
|
|
@ -1,14 +0,0 @@
|
|||
# Web production image: build the SPA, serve via nginx with SPA fallback
|
||||
FROM node:22-alpine AS build
|
||||
WORKDIR /w
|
||||
ARG VITE_API_BASE=http://api:8000/api
|
||||
ENV VITE_API_BASE=$VITE_API_BASE
|
||||
COPY apps/web/package.json apps/web/package-lock.json ./
|
||||
RUN npm ci --no-audit --no-fund
|
||||
COPY apps/web ./
|
||||
RUN npm run build
|
||||
|
||||
FROM nginx:1.27-alpine
|
||||
COPY apps/web/nginx.conf /etc/nginx/conf.d/default.conf
|
||||
COPY --from=build /w/dist /usr/share/nginx/html
|
||||
EXPOSE 80
|
||||
|
|
@ -1,15 +0,0 @@
|
|||
server {
|
||||
listen 80;
|
||||
server_name _;
|
||||
root /usr/share/nginx/html;
|
||||
index index.html;
|
||||
|
||||
location /api/ {
|
||||
proxy_pass http://api:8000/api/;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
location / {
|
||||
try_files $uri $uri/ /index.html;
|
||||
}
|
||||
}
|
||||
|
|
@ -7,25 +7,20 @@ import type {
|
|||
Approval,
|
||||
Artifact,
|
||||
BatchScoringResponse,
|
||||
Cluster,
|
||||
CoverLetterResponse,
|
||||
CritiqueComment,
|
||||
CvImportConfirmResponse,
|
||||
CvImportResponse,
|
||||
CvSection,
|
||||
DemoSeedResponse,
|
||||
EmailSuggestion,
|
||||
InterviewPrepResponse,
|
||||
JobPosting,
|
||||
NotificationLogEntry,
|
||||
PostingsFetchResponse,
|
||||
Profile,
|
||||
RenderCvResponse,
|
||||
ScoreResponse,
|
||||
TailorCvResponse,
|
||||
TaskRun,
|
||||
TodayResponse,
|
||||
TodayResponseV11
|
||||
TodayResponse
|
||||
} from '@/types'
|
||||
|
||||
const API_BASE: string =
|
||||
|
|
@ -205,8 +200,8 @@ export function batchScore(applicationIds: string[]): Promise<BatchScoringRespon
|
|||
})
|
||||
}
|
||||
|
||||
export function getToday(): Promise<TodayResponseV11> {
|
||||
return request<TodayResponseV11>('/today')
|
||||
export function getToday(): Promise<TodayResponse> {
|
||||
return request<TodayResponse>('/today')
|
||||
}
|
||||
|
||||
export function interviewPrep(applicationId: string): Promise<InterviewPrepResponse> {
|
||||
|
|
@ -219,34 +214,6 @@ export function seedDemo(): Promise<DemoSeedResponse> {
|
|||
return request<DemoSeedResponse>('/concierge/seed-demo', { method: 'POST' })
|
||||
}
|
||||
|
||||
// --- v1.1 additions (wave A/B) ---
|
||||
|
||||
export function getSuggestions(): Promise<EmailSuggestion[]> {
|
||||
return request<EmailSuggestion[]>('/suggestions')
|
||||
}
|
||||
|
||||
export function acceptSuggestion(id: string): Promise<EmailSuggestion> {
|
||||
return request<EmailSuggestion>(`/suggestions/${id}/accept`, { method: 'POST' })
|
||||
}
|
||||
|
||||
export function dismissSuggestion(id: string): Promise<EmailSuggestion> {
|
||||
return request<EmailSuggestion>(`/suggestions/${id}/dismiss`, { method: 'POST' })
|
||||
}
|
||||
|
||||
export function getNotificationLog(): Promise<NotificationLogEntry[]> {
|
||||
return request<NotificationLogEntry[]>('/notifications/log')
|
||||
}
|
||||
|
||||
export function getClusters(): Promise<Cluster[]> {
|
||||
return request<Cluster[]>('/clusters')
|
||||
}
|
||||
|
||||
export function tailorCv(applicationId: string): Promise<TailorCvResponse> {
|
||||
return request<TailorCvResponse>(`/applications/${applicationId}/tailor-cv`, {
|
||||
method: 'POST'
|
||||
})
|
||||
}
|
||||
|
||||
// Re-export types for convenience
|
||||
export type {
|
||||
AiAssistResponse,
|
||||
|
|
@ -254,23 +221,18 @@ export type {
|
|||
Approval,
|
||||
Artifact,
|
||||
BatchScoringResponse,
|
||||
Cluster,
|
||||
CoverLetterResponse,
|
||||
CritiqueComment,
|
||||
CvImportConfirmResponse,
|
||||
CvImportResponse,
|
||||
CvSection,
|
||||
DemoSeedResponse,
|
||||
EmailSuggestion,
|
||||
InterviewPrepResponse,
|
||||
JobPosting,
|
||||
NotificationLogEntry,
|
||||
PostingsFetchResponse,
|
||||
Profile,
|
||||
RenderCvResponse,
|
||||
ScoreResponse,
|
||||
TailorCvResponse,
|
||||
TaskRun,
|
||||
TodayResponse,
|
||||
TodayResponseV11
|
||||
TodayResponse
|
||||
}
|
||||
|
|
@ -18,23 +18,6 @@ const totalCost = computed(() =>
|
|||
)
|
||||
const hasCost = computed(() => tasks.value.some((t) => t.cost != null))
|
||||
|
||||
// Group task runs by model name (proxy for provider) and compute totals per group
|
||||
const byProvider = computed(() => {
|
||||
const map = new Map<string, { model: string; tokensIn: number; tokensOut: number; cost: number; count: number }>()
|
||||
for (const t of tasks.value) {
|
||||
const key = t.model || 'unknown'
|
||||
if (!map.has(key)) {
|
||||
map.set(key, { model: key, tokensIn: 0, tokensOut: 0, cost: 0, count: 0 })
|
||||
}
|
||||
const entry = map.get(key)!
|
||||
entry.tokensIn += t.tokens_in ?? 0
|
||||
entry.tokensOut += t.tokens_out ?? 0
|
||||
entry.cost += t.cost ?? 0
|
||||
entry.count += 1
|
||||
}
|
||||
return Array.from(map.values()).sort((a, b) => b.cost - a.cost)
|
||||
})
|
||||
|
||||
async function loadTasks() {
|
||||
try {
|
||||
tasks.value = await api.getTelemetryTasks()
|
||||
|
|
@ -67,31 +50,6 @@ onMounted(loadTasks)
|
|||
<span class="font-medium">{{ totalCost.toFixed(4) }}</span>
|
||||
</div>
|
||||
<div class="text-xs text-gray-400 mt-1">{{ tasks.length }} task runs</div>
|
||||
|
||||
<!-- Provider breakdown -->
|
||||
<div v-if="byProvider.length > 0" class="mt-3 border-t border-gray-100 pt-2">
|
||||
<div class="text-xs font-medium text-gray-500 mb-1">By Provider</div>
|
||||
<table class="w-full text-xs" data-testid="provider-breakdown">
|
||||
<thead class="text-left text-gray-400">
|
||||
<tr>
|
||||
<th class="py-1">Model</th>
|
||||
<th class="py-1 text-right">In</th>
|
||||
<th class="py-1 text-right">Out</th>
|
||||
<th class="py-1 text-right">Cost</th>
|
||||
<th class="py-1 text-right">Runs</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr v-for="p in byProvider" :key="p.model" class="border-t border-gray-50">
|
||||
<td class="py-1">{{ p.model }}</td>
|
||||
<td class="py-1 text-right">{{ p.tokensIn.toLocaleString() }}</td>
|
||||
<td class="py-1 text-right">{{ p.tokensOut.toLocaleString() }}</td>
|
||||
<td class="py-1 text-right">{{ p.cost.toFixed(4) }}</td>
|
||||
<td class="py-1 text-right">{{ p.count }}</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</template>
|
||||
|
|
@ -66,10 +66,8 @@ export interface Application {
|
|||
notes: string
|
||||
state_changed_at: string
|
||||
created_at: string
|
||||
// joined posting info (flat fields from GET /applications)
|
||||
company?: string | null
|
||||
title?: string | null
|
||||
location?: string | null
|
||||
// joined posting info (from GET /applications)
|
||||
posting?: JobPosting
|
||||
}
|
||||
|
||||
export type ArtifactKind = 'cv' | 'cover_letter' | 'email' | 'other'
|
||||
|
|
@ -211,74 +209,3 @@ export interface TaskRun {
|
|||
export interface RedFlagsMap {
|
||||
[applicationId: string]: string[]
|
||||
}
|
||||
|
||||
// --- v1.1 additions (wave A/B) ---
|
||||
|
||||
export interface TodayDeadline {
|
||||
application_id: string
|
||||
title: string
|
||||
company: string
|
||||
apply_by: string
|
||||
}
|
||||
|
||||
export interface TodayResponseV11 extends TodayResponse {
|
||||
deadlines?: TodayDeadline[]
|
||||
}
|
||||
|
||||
export type SuggestionClassification =
|
||||
| 'interview_invite'
|
||||
| 'rejection'
|
||||
| 'question'
|
||||
| 'noise'
|
||||
|
||||
export interface EmailSuggestion {
|
||||
id: string
|
||||
application_id: string | null
|
||||
from_address: string
|
||||
subject: string
|
||||
snippet: string
|
||||
classification: SuggestionClassification
|
||||
created_at: string
|
||||
status: 'pending' | 'accepted' | 'dismissed'
|
||||
}
|
||||
|
||||
export interface NotificationLogEntry {
|
||||
id: string
|
||||
channel: string
|
||||
message: string
|
||||
data: Record<string, unknown> | null
|
||||
created_at: string
|
||||
}
|
||||
|
||||
export interface ClusterPosting {
|
||||
id: string
|
||||
title: string
|
||||
company: string
|
||||
source: string
|
||||
url: string
|
||||
score: number
|
||||
}
|
||||
|
||||
export interface Cluster {
|
||||
cluster_id: string
|
||||
postings: ClusterPosting[]
|
||||
}
|
||||
|
||||
export interface TailorKeywordCoverage {
|
||||
ratio: number
|
||||
matched: string[]
|
||||
missing: string[]
|
||||
}
|
||||
|
||||
export interface TailorChangeLogEntry {
|
||||
action?: string
|
||||
section?: string
|
||||
detail?: string
|
||||
change?: string
|
||||
}
|
||||
|
||||
export interface TailorCvResponse {
|
||||
artifact_id: string
|
||||
change_log: TailorChangeLogEntry[]
|
||||
keyword_coverage: TailorKeywordCoverage
|
||||
}
|
||||
|
|
@ -1,135 +0,0 @@
|
|||
import { describe, it, expect, vi, beforeEach } from 'vitest'
|
||||
import { mount, flushPromises } from '@vue/test-utils'
|
||||
import { createPinia, setActivePinia } from 'pinia'
|
||||
import type { Application, Artifact, TailorCvResponse } from '@/types'
|
||||
|
||||
vi.mock('@/api', () => ({
|
||||
getApplications: vi.fn(),
|
||||
getArtifacts: vi.fn(),
|
||||
createCoverLetter: vi.fn(),
|
||||
createApproval: vi.fn(),
|
||||
confirmApproval: vi.fn(),
|
||||
rejectApproval: vi.fn(),
|
||||
outboxSend: vi.fn(),
|
||||
interviewPrep: vi.fn(),
|
||||
tailorCv: vi.fn(),
|
||||
HttpError: class HttpError extends Error {
|
||||
status: number
|
||||
body: unknown
|
||||
constructor(status: number, body: unknown, msg?: string) {
|
||||
super(msg ?? `HTTP ${status}`)
|
||||
this.status = status
|
||||
this.body = body
|
||||
}
|
||||
}
|
||||
}))
|
||||
|
||||
function makeApp(): Application {
|
||||
return {
|
||||
id: 'app-1',
|
||||
job_posting_id: 'j-1',
|
||||
state: 'drafting',
|
||||
score: 90,
|
||||
score_rationale: null,
|
||||
notes: '',
|
||||
state_changed_at: '2026-01-01T00:00:00Z',
|
||||
created_at: '2026-01-01T00:00:00Z',
|
||||
company: 'Acme',
|
||||
title: 'Engineer',
|
||||
location: 'Remote'
|
||||
}
|
||||
}
|
||||
|
||||
function makeArtifact(): Artifact {
|
||||
return {
|
||||
id: 'art-1',
|
||||
application_id: 'app-1',
|
||||
kind: 'cover_letter',
|
||||
filename: 'cover.pdf',
|
||||
content_hash: 'abcdef0123456789',
|
||||
storage_path: '/tmp/cover.pdf',
|
||||
version: 1,
|
||||
origin: 'user_drafted',
|
||||
created_at: '2026-01-01T00:00:00Z'
|
||||
}
|
||||
}
|
||||
|
||||
function makeTailorResult(): TailorCvResponse {
|
||||
return {
|
||||
artifact_id: 'art-tailor-1',
|
||||
change_log: [
|
||||
{ action: 'experience', detail: 'Reordered to highlight Python backend work' },
|
||||
{ action: 'skills', detail: 'Moved Docker and Kubernetes higher' }
|
||||
],
|
||||
keyword_coverage: { ratio: 0.75, matched: ['python', 'docker'], missing: ['kubernetes'] }
|
||||
}
|
||||
}
|
||||
|
||||
async function mountDetail(app: Application, artifacts: Artifact[]) {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
;(api.getApplications as ReturnType<typeof vi.fn>).mockResolvedValue([app])
|
||||
;(api.getArtifacts as ReturnType<typeof vi.fn>).mockResolvedValue(artifacts)
|
||||
const ApplicationDetail = (await import('@/views/ApplicationDetail.vue')).default
|
||||
const wrapper = mount(ApplicationDetail, { props: { id: 'app-1' } })
|
||||
await flushPromises()
|
||||
return { wrapper }
|
||||
}
|
||||
|
||||
describe('Tailor CV panel', () => {
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks()
|
||||
})
|
||||
|
||||
it('renders change log, coverage bar, and download link after tailoring', async () => {
|
||||
const { wrapper } = await mountDetail(makeApp(), [makeArtifact()])
|
||||
const api = await import('@/api')
|
||||
|
||||
// Panel should not be visible before clicking
|
||||
expect(wrapper.find('[data-testid="tailor-panel"]').exists()).toBe(false)
|
||||
|
||||
// Mock tailorCv to return a result
|
||||
;(api.tailorCv as ReturnType<typeof vi.fn>).mockResolvedValue(makeTailorResult())
|
||||
// After tailoring, getArtifacts is called again to refresh
|
||||
;(api.getArtifacts as ReturnType<typeof vi.fn>).mockResolvedValue([makeArtifact(), {
|
||||
id: 'art-tailor-1',
|
||||
application_id: 'app-1',
|
||||
kind: 'cv',
|
||||
filename: 'tailored_cv.pdf',
|
||||
content_hash: 'deadbeef01234567',
|
||||
storage_path: '/tmp/tailored_cv.pdf',
|
||||
version: 1,
|
||||
origin: 'ai_drafted',
|
||||
created_at: '2026-07-30T00:00:00Z'
|
||||
}])
|
||||
|
||||
// Click the Tailor CV button
|
||||
const btn = wrapper.find('[data-testid="tailor-cv-btn"]')
|
||||
expect(btn.exists()).toBe(true)
|
||||
await btn.trigger('click')
|
||||
await flushPromises()
|
||||
|
||||
// Panel should now be visible
|
||||
const panel = wrapper.find('[data-testid="tailor-panel"]')
|
||||
expect(panel.exists()).toBe(true)
|
||||
|
||||
// Change log entries should be visible
|
||||
expect(panel.text()).toContain('Reordered to highlight Python backend work')
|
||||
expect(panel.text()).toContain('Moved Docker and Kubernetes higher')
|
||||
|
||||
// Coverage bar should be present with 75%
|
||||
expect(panel.text()).toContain('Keyword Coverage')
|
||||
expect(panel.text()).toContain('75%')
|
||||
const bar = panel.find('[data-testid="coverage-bar"]')
|
||||
expect(bar.exists()).toBe(true)
|
||||
expect(bar.attributes('style')).toContain('width: 75%')
|
||||
|
||||
// Download link should be present
|
||||
const dl = panel.find('[data-testid="download-link"]')
|
||||
expect(dl.exists()).toBe(true)
|
||||
expect(dl.text()).toContain('Download tailored CV')
|
||||
|
||||
// Tailored artifact should appear in artifacts list
|
||||
expect(wrapper.text()).toContain('tailored_cv.pdf')
|
||||
})
|
||||
})
|
||||
|
|
@ -34,9 +34,11 @@ function makeApp(): Application {
|
|||
notes: '',
|
||||
state_changed_at: '2026-01-01T00:00:00Z',
|
||||
created_at: '2026-01-01T00:00:00Z',
|
||||
company: 'Acme',
|
||||
title: 'Engineer',
|
||||
location: 'Remote'
|
||||
posting: {
|
||||
id: 'j-1', source: 'manual_url', external_id: null, url: 'http://x',
|
||||
company: 'Acme', title: 'Engineer', location: 'Remote', description: '',
|
||||
raw: {}, fetched_at: '2026-01-01T00:00:00Z'
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -10,8 +10,7 @@ import type {
|
|||
Approval,
|
||||
ApprovalAction,
|
||||
CoverLetterResponse,
|
||||
CritiqueComment,
|
||||
TailorCvResponse
|
||||
CritiqueComment
|
||||
} from '@/types'
|
||||
|
||||
const props = defineProps<{ id: string }>()
|
||||
|
|
@ -37,10 +36,6 @@ const sending = ref(false)
|
|||
// Interview prep modal
|
||||
const showPrepModal = ref(false)
|
||||
|
||||
// Tailor CV
|
||||
const tailoring = ref(false)
|
||||
const tailorResult = ref<TailorCvResponse | null>(null)
|
||||
|
||||
const isConfirmed = computed(() => approval.value?.confirmed_by_user ?? false)
|
||||
const canSend = computed(() => isConfirmed.value && !sending.value)
|
||||
|
||||
|
|
@ -52,29 +47,6 @@ const severityClass: Record<string, string> = {
|
|||
low: 'bg-blue-50 border-blue-200'
|
||||
}
|
||||
|
||||
const coverageRatio = computed(() => {
|
||||
if (!tailorResult.value) return 0
|
||||
const kc = tailorResult.value.keyword_coverage
|
||||
return typeof kc === 'object' && kc !== null ? (kc.ratio ?? 0) : Number(kc) || 0
|
||||
})
|
||||
|
||||
const coverageColor = computed(() => {
|
||||
if (!tailorResult.value) return 'bg-gray-300'
|
||||
const c = coverageRatio.value
|
||||
if (c >= 0.7) return 'bg-green-500'
|
||||
if (c >= 0.4) return 'bg-yellow-500'
|
||||
return 'bg-red-500'
|
||||
})
|
||||
|
||||
const coveragePercent = computed(() => {
|
||||
return Math.round(coverageRatio.value * 100)
|
||||
})
|
||||
|
||||
const downloadUrl = computed(() => {
|
||||
if (!tailorResult.value) return ''
|
||||
return `${import.meta.env.VITE_API_BASE ?? 'http://localhost:8000/api'}/artifacts/${tailorResult.value.artifact_id}/download`
|
||||
})
|
||||
|
||||
async function loadData() {
|
||||
try {
|
||||
const apps = await api.getApplications()
|
||||
|
|
@ -145,7 +117,7 @@ async function sendOutbox() {
|
|||
if (!approval.value || !canSend.value) return
|
||||
sending.value = true
|
||||
try {
|
||||
await api.outboxSend(approval.value.id, { to: application.value?.company ?? '' })
|
||||
await api.outboxSend(approval.value.id, { to: application.value?.posting?.company ?? '' })
|
||||
toast.push('Sent successfully', 'success')
|
||||
} catch (err) {
|
||||
let msg = 'Send failed'
|
||||
|
|
@ -167,27 +139,6 @@ function closeInterviewPrep() {
|
|||
showPrepModal.value = false
|
||||
}
|
||||
|
||||
async function tailorCv() {
|
||||
tailoring.value = true
|
||||
tailorResult.value = null
|
||||
try {
|
||||
const res = await api.tailorCv(props.id)
|
||||
tailorResult.value = res
|
||||
// Refresh artifacts to show the new tailored CV variant
|
||||
artifacts.value = await api.getArtifacts(props.id)
|
||||
toast.push('CV tailored for this job', 'success')
|
||||
} catch (err) {
|
||||
let msg = 'Failed to tailor CV'
|
||||
if (err instanceof HttpError) {
|
||||
const body = err.body as { error?: { message?: string } } | null
|
||||
msg = body?.error?.message ?? msg
|
||||
}
|
||||
toast.push(msg, 'error')
|
||||
} finally {
|
||||
tailoring.value = false
|
||||
}
|
||||
}
|
||||
|
||||
onMounted(loadData)
|
||||
</script>
|
||||
|
||||
|
|
@ -200,77 +151,14 @@ onMounted(loadData)
|
|||
<template v-if="!loading && application">
|
||||
<!-- Posting info -->
|
||||
<section class="bg-white rounded-lg border border-gray-200 p-4">
|
||||
<div class="font-semibold text-lg">{{ application.company ?? 'Unknown' }}</div>
|
||||
<div class="text-gray-600">{{ application.title ?? 'No title' }}</div>
|
||||
<div class="font-semibold text-lg">{{ application.posting?.company ?? 'Unknown' }}</div>
|
||||
<div class="text-gray-600">{{ application.posting?.title ?? 'No title' }}</div>
|
||||
<div class="text-sm text-gray-500 mt-1">
|
||||
State: <span class="capitalize font-medium">{{ application.state }}</span>
|
||||
<span v-if="application.score != null" class="ml-3">Score: {{ application.score }}</span>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<!-- Tailor CV -->
|
||||
<section class="bg-white rounded-lg border border-gray-200 p-4 space-y-3">
|
||||
<h2 class="font-semibold">Tailor CV for this Job</h2>
|
||||
<p class="text-sm text-gray-600">
|
||||
Generate a tailored CV variant that reorders and rephrases your existing sections toward this posting's keywords. Your facts are never invented, only rephrased.
|
||||
</p>
|
||||
<button
|
||||
@click="tailorCv"
|
||||
:disabled="tailoring"
|
||||
class="bg-indigo-600 text-white px-4 py-2 rounded text-sm disabled:opacity-50"
|
||||
data-testid="tailor-cv-btn"
|
||||
>
|
||||
{{ tailoring ? 'Tailoring...' : 'Tailor My CV' }}
|
||||
</button>
|
||||
|
||||
<!-- Tailor result panel -->
|
||||
<div v-if="tailorResult" class="space-y-4 border-t border-gray-100 pt-3" data-testid="tailor-panel">
|
||||
<!-- Keyword coverage bar -->
|
||||
<div>
|
||||
<div class="flex items-center justify-between text-sm mb-1">
|
||||
<span class="text-gray-600">Keyword Coverage</span>
|
||||
<span class="font-medium">{{ coveragePercent }}%</span>
|
||||
</div>
|
||||
<div class="w-full bg-gray-200 rounded-full h-3">
|
||||
<div
|
||||
class="h-3 rounded-full transition-all"
|
||||
:class="coverageColor"
|
||||
:style="{ width: coveragePercent + '%' }"
|
||||
data-testid="coverage-bar"
|
||||
></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<!-- Change log -->
|
||||
<div>
|
||||
<h3 class="font-medium text-sm mb-2">Changes Made</h3>
|
||||
<ul class="text-sm space-y-1">
|
||||
<li
|
||||
v-for="(entry, i) in tailorResult.change_log"
|
||||
:key="i"
|
||||
class="border-b border-gray-100 py-1"
|
||||
>
|
||||
<span class="font-medium text-gray-700">{{ entry.action || entry.section || 'change' }}:</span>
|
||||
<span class="text-gray-600 ml-1">{{ entry.detail || entry.change }}</span>
|
||||
</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
<!-- Download link -->
|
||||
<div>
|
||||
<a
|
||||
:href="downloadUrl"
|
||||
target="_blank"
|
||||
rel="noopener"
|
||||
class="text-sm text-indigo-600 hover:underline"
|
||||
data-testid="download-link"
|
||||
>
|
||||
Download tailored CV (PDF)
|
||||
</a>
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<!-- Interview prep -->
|
||||
<section class="bg-white rounded-lg border border-gray-200 p-4 space-y-3">
|
||||
<h2 class="font-semibold">Interview Prep</h2>
|
||||
|
|
@ -296,6 +184,7 @@ onMounted(loadData)
|
|||
</ul>
|
||||
<p v-else class="text-sm text-gray-400">No artifacts yet.</p>
|
||||
</section>
|
||||
|
||||
<!-- Cover letter editor + critique -->
|
||||
<section class="bg-white rounded-lg border border-gray-200 p-4 space-y-3">
|
||||
<h2 class="font-semibold">Cover Letter</h2>
|
||||
|
|
|
|||
|
|
@ -30,9 +30,11 @@ function makeApp(id: string, state: string, company: string): Application {
|
|||
notes: '',
|
||||
state_changed_at: '2026-01-01T00:00:00Z',
|
||||
created_at: '2026-01-01T00:00:00Z',
|
||||
company,
|
||||
title: 'Engineer',
|
||||
location: 'Remote'
|
||||
posting: {
|
||||
id: 'j-' + id, source: 'manual_url', external_id: null, url: 'http://x',
|
||||
company, title: 'Engineer', location: 'Remote', description: '',
|
||||
raw: {}, fetched_at: '2026-01-01T00:00:00Z'
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -31,9 +31,11 @@ function fixture(): Application[] {
|
|||
notes: '',
|
||||
state_changed_at: '2026-01-01T00:00:00Z',
|
||||
created_at: '2026-01-01T00:00:00Z',
|
||||
company: 'Acme',
|
||||
title: 'Engineer',
|
||||
location: 'Remote'
|
||||
posting: {
|
||||
id: 'j-1', source: 'manual_url', external_id: null, url: 'http://x',
|
||||
company: 'Acme', title: 'Engineer', location: 'Remote', description: '',
|
||||
raw: {}, fetched_at: '2026-01-01T00:00:00Z'
|
||||
}
|
||||
},
|
||||
{
|
||||
id: 'app-2',
|
||||
|
|
@ -44,9 +46,11 @@ function fixture(): Application[] {
|
|||
notes: '',
|
||||
state_changed_at: '2026-01-01T00:00:00Z',
|
||||
created_at: '2026-01-01T00:00:00Z',
|
||||
company: 'Globex',
|
||||
title: 'Manager',
|
||||
location: 'Malmo'
|
||||
posting: {
|
||||
id: 'j-2', source: 'linkedin', external_id: null, url: 'http://y',
|
||||
company: 'Globex', title: 'Manager', location: 'Malmo', description: '',
|
||||
raw: {}, fetched_at: '2026-01-01T00:00:00Z'
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
|
|||
|
|
@ -41,10 +41,7 @@ function hasRedFlags(app: Application): boolean {
|
|||
}
|
||||
|
||||
function redFlagsFor(app: Application): string[] {
|
||||
const fromBatch = redFlagsMap.value[app.id]
|
||||
if (fromBatch) return fromBatch
|
||||
const stored = (app.score_rationale as { red_flags?: string[] } | null)?.red_flags
|
||||
return stored ?? []
|
||||
return redFlagsMap.value[app.id] ?? []
|
||||
}
|
||||
|
||||
function hasNudge(app: Application): boolean {
|
||||
|
|
@ -56,11 +53,7 @@ async function loadApplications() {
|
|||
applications.value = await api.getApplications()
|
||||
// Load red flags via batch scoring and nudges via today endpoint
|
||||
const [batchResult, todayResult] = await Promise.allSettled([
|
||||
api.batchScore(
|
||||
applications.value
|
||||
.filter((a) => a.state === 'discovered' || a.state === 'scored')
|
||||
.map((a) => a.id)
|
||||
),
|
||||
api.batchScore(applications.value.map((a) => a.id)),
|
||||
api.getToday()
|
||||
])
|
||||
if (batchResult.status === 'fulfilled') {
|
||||
|
|
@ -171,9 +164,9 @@ onMounted(loadApplications)
|
|||
class="inline-block w-2 h-2 rounded-full bg-orange-500 flex-shrink-0"
|
||||
title="Follow-up nudge pending"
|
||||
></span>
|
||||
<div class="font-medium text-sm truncate">{{ app.company ?? 'Unknown' }}</div>
|
||||
<div class="font-medium text-sm truncate">{{ app.posting?.company ?? 'Unknown' }}</div>
|
||||
</div>
|
||||
<div class="text-xs text-gray-500 truncate">{{ app.title ?? 'No title' }}</div>
|
||||
<div class="text-xs text-gray-500 truncate">{{ app.posting?.title ?? 'No title' }}</div>
|
||||
<div v-if="app.score != null" class="text-xs text-green-700 mt-1">
|
||||
Score: {{ app.score }}
|
||||
</div>
|
||||
|
|
|
|||
|
|
@ -1,103 +0,0 @@
|
|||
import { describe, it, expect, vi } from 'vitest'
|
||||
import { mount, flushPromises } from '@vue/test-utils'
|
||||
import { createPinia, setActivePinia } from 'pinia'
|
||||
|
||||
vi.mock('@/api', () => ({
|
||||
getPostings: vi.fn(),
|
||||
getClusters: vi.fn(),
|
||||
createPosting: vi.fn(),
|
||||
fetchPostings: vi.fn(),
|
||||
scorePosting: vi.fn(),
|
||||
batchScore: vi.fn().mockResolvedValue({ results: [] }),
|
||||
HttpError: class HttpError extends Error {
|
||||
status: number
|
||||
body: unknown
|
||||
constructor(status: number, body: unknown, msg?: string) {
|
||||
super(msg ?? `HTTP ${status}`)
|
||||
this.status = status
|
||||
this.body = body
|
||||
}
|
||||
}
|
||||
}))
|
||||
|
||||
describe('Research cluster alternates', () => {
|
||||
it('renders cluster header with alternate count and expands to show alternates', async () => {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
const Research = (await import('@/views/Research.vue')).default
|
||||
|
||||
// Two postings in the same cluster
|
||||
const postings = [
|
||||
{
|
||||
id: 'p-1', source: 'manual_url', external_id: null, url: 'http://a',
|
||||
company: 'Acme', title: 'Backend Dev', location: 'Malmo', description: '',
|
||||
raw: {}, fetched_at: '2026-07-30T00:00:00Z', cluster_id: 'c-1'
|
||||
},
|
||||
{
|
||||
id: 'p-2', source: 'linkedin', external_id: null, url: 'http://b',
|
||||
company: 'Acme', title: 'Backend Dev', location: 'Remote', description: '',
|
||||
raw: {}, fetched_at: '2026-07-30T00:00:00Z', cluster_id: 'c-1'
|
||||
}
|
||||
]
|
||||
|
||||
// Clusters endpoint returns the cluster with alternate postings
|
||||
const clusters = [
|
||||
{
|
||||
cluster_id: 'c-1',
|
||||
postings: [
|
||||
{ id: 'p-1', title: 'Backend Dev', company: 'Acme', source: 'manual_url', url: 'http://a', score: 85 },
|
||||
{ id: 'p-2', title: 'Backend Dev', company: 'Acme', source: 'linkedin', url: 'http://b', score: 80 }
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
;(api.getPostings as ReturnType<typeof vi.fn>).mockResolvedValue(postings)
|
||||
;(api.getClusters as ReturnType<typeof vi.fn>).mockResolvedValue(clusters)
|
||||
|
||||
const wrapper = mount(Research)
|
||||
await flushPromises()
|
||||
|
||||
// Cluster header should show "Also via 1 more"
|
||||
expect(wrapper.text()).toContain('Also via 1 more')
|
||||
|
||||
// Alternates should NOT be visible before expanding
|
||||
const alternatesBefore = wrapper.find('[data-testid="cluster-alternates"]')
|
||||
expect(alternatesBefore.exists()).toBe(false)
|
||||
|
||||
// Click to expand
|
||||
const toggle = wrapper.find('[data-testid="cluster-alternates-toggle"]')
|
||||
expect(toggle.exists()).toBe(true)
|
||||
await toggle.trigger('click')
|
||||
await flushPromises()
|
||||
|
||||
// Alternates should now be visible
|
||||
const alternatesAfter = wrapper.find('[data-testid="cluster-alternates"]')
|
||||
expect(alternatesAfter.exists()).toBe(true)
|
||||
// Should show the alternate source (linkedin) and link
|
||||
expect(alternatesAfter.text()).toContain('linkedin')
|
||||
})
|
||||
|
||||
it('does not show cluster header for single postings without alternates', async () => {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
const Research = (await import('@/views/Research.vue')).default
|
||||
|
||||
const postings = [
|
||||
{
|
||||
id: 'p-3', source: 'manual_url', external_id: null, url: 'http://c',
|
||||
company: 'Globex', title: 'Manager', location: 'Stockholm', description: '',
|
||||
raw: {}, fetched_at: '2026-07-30T00:00:00Z'
|
||||
}
|
||||
]
|
||||
|
||||
;(api.getPostings as ReturnType<typeof vi.fn>).mockResolvedValue(postings)
|
||||
;(api.getClusters as ReturnType<typeof vi.fn>).mockResolvedValue([])
|
||||
|
||||
const wrapper = mount(Research)
|
||||
await flushPromises()
|
||||
|
||||
// Should not show "Also via" since no alternates
|
||||
expect(wrapper.text()).not.toContain('Also via')
|
||||
expect(wrapper.text()).toContain('Globex')
|
||||
})
|
||||
})
|
||||
|
|
@ -1,20 +1,18 @@
|
|||
<script setup lang="ts">
|
||||
import { onMounted, ref, computed } from 'vue'
|
||||
import { onMounted, ref } from 'vue'
|
||||
import { useToastStore } from '@/stores/toast'
|
||||
import * as api from '@/api'
|
||||
import { HttpError } from '@/api'
|
||||
import type { JobPosting, Cluster } from '@/types'
|
||||
import type { JobPosting } from '@/types'
|
||||
|
||||
const toast = useToastStore()
|
||||
|
||||
const postings = ref<JobPosting[]>([])
|
||||
const clusters = ref<Cluster[]>([])
|
||||
const loading = ref(true)
|
||||
const newUrl = ref('')
|
||||
const scoringId = ref<string | null>(null)
|
||||
const scoreMap = ref<Record<string, { score: number; rationale: Record<string, unknown> }>>({})
|
||||
const redFlagsMap = ref<Record<string, string[]>>({})
|
||||
const expandedClusters = ref<Set<string>>(new Set())
|
||||
|
||||
// Fetch form
|
||||
const fetchQuery = ref('')
|
||||
|
|
@ -22,51 +20,9 @@ const fetchRegion = ref('')
|
|||
const fetching = ref(false)
|
||||
const fetchResult = ref<{ new: number; dupes: number } | null>(null)
|
||||
|
||||
// Group postings by cluster_id from the posting data. Postings without cluster_id get unique singleton groups.
|
||||
const groupedPostings = computed(() => {
|
||||
const map = new Map<string, JobPosting[]>()
|
||||
for (const p of postings.value) {
|
||||
const cid = (p as JobPosting & { cluster_id?: string }).cluster_id ?? `solo-${p.id}`
|
||||
if (!map.has(cid)) map.set(cid, [])
|
||||
map.get(cid)!.push(p)
|
||||
}
|
||||
return Array.from(map.entries()).map(([cluster_id, items]) => ({ cluster_id, items }))
|
||||
})
|
||||
|
||||
// Alternates for a cluster (from GET /clusters endpoint)
|
||||
function clusterAlternatives(clusterId: string): Cluster['postings'] {
|
||||
const cluster = clusters.value.find((c) => c.cluster_id === clusterId)
|
||||
if (!cluster) return []
|
||||
// Return postings other than the first/best one
|
||||
return cluster.postings.slice(1)
|
||||
}
|
||||
|
||||
function isExpanded(clusterId: string): boolean {
|
||||
return expandedClusters.value.has(clusterId)
|
||||
}
|
||||
|
||||
function toggleExpand(clusterId: string) {
|
||||
const next = new Set(expandedClusters.value)
|
||||
if (next.has(clusterId)) {
|
||||
next.delete(clusterId)
|
||||
} else {
|
||||
next.add(clusterId)
|
||||
}
|
||||
expandedClusters.value = next
|
||||
}
|
||||
|
||||
async function loadPostings() {
|
||||
try {
|
||||
const [postingsRes, clustersRes] = await Promise.allSettled([
|
||||
api.getPostings(),
|
||||
api.getClusters()
|
||||
])
|
||||
if (postingsRes.status === 'fulfilled') {
|
||||
postings.value = postingsRes.value
|
||||
}
|
||||
if (clustersRes.status === 'fulfilled') {
|
||||
clusters.value = clustersRes.value
|
||||
}
|
||||
postings.value = await api.getPostings()
|
||||
// Load red flags for existing postings via batch scoring
|
||||
if (postings.value.length > 0) {
|
||||
try {
|
||||
|
|
@ -153,7 +109,7 @@ onMounted(loadPostings)
|
|||
|
||||
<!-- Fetch form (Arbetsformedlingen connector) -->
|
||||
<section class="bg-white rounded-lg border border-gray-200 p-4 space-y-3">
|
||||
<h2 class="font-semibold">Fetch from Arbetsförmedlingen</h2>
|
||||
<h2 class="font-semibold">Fetch from Arbetsformedlingen</h2>
|
||||
<div class="flex gap-2">
|
||||
<input
|
||||
v-model="fetchQuery"
|
||||
|
|
@ -194,101 +150,49 @@ onMounted(loadPostings)
|
|||
|
||||
<div v-if="loading" class="text-gray-500">Loading...</div>
|
||||
|
||||
<!-- Cluster grouped postings -->
|
||||
<div v-if="!loading" class="space-y-4">
|
||||
<div
|
||||
v-for="group in groupedPostings"
|
||||
:key="group.cluster_id"
|
||||
class="bg-white rounded-lg border border-gray-200"
|
||||
>
|
||||
<!-- Cluster header -->
|
||||
<div
|
||||
v-if="clusterAlternatives(group.cluster_id).length > 0"
|
||||
class="flex items-center justify-between px-4 py-2 border-b border-gray-100 cursor-pointer hover:bg-gray-50"
|
||||
@click="toggleExpand(group.cluster_id)"
|
||||
>
|
||||
<span class="text-sm font-medium text-gray-700">
|
||||
{{ group.items[0]?.company ?? 'Unknown' }} - {{ group.items[0]?.title ?? 'No title' }}
|
||||
</span>
|
||||
<span class="text-xs text-gray-500" data-testid="cluster-alternates-toggle">
|
||||
Also via {{ clusterAlternatives(group.cluster_id).length }} more
|
||||
<span v-if="isExpanded(group.cluster_id)">▲</span>
|
||||
<span v-else>▼</span>
|
||||
</span>
|
||||
</div>
|
||||
|
||||
<!-- Main posting table for this cluster -->
|
||||
<table class="w-full text-sm">
|
||||
<thead class="bg-gray-50 text-left">
|
||||
<tr>
|
||||
<th class="px-3 py-2">Company</th>
|
||||
<th class="px-3 py-2">Title</th>
|
||||
<th class="px-3 py-2">Location</th>
|
||||
<th class="px-3 py-2">Source</th>
|
||||
<th class="px-3 py-2">Scam</th>
|
||||
<th class="px-3 py-2">Fetched</th>
|
||||
<th class="px-3 py-2">Actions</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr v-for="p in group.items" :key="p.id" class="border-t border-gray-100">
|
||||
<td class="px-3 py-2">{{ p.company }}</td>
|
||||
<td class="px-3 py-2">{{ p.title }}</td>
|
||||
<td class="px-3 py-2">{{ p.location }}</td>
|
||||
<td class="px-3 py-2">{{ p.source }}</td>
|
||||
<td class="px-3 py-2">
|
||||
<span
|
||||
v-if="hasScamFlag(p)"
|
||||
class="text-red-600 font-bold"
|
||||
:title="scamFlagsFor(p).join('; ')"
|
||||
>
|
||||
⚠
|
||||
</span>
|
||||
<span v-else class="text-gray-400">-</span>
|
||||
</td>
|
||||
<td class="px-3 py-2 text-gray-500">{{ p.fetched_at?.slice(0, 10) }}</td>
|
||||
<td class="px-3 py-2">
|
||||
<button
|
||||
@click="scorePosting(p)"
|
||||
:disabled="scoringId === p.id"
|
||||
class="text-indigo-600 hover:underline text-sm"
|
||||
>
|
||||
{{ scoringId === p.id ? 'Scoring...' : 'Score' }}
|
||||
</button>
|
||||
<span v-if="scoreMap[p.id]" class="ml-2 text-xs bg-green-100 text-green-800 rounded px-2 py-0.5">
|
||||
{{ scoreMap[p.id].score }}
|
||||
</span>
|
||||
</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
|
||||
<!-- Expandable alternates -->
|
||||
<div
|
||||
v-if="isExpanded(group.cluster_id) && clusterAlternatives(group.cluster_id).length > 0"
|
||||
class="border-t border-gray-100 px-4 py-3 bg-gray-50"
|
||||
data-testid="cluster-alternates"
|
||||
>
|
||||
<div class="text-xs font-medium text-gray-500 mb-2">Alternate sources for this role:</div>
|
||||
<ul class="text-sm space-y-1">
|
||||
<li
|
||||
v-for="alt in clusterAlternatives(group.cluster_id)"
|
||||
:key="alt.id"
|
||||
class="flex items-center justify-between"
|
||||
<table v-if="!loading" class="w-full bg-white rounded-lg border border-gray-200 text-sm">
|
||||
<thead class="bg-gray-50 text-left">
|
||||
<tr>
|
||||
<th class="px-3 py-2">Company</th>
|
||||
<th class="px-3 py-2">Title</th>
|
||||
<th class="px-3 py-2">Location</th>
|
||||
<th class="px-3 py-2">Source</th>
|
||||
<th class="px-3 py-2">Scam</th>
|
||||
<th class="px-3 py-2">Fetched</th>
|
||||
<th class="px-3 py-2">Actions</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<tr v-for="p in postings" :key="p.id" class="border-t border-gray-100">
|
||||
<td class="px-3 py-2">{{ p.company }}</td>
|
||||
<td class="px-3 py-2">{{ p.title }}</td>
|
||||
<td class="px-3 py-2">{{ p.location }}</td>
|
||||
<td class="px-3 py-2">{{ p.source }}</td>
|
||||
<td class="px-3 py-2">
|
||||
<span
|
||||
v-if="hasScamFlag(p)"
|
||||
class="text-red-600 font-bold"
|
||||
:title="scamFlagsFor(p).join('; ')"
|
||||
>
|
||||
<span>
|
||||
<a :href="alt.url" target="_blank" rel="noopener" class="text-indigo-600 hover:underline">
|
||||
{{ alt.company }}
|
||||
</a>
|
||||
<span class="text-gray-400 ml-2">({{ alt.source }})</span>
|
||||
</span>
|
||||
<span v-if="alt.score" class="text-xs bg-green-100 text-green-800 rounded px-2 py-0.5">
|
||||
{{ alt.score }}
|
||||
</span>
|
||||
</li>
|
||||
</ul>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
⚠
|
||||
</span>
|
||||
<span v-else class="text-gray-400">-</span>
|
||||
</td>
|
||||
<td class="px-3 py-2 text-gray-500">{{ p.fetched_at?.slice(0, 10) }}</td>
|
||||
<td class="px-3 py-2">
|
||||
<button
|
||||
@click="scorePosting(p)"
|
||||
:disabled="scoringId === p.id"
|
||||
class="text-indigo-600 hover:underline text-sm"
|
||||
>
|
||||
{{ scoringId === p.id ? 'Scoring...' : 'Score' }}
|
||||
</button>
|
||||
<span v-if="scoreMap[p.id]" class="ml-2 text-xs bg-green-100 text-green-800 rounded px-2 py-0.5">
|
||||
{{ scoreMap[p.id].score }}
|
||||
</span>
|
||||
</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</template>
|
||||
|
|
@ -1,79 +0,0 @@
|
|||
import { describe, it, expect, vi } from 'vitest'
|
||||
import { mount, flushPromises } from '@vue/test-utils'
|
||||
import { createPinia, setActivePinia } from 'pinia'
|
||||
|
||||
vi.mock('@/api', () => ({
|
||||
getToday: vi.fn(),
|
||||
getSuggestions: vi.fn().mockResolvedValue([]),
|
||||
getNotificationLog: vi.fn().mockResolvedValue([]),
|
||||
acceptSuggestion: vi.fn(),
|
||||
dismissSuggestion: vi.fn(),
|
||||
getTelemetryTasks: vi.fn().mockResolvedValue([]),
|
||||
HttpError: class HttpError extends Error {
|
||||
status: number
|
||||
body: unknown
|
||||
constructor(status: number, body: unknown, msg?: string) {
|
||||
super(msg ?? `HTTP ${status}`)
|
||||
this.status = status
|
||||
this.body = body
|
||||
}
|
||||
}
|
||||
}))
|
||||
|
||||
describe('TodayView deadlines strip', () => {
|
||||
it('renders deadline cards with urgent styling when <= 2 days', async () => {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
const TodayView = (await import('@/views/TodayView.vue')).default
|
||||
|
||||
// Build deadlines: one urgent (tomorrow) and one normal (5 days)
|
||||
const tomorrow = new Date()
|
||||
tomorrow.setDate(tomorrow.getDate() + 1)
|
||||
const fiveDays = new Date()
|
||||
fiveDays.setDate(fiveDays.getDate() + 5)
|
||||
|
||||
;(api.getToday as ReturnType<typeof vi.fn>).mockResolvedValue({
|
||||
digest: [],
|
||||
nudges: [],
|
||||
pending_approvals: 0,
|
||||
deadlines: [
|
||||
{ application_id: 'app-1', title: 'Backend Dev', company: 'Acme', apply_by: tomorrow.toISOString().slice(0, 10) },
|
||||
{ application_id: 'app-2', title: 'Frontend Dev', company: 'Globex', apply_by: fiveDays.toISOString().slice(0, 10) }
|
||||
]
|
||||
})
|
||||
|
||||
const wrapper = mount(TodayView)
|
||||
await flushPromises()
|
||||
|
||||
// Section heading present
|
||||
expect(wrapper.text()).toContain('Deadlines This Week')
|
||||
|
||||
// Both companies shown
|
||||
expect(wrapper.text()).toContain('Acme')
|
||||
expect(wrapper.text()).toContain('Globex')
|
||||
|
||||
// Urgent card has red background class
|
||||
const urgentCard = wrapper.findAll('.bg-red-50')
|
||||
expect(urgentCard.length).toBeGreaterThanOrEqual(1)
|
||||
expect(urgentCard[0].text()).toContain('Backend Dev')
|
||||
expect(urgentCard[0].text()).toContain('Acme')
|
||||
})
|
||||
|
||||
it('does not render deadlines section when no deadlines', async () => {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
const TodayView = (await import('@/views/TodayView.vue')).default
|
||||
|
||||
;(api.getToday as ReturnType<typeof vi.fn>).mockResolvedValue({
|
||||
digest: [],
|
||||
nudges: [],
|
||||
pending_approvals: 0,
|
||||
deadlines: []
|
||||
})
|
||||
|
||||
const wrapper = mount(TodayView)
|
||||
await flushPromises()
|
||||
|
||||
expect(wrapper.text()).not.toContain('Deadlines This Week')
|
||||
})
|
||||
})
|
||||
|
|
@ -1,125 +0,0 @@
|
|||
import { describe, it, expect, vi } from 'vitest'
|
||||
import { mount, flushPromises } from '@vue/test-utils'
|
||||
import { createPinia, setActivePinia } from 'pinia'
|
||||
|
||||
vi.mock('@/api', () => ({
|
||||
getToday: vi.fn(),
|
||||
getSuggestions: vi.fn(),
|
||||
getNotificationLog: vi.fn().mockResolvedValue([]),
|
||||
acceptSuggestion: vi.fn(),
|
||||
dismissSuggestion: vi.fn(),
|
||||
getTelemetryTasks: vi.fn().mockResolvedValue([]),
|
||||
HttpError: class HttpError extends Error {
|
||||
status: number
|
||||
body: unknown
|
||||
constructor(status: number, body: unknown, msg?: string) {
|
||||
super(msg ?? `HTTP ${status}`)
|
||||
this.status = status
|
||||
this.body = body
|
||||
}
|
||||
}
|
||||
}))
|
||||
|
||||
describe('TodayView suggestions accept flow', () => {
|
||||
it('calls acceptSuggestion API and removes card from pending list', async () => {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
const TodayView = (await import('@/views/TodayView.vue')).default
|
||||
|
||||
const suggestions = [
|
||||
{
|
||||
id: 'sug-1',
|
||||
application_id: 'app-1',
|
||||
from_address: 'recruiter@acme.com',
|
||||
subject: 'Interview Invitation',
|
||||
snippet: 'We would like to invite you...',
|
||||
classification: 'interview_invite',
|
||||
created_at: '2026-07-30T10:00:00Z',
|
||||
status: 'pending'
|
||||
},
|
||||
{
|
||||
id: 'sug-2',
|
||||
application_id: 'app-2',
|
||||
from_address: 'noreply@globex.com',
|
||||
subject: 'Application Update',
|
||||
snippet: 'Thank you for applying...',
|
||||
classification: 'rejection',
|
||||
created_at: '2026-07-30T11:00:00Z',
|
||||
status: 'pending'
|
||||
}
|
||||
]
|
||||
|
||||
;(api.getToday as ReturnType<typeof vi.fn>).mockResolvedValue({
|
||||
digest: [],
|
||||
nudges: [],
|
||||
pending_approvals: 0,
|
||||
deadlines: []
|
||||
})
|
||||
;(api.getSuggestions as ReturnType<typeof vi.fn>).mockResolvedValue(suggestions)
|
||||
// Accept returns the updated suggestion with status 'accepted'
|
||||
;(api.acceptSuggestion as ReturnType<typeof vi.fn>).mockResolvedValue({ ...suggestions[0], status: 'accepted' })
|
||||
|
||||
const wrapper = mount(TodayView)
|
||||
await flushPromises()
|
||||
|
||||
// Both suggestions visible initially
|
||||
expect(wrapper.text()).toContain('Interview Invitation')
|
||||
expect(wrapper.text()).toContain('Application Update')
|
||||
expect(wrapper.text()).toContain('Interview Invite')
|
||||
|
||||
// Click Accept on first suggestion
|
||||
const acceptBtns = wrapper.findAll('[data-testid="accept-suggestion"]')
|
||||
expect(acceptBtns.length).toBe(2)
|
||||
await acceptBtns[0].trigger('click')
|
||||
await flushPromises()
|
||||
|
||||
// API was called with the right id
|
||||
expect(api.acceptSuggestion).toHaveBeenCalledWith('sug-1')
|
||||
|
||||
// The accepted suggestion should no longer appear in the pending list
|
||||
expect(wrapper.text()).not.toContain('Interview Invitation')
|
||||
// The other suggestion should still be present
|
||||
expect(wrapper.text()).toContain('Application Update')
|
||||
})
|
||||
|
||||
it('calls dismissSuggestion API and removes card from pending list', async () => {
|
||||
setActivePinia(createPinia())
|
||||
const api = await import('@/api')
|
||||
const TodayView = (await import('@/views/TodayView.vue')).default
|
||||
|
||||
const suggestions = [
|
||||
{
|
||||
id: 'sug-3',
|
||||
application_id: null,
|
||||
from_address: 'spam@noise.com',
|
||||
subject: 'Some spam',
|
||||
snippet: 'Buy our product...',
|
||||
classification: 'noise',
|
||||
created_at: '2026-07-30T12:00:00Z',
|
||||
status: 'pending'
|
||||
}
|
||||
]
|
||||
|
||||
;(api.getToday as ReturnType<typeof vi.fn>).mockResolvedValue({
|
||||
digest: [],
|
||||
nudges: [],
|
||||
pending_approvals: 0,
|
||||
deadlines: []
|
||||
})
|
||||
;(api.getSuggestions as ReturnType<typeof vi.fn>).mockResolvedValue(suggestions)
|
||||
;(api.dismissSuggestion as ReturnType<typeof vi.fn>).mockResolvedValue({ ...suggestions[0], status: 'dismissed' })
|
||||
|
||||
const wrapper = mount(TodayView)
|
||||
await flushPromises()
|
||||
|
||||
expect(wrapper.text()).toContain('Some spam')
|
||||
|
||||
const dismissBtn = wrapper.find('[data-testid="dismiss-suggestion"]')
|
||||
expect(dismissBtn.exists()).toBe(true)
|
||||
await dismissBtn.trigger('click')
|
||||
await flushPromises()
|
||||
|
||||
expect(api.dismissSuggestion).toHaveBeenCalledWith('sug-3')
|
||||
expect(wrapper.text()).not.toContain('Some spam')
|
||||
})
|
||||
})
|
||||
|
|
@ -4,73 +4,19 @@ import { useRouter } from 'vue-router'
|
|||
import { useToastStore } from '@/stores/toast'
|
||||
import * as api from '@/api'
|
||||
import CostDisplay from '@/components/CostDisplay.vue'
|
||||
import type { TodayResponseV11, TodayDeadline, EmailSuggestion, NotificationLogEntry, SuggestionClassification } from '@/types'
|
||||
import type { TodayResponse } from '@/types'
|
||||
|
||||
const toast = useToastStore()
|
||||
const router = useRouter()
|
||||
|
||||
const today = ref<TodayResponseV11 | null>(null)
|
||||
const today = ref<TodayResponse | null>(null)
|
||||
const loading = ref(true)
|
||||
const deadlines = ref<TodayDeadline[]>([])
|
||||
const suggestions = ref<EmailSuggestion[]>([])
|
||||
const notifications = ref<NotificationLogEntry[]>([])
|
||||
const suggestionActioningId = ref<string | null>(null)
|
||||
|
||||
const nudgeIds = computed(() => new Set(today.value?.nudges.map((n) => n.application_id) ?? []))
|
||||
|
||||
const pendingSuggestions = computed(() =>
|
||||
suggestions.value.filter((s) => s.status === 'pending')
|
||||
)
|
||||
|
||||
const classificationChipClass: Record<SuggestionClassification, string> = {
|
||||
interview_invite: 'bg-green-100 text-green-800',
|
||||
rejection: 'bg-red-100 text-red-800',
|
||||
question: 'bg-yellow-100 text-yellow-800',
|
||||
noise: 'bg-gray-100 text-gray-600'
|
||||
}
|
||||
|
||||
const classificationLabel: Record<SuggestionClassification, string> = {
|
||||
interview_invite: 'Interview Invite',
|
||||
rejection: 'Rejection',
|
||||
question: 'Question',
|
||||
noise: 'Noise'
|
||||
}
|
||||
|
||||
function daysUntil(dateStr: string): number {
|
||||
const today = new Date()
|
||||
today.setHours(0, 0, 0, 0)
|
||||
const target = new Date(dateStr)
|
||||
target.setHours(0, 0, 0, 0)
|
||||
const diff = Math.round((target.getTime() - today.getTime()) / (1000 * 60 * 60 * 24))
|
||||
return diff
|
||||
}
|
||||
|
||||
function isUrgent(dateStr: string): boolean {
|
||||
return daysUntil(dateStr) <= 2
|
||||
}
|
||||
|
||||
function formatDate(dateStr: string): string {
|
||||
const d = new Date(dateStr)
|
||||
return d.toLocaleDateString('en-US', { month: 'short', day: 'numeric' })
|
||||
}
|
||||
|
||||
async function loadToday() {
|
||||
try {
|
||||
const [todayRes, suggestionsRes, notifRes] = await Promise.allSettled([
|
||||
api.getToday(),
|
||||
api.getSuggestions(),
|
||||
api.getNotificationLog()
|
||||
])
|
||||
if (todayRes.status === 'fulfilled') {
|
||||
today.value = todayRes.value
|
||||
deadlines.value = todayRes.value.deadlines ?? []
|
||||
}
|
||||
if (suggestionsRes.status === 'fulfilled') {
|
||||
suggestions.value = suggestionsRes.value
|
||||
}
|
||||
if (notifRes.status === 'fulfilled') {
|
||||
notifications.value = notifRes.value.slice(0, 5)
|
||||
}
|
||||
today.value = await api.getToday()
|
||||
} catch {
|
||||
toast.push('Failed to load today digest', 'error')
|
||||
} finally {
|
||||
|
|
@ -78,36 +24,6 @@ async function loadToday() {
|
|||
}
|
||||
}
|
||||
|
||||
async function acceptSuggestion(id: string) {
|
||||
suggestionActioningId.value = id
|
||||
try {
|
||||
await api.acceptSuggestion(id)
|
||||
suggestions.value = suggestions.value.map((s) =>
|
||||
s.id === id ? { ...s, status: 'accepted' } : s
|
||||
)
|
||||
toast.push('Suggestion accepted', 'success')
|
||||
} catch {
|
||||
toast.push('Failed to accept suggestion', 'error')
|
||||
} finally {
|
||||
suggestionActioningId.value = null
|
||||
}
|
||||
}
|
||||
|
||||
async function dismissSuggestion(id: string) {
|
||||
suggestionActioningId.value = id
|
||||
try {
|
||||
await api.dismissSuggestion(id)
|
||||
suggestions.value = suggestions.value.map((s) =>
|
||||
s.id === id ? { ...s, status: 'dismissed' } : s
|
||||
)
|
||||
toast.push('Suggestion dismissed', 'success')
|
||||
} catch {
|
||||
toast.push('Failed to dismiss suggestion', 'error')
|
||||
} finally {
|
||||
suggestionActioningId.value = null
|
||||
}
|
||||
}
|
||||
|
||||
function goToApplication(id: string) {
|
||||
router.push(`/applications/${id}`)
|
||||
}
|
||||
|
|
@ -140,76 +56,6 @@ onMounted(loadToday)
|
|||
</span>
|
||||
</div>
|
||||
|
||||
<!-- Deadlines this week strip -->
|
||||
<section v-if="deadlines.length > 0">
|
||||
<h2 class="font-semibold text-lg mb-3">Deadlines This Week</h2>
|
||||
<div class="grid gap-3 sm:grid-cols-2 lg:grid-cols-3">
|
||||
<div
|
||||
v-for="d in deadlines"
|
||||
:key="d.application_id"
|
||||
class="rounded-lg border p-4 cursor-pointer hover:shadow-md transition-shadow"
|
||||
:class="isUrgent(d.apply_by) ? 'bg-red-50 border-red-300' : 'bg-white border-gray-200'"
|
||||
@click="goToApplication(d.application_id)"
|
||||
>
|
||||
<div class="font-medium">{{ d.title }}</div>
|
||||
<div class="text-sm text-gray-600">{{ d.company }}</div>
|
||||
<div
|
||||
class="mt-2 text-sm font-medium"
|
||||
:class="isUrgent(d.apply_by) ? 'text-red-700' : 'text-gray-600'"
|
||||
>
|
||||
Apply by {{ formatDate(d.apply_by) }}
|
||||
<span v-if="daysUntil(d.apply_by) === 0" class="ml-1">(today)</span>
|
||||
<span v-else-if="daysUntil(d.apply_by) === 1" class="ml-1">(tomorrow)</span>
|
||||
<span v-else class="ml-1">({{ daysUntil(d.apply_by) }} days)</span>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<!-- Inbox insights strip -->
|
||||
<section v-if="pendingSuggestions.length > 0">
|
||||
<h2 class="font-semibold text-lg mb-3">Inbox Insights</h2>
|
||||
<div class="space-y-3">
|
||||
<div
|
||||
v-for="s in pendingSuggestions"
|
||||
:key="s.id"
|
||||
class="bg-white rounded-lg border border-gray-200 p-4"
|
||||
>
|
||||
<div class="flex items-center justify-between">
|
||||
<div class="flex items-center gap-2">
|
||||
<span class="text-sm font-medium text-gray-700">{{ s.from_address }}</span>
|
||||
<span
|
||||
class="text-xs rounded px-2 py-0.5 font-medium"
|
||||
:class="classificationChipClass[s.classification]"
|
||||
>
|
||||
{{ classificationLabel[s.classification] }}
|
||||
</span>
|
||||
</div>
|
||||
<div class="flex gap-2">
|
||||
<button
|
||||
class="text-sm bg-green-600 text-white px-3 py-1 rounded hover:bg-green-700 disabled:opacity-50"
|
||||
:disabled="suggestionActioningId === s.id"
|
||||
data-testid="accept-suggestion"
|
||||
@click.stop="acceptSuggestion(s.id)"
|
||||
>
|
||||
Accept
|
||||
</button>
|
||||
<button
|
||||
class="text-sm bg-gray-200 text-gray-700 px-3 py-1 rounded hover:bg-gray-300 disabled:opacity-50"
|
||||
:disabled="suggestionActioningId === s.id"
|
||||
data-testid="dismiss-suggestion"
|
||||
@click.stop="dismissSuggestion(s.id)"
|
||||
>
|
||||
Dismiss
|
||||
</button>
|
||||
</div>
|
||||
</div>
|
||||
<div class="font-medium text-sm mt-2">{{ s.subject }}</div>
|
||||
<div class="text-sm text-gray-500 mt-1">{{ s.snippet }}</div>
|
||||
</div>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<!-- Digest cards -->
|
||||
<section>
|
||||
<h2 class="font-semibold text-lg mb-3">Top Matches Today</h2>
|
||||
|
|
@ -269,18 +115,6 @@ onMounted(loadToday)
|
|||
</div>
|
||||
</section>
|
||||
|
||||
<!-- Notification mini-log -->
|
||||
<section v-if="notifications.length > 0">
|
||||
<h2 class="font-semibold text-lg mb-3">Recent Notifications</h2>
|
||||
<ul class="text-sm space-y-1 bg-white rounded-lg border border-gray-200 p-3">
|
||||
<li v-for="n in notifications" :key="n.id" class="border-b border-gray-100 py-1 last:border-0">
|
||||
<span class="text-gray-400 text-xs">{{ n.created_at?.slice(0, 16).replace('T', ' ') }}</span>
|
||||
<span class="ml-2 text-gray-700">{{ n.message }}</span>
|
||||
<span class="ml-2 text-xs text-gray-400">({{ n.channel }})</span>
|
||||
</li>
|
||||
</ul>
|
||||
</section>
|
||||
|
||||
<!-- Cost display -->
|
||||
<CostDisplay />
|
||||
</template>
|
||||
|
|
|
|||
|
|
@ -223,7 +223,7 @@ function prev() {
|
|||
<div v-if="step === 2" class="bg-white rounded-lg border border-gray-200 p-6 space-y-4">
|
||||
<h2 class="font-semibold text-lg">Fetch Job Postings</h2>
|
||||
<p class="text-sm text-gray-600">
|
||||
Search for job postings from the Arbetsförmedlingen connector. New postings will be added to your applications.
|
||||
Search for job postings from the Arbetsformedlingen connector. New postings will be added to your applications.
|
||||
</p>
|
||||
<div class="space-y-2">
|
||||
<label class="block">
|
||||
|
|
|
|||
|
|
@ -1,57 +0,0 @@
|
|||
# Production stack for jobhunt-platform.
|
||||
# Built and started by .forgejo/workflows/deploy.yml on the host docker daemon.
|
||||
# Web UI is published on http://<host>:8085, API on :8000. Postgres is internal only.
|
||||
services:
|
||||
postgres:
|
||||
image: postgres:16
|
||||
container_name: jobhunt-postgres
|
||||
environment:
|
||||
POSTGRES_USER: jobhunt
|
||||
POSTGRES_PASSWORD: jobhunt
|
||||
POSTGRES_DB: jobhunt
|
||||
volumes:
|
||||
- jobhunt_pgdata:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U jobhunt -d jobhunt"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
retries: 10
|
||||
restart: unless-stopped
|
||||
|
||||
api:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: apps/api/Dockerfile.test
|
||||
image: jobhunt-api
|
||||
container_name: jobhunt-api
|
||||
entrypoint: uvicorn app.main:app --host 0.0.0.0 --port 8000
|
||||
env_file: .env
|
||||
environment:
|
||||
DATABASE_URL: postgresql://jobhunt:jobhunt@postgres:5432/jobhunt
|
||||
working_dir: /app/apps/api
|
||||
ports:
|
||||
- "8000:8000"
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
restart: unless-stopped
|
||||
|
||||
web:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: apps/web/Dockerfile
|
||||
args:
|
||||
# Baked into the SPA at build time. Relative /api goes through the
|
||||
# nginx proxy in apps/web/nginx.conf -> http://api:8000/api/
|
||||
VITE_API_BASE: /api
|
||||
image: jobhunt-web
|
||||
container_name: jobhunt-web
|
||||
ports:
|
||||
- "8085:80"
|
||||
depends_on:
|
||||
- api
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
jobhunt_pgdata:
|
||||
name: jobhunt_pgdata
|
||||
|
|
@ -8,8 +8,8 @@ services:
|
|||
POSTGRES_USER: jobhunt
|
||||
POSTGRES_PASSWORD: jobhunt
|
||||
POSTGRES_DB: jobhunt
|
||||
# No host port publishing: CI/tests run inside the compose network, and on
|
||||
# this host 5433 is already taken by bilhej-postgres-prod.
|
||||
ports:
|
||||
- "5433:5432"
|
||||
volumes:
|
||||
- jobhunt_pgdata:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
|
|
@ -29,42 +29,6 @@ services:
|
|||
condition: service_healthy
|
||||
restart: "no"
|
||||
|
||||
api:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: apps/api/Dockerfile.test
|
||||
image: jobhunt-platform-api-test
|
||||
entrypoint: uvicorn app.main:app --host 0.0.0.0 --port 8000
|
||||
environment:
|
||||
DATABASE_URL: postgresql://jobhunt:jobhunt@postgres:5432/jobhunt
|
||||
working_dir: /app/apps/api
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
restart: "no"
|
||||
|
||||
web:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: apps/web/Dockerfile
|
||||
args:
|
||||
VITE_API_BASE: http://api:8000/api
|
||||
depends_on:
|
||||
- api
|
||||
restart: "no"
|
||||
|
||||
shots:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: scripts/Dockerfile.shots
|
||||
volumes:
|
||||
- shots_out:/out
|
||||
depends_on:
|
||||
- api
|
||||
- web
|
||||
restart: "no"
|
||||
|
||||
volumes:
|
||||
jobhunt_pgdata:
|
||||
name: jobhunt_pgdata
|
||||
shots_out:
|
||||
|
|
@ -1,27 +0,0 @@
|
|||
# ADR-0003: v1.1 scope (feel-alive features + killer demo)
|
||||
|
||||
Status: accepted (2026-07-30)
|
||||
|
||||
## Features
|
||||
|
||||
1. **Email reply tracking (read-only)**: IMAP poll (stdlib imaplib, env `IMAP_HOST/PORT/USER/PASS`, flag `EMAIL_WATCH_ENABLED` default false) every 15 min via scheduler. New messages matched to applications by contact domain/company; cheap-LLM classify into `interview_invite | rejection | question | noise`. Result stored as `suggestion` rows; user confirms card moves (state transitions stay user-gated per ADR-0001, no auto-move in v1.1).
|
||||
2. **Notifications**: `NotificationChannel` interface; v1.1 implementations: `LogChannel` (default), `WebhookChannel` (generic POST to user URL, documented Hermes-webhook example). Triggers: daily digest (07:30), interview-invite suggestion. Payload: text + data JSON.
|
||||
3. **Agency duplicate detection**: deterministic similarity (packages/matching with rapidfuzz: normalized employer name match OR token_set_ratio(title)>=85 AND token_set_ratio(description)>=80 -> same cluster). No LLM. `cluster_id` groups postings; API surfaces alternates ("same role via 3 agencies").
|
||||
4. **CV tailoring per posting**: strong-class task `cv_tailor` -> tailored CV variant JSON (reordered skills, rephrased bullets toward posting keywords, unchanged facts — hallucination guard: only reorder/rephrase existing content, never invent). Stored as artifact(kind=cv) variant linked to application. ATS keyword report: deterministic keyword coverage (tokenizer intersection, no LLM).
|
||||
5. **Deadline radar**: cheap extraction task `deadline_extract` during scoring; nullable `apply_by date` on job_posting (migration 004); /today adds `deadlines` strip (next 7 days).
|
||||
|
||||
## Rules kept
|
||||
|
||||
- Approval gate untouched; email watch is read-only.
|
||||
- Zero-config still works: everything above degrades to mock/log/no-op.
|
||||
- No paid-provider fallback for cheap classes.
|
||||
|
||||
## v1.1 delivery shape
|
||||
|
||||
Wave A (parallel, no shared files):
|
||||
- WA1 apps/api: email watch + notifications + suggestion endpoints (owns main.py/schemas.py/migrations this wave)
|
||||
- WA2 packages/matching (new) + packages/llm-gateway (mock additions only)
|
||||
|
||||
Wave B (after merge):
|
||||
- WB1 apps/api: dedupe integration + cv-tailor + deadline endpoints (owns main.py etc.)
|
||||
- WB2 apps/web: v1.1 UI (suggestions inbox strip, cluster alternates, tailor button + variant viewer, deadlines strip, notification settings stub)
|
||||
|
|
@ -117,32 +117,3 @@ This transparency helps you make informed decisions about when to use AI feature
|
|||
- Check the Today page daily for new nudges and digest items.
|
||||
- Always review AI-generated content before sending. The system assists you, but you are the decision maker.
|
||||
- Use the scam/red-flag indicators to avoid suspicious postings.
|
||||
|
||||
## v1.1: email radar, dedupe, tailor CV
|
||||
|
||||
Three new features help you move faster without missing anything.
|
||||
|
||||
### Email radar on the Today page
|
||||
|
||||
The Today page now has two extra strips above the digest:
|
||||
|
||||
- **Deadlines This Week** shows upcoming application deadlines as cards. Cards turn red when the deadline is within two days. Click a card to jump to the application.
|
||||
- **Inbox Insights** lists classified email suggestions (interview invite, rejection, question, noise) pulled from your inbox monitoring. Each card has Accept and Dismiss buttons. Accepted suggestions stay on file; dismissed ones disappear. A **Recent Notifications** mini-log at the bottom shows the last five system events so you can see what happened recently.
|
||||
|
||||
### Dedupe in Research
|
||||
|
||||
The Research table now groups duplicate postings by cluster. When the same role appears through multiple agencies or sources, the cluster header shows "also via N more." Click the header to expand the alternates list and see all sources side by side with their scores. This saves you from applying to the same job three times.
|
||||
|
||||
### Tailor CV
|
||||
|
||||
On any application detail page, click **Tailor My CV** to generate a CV variant tuned to that specific posting. The panel shows:
|
||||
|
||||
- A keyword coverage bar indicating how well your CV matches the posting description.
|
||||
- A change log listing every modification (reordered sections, rephrased bullets). No facts are invented; only rephrased and reordered.
|
||||
- A download link for the tailored CV as a PDF.
|
||||
|
||||
The tailored CV appears in the artifacts list for that application, ready to use in the approval and send flow.
|
||||
|
||||
### Cost breakdown by provider
|
||||
|
||||
The Cost Summary on the Today page now includes a per-model breakdown table showing tokens in, tokens out, cost, and run count for each LLM model used. This helps you compare spending across providers at a glance.
|
||||
|
|
@ -1,55 +0,0 @@
|
|||
# v1.1 worker dispatch cards
|
||||
|
||||
Global v1 rules still binding (see v1-tasks.md header): own paths only, uv, no ORM, no em dashes, mock-first, approval gate untouched, DinD test pattern `docker compose run --rm api-test`, `COPY dir ./dir` not `COPY dir dest`, commit early and often on your own branch, never discard files you did not create (no git clean/reset --hard/checkout --).
|
||||
|
||||
## WA1: apps/api — email watch + notifications + suggestions
|
||||
|
||||
Paths: apps/api/** only (this wave you OWN apps/api; WA2 never touches it).
|
||||
|
||||
Migration `003_email_notify.sql`:
|
||||
```sql
|
||||
CREATE TABLE IF NOT EXISTS email_suggestion (
|
||||
id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
application_id uuid REFERENCES application(id) ON DELETE SET NULL,
|
||||
mailbox_from text NOT NULL,
|
||||
subject text NOT NULL,
|
||||
snippet text NOT NULL,
|
||||
classification text NOT NULL CHECK (classification IN ('interview_invite','rejection','question','noise')),
|
||||
state_proposal text, -- e.g. 'interviewing', null = no move suggested
|
||||
status text NOT NULL DEFAULT 'pending' CHECK (status IN ('pending','accepted','dismissed')),
|
||||
received_at timestamptz NOT NULL,
|
||||
created_at timestamptz DEFAULT now()
|
||||
);
|
||||
CREATE TABLE IF NOT EXISTS notification_log (
|
||||
id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||
channel text NOT NULL,
|
||||
kind text NOT NULL, -- 'daily_digest' | 'email_suggestion'
|
||||
payload jsonb NOT NULL,
|
||||
delivered boolean NOT NULL,
|
||||
error text,
|
||||
created_at timestamptz DEFAULT now()
|
||||
);
|
||||
```
|
||||
|
||||
Deliverables:
|
||||
- `app/imap_watch.py`: stdlib imaplib client (SSL, env IMAP_HOST/PORT/USER/PASS; disabled unless EMAIL_WATCH_ENABLED=true). Fetch UNSEEN since last poll; match sender domain + subject/body keywords to open applications (status in sent/interviewing) via repo query (company name in subject/body, or sender domain in posting URL raw); cheap-class llm task `email_classify` -> {classification, state_proposal?, reason}; insert email_suggestion rows, skip noise↔noise spam dedupe (same from+subject+day -> skip). Tests use a FakeImap (no network).
|
||||
- `app/notify.py`: NotificationChannel protocol; LogChannel (writes notification_log delivered=true); WebhookChannel (env NOTIFY_WEBHOOK_URL, httpx POST {kind,text,data}, 2xx=delivered else error row). `send_notification(kind,text,data)` used by scheduler + email watch.
|
||||
- Scheduler additions: imap poll job every 15 min (only when enabled); daily digest 07:30 -> /today payload text.
|
||||
- Endpoints: `GET /suggestions` (pending), `POST /suggestions/{id}/accept` (applies state_proposal via normal guarded transition path; apply last_activity), `POST /suggestions/{id}/dismiss`, `GET /notifications/log` (last 50).
|
||||
- Mock additions NOT your job (WA2 adds email_classify mock to llm-gateway); your code calls gateway task `email_classify` defensively (mock mode must work; ship a fallback inline mock dict in app/llm.py like existing tasks so tests pass even before WA2 lands).
|
||||
- Tests (+ target 25): imap matching logic, classifier->row, noise dedupe, accept applies transition through guard, webhook success/failure rows, digest payload shape.
|
||||
- Run `docker compose run --rm api-test`, keep suite green (previous 90 + yours).
|
||||
|
||||
## WA2: packages/matching + llm-gateway mocks
|
||||
|
||||
Paths: `packages/matching/**` (new), `packages/llm-gateway/src/llm_gateway/mock.py` + its test file ONLY.
|
||||
|
||||
- packages/matching:
|
||||
- `similarity.py`: normalize (lowercase, strip agency suffixes like AB/Consulting... keep conservative), `title_score(a,b)` rapidfuzz token_set_ratio, `employer_match(a,b)` normalized equality, `desc_score(a,b)` token_set_ratio on first 2000 chars.
|
||||
- `dedupe.py`: `cluster(postings: list[dict]) -> map[cluster_id, list[id]]` with rule: same employer OR (title>=85 AND desc>=80). Deterministic, sorted cluster ids c1..cN by max score desc.
|
||||
- `keywords.py`: `extract_keywords(text, top_n=30)` (freq, drop swedish+english stopwords, keep tech multiwords like "fast api"->fastapi ok simple), `coverage(cv_text, posting_text) -> {matched, missing, ratio}`.
|
||||
- pyproject (uv/hatchling), README, pytest suite (>=20 tests incl. agency repost fixture pairs: invent 3 realistic triples, one of them being legit-different jobs at same agency that must NOT cluster).
|
||||
- llm-gateway mock additions: deterministic outputs for `email_classify` (interview_invite w/ state_proposal interviewing), `cv_tailor` (reordered sections + change_log list), `deadline_extract` ({apply_by: null or ISO date}); register in task->class map (email_classify+deadline_extract = CHEAP, cv_tailor = STRONG); extend tests (+6).
|
||||
- uv venv per package, pytest green, branch feat/WA2-matching, push.
|
||||
|
||||
# (Wave B cards get dispatched after Wave A merges — see ADR-0003)
|
||||
|
|
@ -1,36 +0,0 @@
|
|||
# v1.1 worker dispatch — wave B
|
||||
|
||||
Global v1 rules binding (see v1-tasks.md header). Wave A is merged into master: packages/matching exists (cluster(), keywords coverage()), llm-gateway has mocks for cv_tailor (STRONG) + deadline_extract (CHEAP) + email_classify, api has email_suggestion + notification_log tables and /suggestions + /notifications/log endpoints (125 api tests green).
|
||||
|
||||
## WB1: apps/api — dedupe + tailor + deadline integration
|
||||
|
||||
Paths: apps/api/** ONLY (you own apps/api this wave).
|
||||
|
||||
Migration `004_dedupe_deadline.sql`:
|
||||
```sql
|
||||
ALTER TABLE job_posting ADD COLUMN IF NOT EXISTS cluster_id text;
|
||||
ALTER TABLE job_posting ADD COLUMN IF NOT EXISTS apply_by date;
|
||||
```
|
||||
|
||||
Deliverables:
|
||||
- Cluster assignment: on job_posting creation (manual POST /postings AND /postings/fetch), run packages/matching cluster() over the new posting + all existing postings (small N, fine at v1 scale); persist cluster_id; new clusters only when no match (cluster() output may re-group - reconcile: prefer stability, assign new posting into existing cluster_id when rule matches, else fresh id).
|
||||
- Read: `GET /postings` gains `cluster_id`; new `GET /clusters` -> [{cluster_id, postings: [{id, title, company, source, url, score}]}] sorted by best score desc; UI uses this for "same role via 3 agencies".
|
||||
- Tailor CV: `POST /applications/{id}/tailor-cv` -> gateway task cv_tailor (STRONG) with prompt = profile + sections + posting description; validate output schema {sections, change_log[]}; hallucination guard check: every tailored bullet must map to a source bullet id from input (reject + 502 on unmapped bullet); store artifact kind='cv' origin='ai_drafted' + render PDF via packages/artifacts (bytes -> hash -> storage); return {artifact_id, change_log, keyword_coverage: coverage(cv_text, posting.description)}.
|
||||
- Deadline: scoring endpoints (single + batch) additionally run deadline_extract (CHEAP) and persist apply_by when non-null; /today adds `deadlines: [{application_id, title, company, apply_by}]` for apply_by within next 7 days.
|
||||
- Dockerfile.test: add `-e /app/packages/matching` install.
|
||||
- Tests (+ >= 20): cluster assignment on create, cluster stability across re-imports, clusters endpoint shape, tailor-cv happy path + hallucination rejection (fabricate mock returning bullet without source id -> 502), keyword coverage numbers vs fixture, deadline persisted + /today deadlines filter window.
|
||||
- `docker compose run --rm api-test` all green (125 + yours). Branch feat/WB1-dedupe-tailor, commit incrementally, push.
|
||||
|
||||
## WB2: apps/web — v1.1 UI
|
||||
|
||||
Paths: apps/web/** + docs/user-guide.md (edit allowed, append section) ONLY.
|
||||
|
||||
Backend per docs/api-contract-v2.md + wave A/B adds: /suggestions (accept/dismiss), /notifications/log, /clusters, tailor-cv, /today.deadlines. Mock these in tests like before.
|
||||
|
||||
- Today view: new "Deadlines this week" strip (cards with company/title/date, red when <=2 days) from GET /today.deadlines; "Inbox insights" strip listing pending email_suggestion rows (from/subject/snippet/classification chip) with Accept/Dismiss buttons -> POST endpoints, then refresh; notifications mini-log (last 5) optional.
|
||||
- Research/Postings: group rows by cluster; cluster rows show "also via N more" expandable alternates list (GET /clusters).
|
||||
- Application detail: "Tailor CV for this job" button -> POST tailor-cv -> panel showing change_log bullets + keyword coverage bar + link to download artifact; variant appears in artifacts list.
|
||||
- CostDisplay: add totals by provider (group /telemetry/tasks client-side).
|
||||
- Vitest: +4 tests (deadlines strip render, suggestions accept flow, cluster alternates render, tailor panel render from fixture). Keep all existing green. npm run build + npm test green.
|
||||
- docs/user-guide.md: append "v1.1: email radar, dedupe, tailor CV" short section (plain language, no em dashes).
|
||||
- Branch feat/WB2-web-v11, commit incrementally, push.
|
||||
|
|
@ -30,12 +30,9 @@ TASK_CLASS_MAP: dict[str, TaskClass] = {
|
|||
"score": TaskClass.CHEAP,
|
||||
"extract": TaskClass.CHEAP,
|
||||
"cv_assist": TaskClass.CHEAP,
|
||||
"email_classify": TaskClass.CHEAP,
|
||||
"deadline_extract": TaskClass.CHEAP,
|
||||
"cl_critique": TaskClass.STRONG,
|
||||
"critique": TaskClass.STRONG,
|
||||
"research": TaskClass.STRONG,
|
||||
"cv_tailor": TaskClass.STRONG,
|
||||
}
|
||||
|
||||
# Default budgets (max output tokens) per task name.
|
||||
|
|
@ -43,12 +40,9 @@ DEFAULT_BUDGETS: dict[str, int] = {
|
|||
"score": 2000,
|
||||
"extract": 4000,
|
||||
"cv_assist": 2000,
|
||||
"email_classify": 1000,
|
||||
"deadline_extract": 500,
|
||||
"cl_critique": 4000,
|
||||
"critique": 6000,
|
||||
"research": 4000,
|
||||
"cv_tailor": 6000,
|
||||
}
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -59,41 +59,6 @@ MOCK_OUTPUTS: dict[str, dict] = {
|
|||
"summary": "The company is a mid-size tech firm focused on cloud infrastructure.",
|
||||
"key_points": ["Founded in 2015", "Series B funding", "Remote-first culture"],
|
||||
},
|
||||
"email_classify": {
|
||||
"classification": "interview_invite",
|
||||
"state_proposal": "interviewing",
|
||||
"reason": "The email contains an invitation to schedule an interview.",
|
||||
},
|
||||
"cv_tailor": {
|
||||
"tailored_cv": {
|
||||
"summary": "Senior Python Developer with 6+ years building scalable backend systems.",
|
||||
"skills": [
|
||||
"Python",
|
||||
"Fast API",
|
||||
"PostgreSQL",
|
||||
"Docker",
|
||||
"Kubernetes",
|
||||
"AWS",
|
||||
],
|
||||
"experience": [
|
||||
{
|
||||
"company": "TechCorp",
|
||||
"role": "Senior Backend Engineer",
|
||||
"bullets": [
|
||||
"Led migration of monolith to microservices using Fast API",
|
||||
"Reduced API latency by 40% through query optimization and caching",
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
"change_log": [
|
||||
{"action": "reordered", "detail": "Moved Python and Fast API to top of skills"},
|
||||
{"action": "rephrased", "detail": "Rewrote first experience bullet to emphasize Fast API"},
|
||||
],
|
||||
},
|
||||
"deadline_extract": {
|
||||
"apply_by": None,
|
||||
},
|
||||
}
|
||||
|
||||
# Default mock output for unknown task names.
|
||||
|
|
|
|||
|
|
@ -270,11 +270,8 @@ class TestGatewayConfig:
|
|||
assert config.get_task_class("score") == TaskClass.CHEAP
|
||||
assert config.get_task_class("extract") == TaskClass.CHEAP
|
||||
assert config.get_task_class("cv_assist") == TaskClass.CHEAP
|
||||
assert config.get_task_class("email_classify") == TaskClass.CHEAP
|
||||
assert config.get_task_class("deadline_extract") == TaskClass.CHEAP
|
||||
assert config.get_task_class("critique") == TaskClass.STRONG
|
||||
assert config.get_task_class("cl_critique") == TaskClass.STRONG
|
||||
assert config.get_task_class("cv_tailor") == TaskClass.STRONG
|
||||
|
||||
def test_get_model_routing(self) -> None:
|
||||
config = mock_config(cheap_model="cheap-model", strong_model="strong-model")
|
||||
|
|
|
|||
|
|
@ -1,125 +0,0 @@
|
|||
"""Tests for new v1.1 mock tasks: email_classify, cv_tailor, deadline_extract."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from llm_gateway.config import GatewayConfig, ProviderConfig, TaskClass
|
||||
from llm_gateway.gateway import Gateway
|
||||
from llm_gateway.mock import get_mock_output, MOCK_OUTPUTS
|
||||
|
||||
|
||||
def mock_config(**overrides) -> GatewayConfig:
|
||||
"""Build a config in mock mode (no API key)."""
|
||||
primary = ProviderConfig(
|
||||
name="primary",
|
||||
base_url="https://mock.example.com/v1",
|
||||
api_key="",
|
||||
model="glm-5.2",
|
||||
)
|
||||
defaults = {
|
||||
"primary": primary,
|
||||
"fallback": None,
|
||||
"cheap_model": "glm-5.2",
|
||||
"strong_model": "glm-5.2",
|
||||
"budgets": {
|
||||
"score": 2000,
|
||||
"extract": 4000,
|
||||
"email_classify": 1000,
|
||||
"deadline_extract": 500,
|
||||
"cv_tailor": 6000,
|
||||
"default": 4000,
|
||||
},
|
||||
"max_retries": 2,
|
||||
}
|
||||
defaults.update(overrides)
|
||||
return GatewayConfig(**defaults)
|
||||
|
||||
|
||||
class TestEmailClassifyMock:
|
||||
async def test_email_classify_returns_deterministic(self) -> None:
|
||||
"""email_classify mock returns interview_invite classification."""
|
||||
config = mock_config()
|
||||
gw = Gateway(config)
|
||||
result_a = await gw.run_task("email_classify", "Email from recruiter")
|
||||
result_b = await gw.run_task("email_classify", "Email from recruiter")
|
||||
assert result_a == result_b
|
||||
assert result_a["classification"] == "interview_invite"
|
||||
assert result_a["state_proposal"] == "interviewing"
|
||||
assert "reason" in result_a
|
||||
await gw.aclose()
|
||||
|
||||
async def test_email_classify_is_cheap(self) -> None:
|
||||
"""email_classify should be classified as CHEAP."""
|
||||
config = mock_config()
|
||||
assert config.get_task_class("email_classify") == TaskClass.CHEAP
|
||||
assert config.get_model("email_classify") == config.cheap_model
|
||||
|
||||
def test_email_classify_in_mock_outputs(self) -> None:
|
||||
"""email_classify should be in MOCK_OUTPUTS."""
|
||||
assert "email_classify" in MOCK_OUTPUTS
|
||||
output = get_mock_output("email_classify")
|
||||
assert output["classification"] == "interview_invite"
|
||||
assert output["state_proposal"] == "interviewing"
|
||||
|
||||
|
||||
class TestCvTailorMock:
|
||||
async def test_cv_tailor_returns_deterministic(self) -> None:
|
||||
"""cv_tailor mock returns tailored CV with change_log."""
|
||||
config = mock_config()
|
||||
gw = Gateway(config)
|
||||
result_a = await gw.run_task("cv_tailor", "Tailor CV for posting")
|
||||
result_b = await gw.run_task("cv_tailor", "Tailor CV for posting")
|
||||
assert result_a == result_b
|
||||
assert "tailored_cv" in result_a
|
||||
assert "change_log" in result_a
|
||||
assert isinstance(result_a["change_log"], list)
|
||||
assert len(result_a["change_log"]) >= 1
|
||||
# Check change_log entries have action and detail.
|
||||
for entry in result_a["change_log"]:
|
||||
assert "action" in entry
|
||||
assert "detail" in entry
|
||||
await gw.aclose()
|
||||
|
||||
async def test_cv_tailor_is_strong(self) -> None:
|
||||
"""cv_tailor should be classified as STRONG."""
|
||||
config = mock_config()
|
||||
assert config.get_task_class("cv_tailor") == TaskClass.STRONG
|
||||
assert config.get_model("cv_tailor") == config.strong_model
|
||||
|
||||
def test_cv_tailor_in_mock_outputs(self) -> None:
|
||||
"""cv_tailor should be in MOCK_OUTPUTS."""
|
||||
assert "cv_tailor" in MOCK_OUTPUTS
|
||||
output = get_mock_output("cv_tailor")
|
||||
assert "tailored_cv" in output
|
||||
assert "change_log" in output
|
||||
# Check it has skills and experience.
|
||||
assert "skills" in output["tailored_cv"]
|
||||
assert "experience" in output["tailored_cv"]
|
||||
|
||||
|
||||
class TestDeadlineExtractMock:
|
||||
async def test_deadline_extract_returns_deterministic(self) -> None:
|
||||
"""deadline_extract mock returns apply_by (null by default)."""
|
||||
config = mock_config()
|
||||
gw = Gateway(config)
|
||||
result_a = await gw.run_task("deadline_extract", "Extract deadline")
|
||||
result_b = await gw.run_task("deadline_extract", "Extract deadline")
|
||||
assert result_a == result_b
|
||||
assert "apply_by" in result_a
|
||||
# Default mock has null deadline.
|
||||
assert result_a["apply_by"] is None
|
||||
await gw.aclose()
|
||||
|
||||
async def test_deadline_extract_is_cheap(self) -> None:
|
||||
"""deadline_extract should be classified as CHEAP."""
|
||||
config = mock_config()
|
||||
assert config.get_task_class("deadline_extract") == TaskClass.CHEAP
|
||||
assert config.get_model("deadline_extract") == config.cheap_model
|
||||
|
||||
def test_deadline_extract_in_mock_outputs(self) -> None:
|
||||
"""deadline_extract should be in MOCK_OUTPUTS."""
|
||||
assert "deadline_extract" in MOCK_OUTPUTS
|
||||
output = get_mock_output("deadline_extract")
|
||||
assert "apply_by" in output
|
||||
assert output["apply_by"] is None
|
||||
|
|
@ -1,40 +0,0 @@
|
|||
# packages/matching
|
||||
|
||||
Job posting similarity, dedupe clustering, and keyword coverage for the
|
||||
jobhunt-platform v1.1 agency duplicate detection feature (ADR-0003).
|
||||
|
||||
## Modules
|
||||
|
||||
### similarity.py
|
||||
- `normalize_employer(name)` -- lowercase, strip agency/legal suffixes (AB, Consulting, etc.), remove punctuation.
|
||||
- `title_score(a, b)` -- rapidfuzz token_set_ratio on job titles (0-100).
|
||||
- `employer_match(a, b)` -- True if normalized employer names are equal.
|
||||
- `desc_score(a, b, max_chars=2000)` -- token_set_ratio on first 2000 chars of descriptions.
|
||||
|
||||
### dedupe.py
|
||||
- `cluster(postings: list[dict]) -> dict[str, list[str]]` -- group postings into duplicate clusters.
|
||||
|
||||
Clustering rule (per ADR-0003):
|
||||
- Same employer (normalized) **OR**
|
||||
- Title similarity >= 85 **AND** description similarity >= 80
|
||||
|
||||
Uses union-find for transitive grouping. Clusters are sorted by descending
|
||||
max pairwise score (`c1` = tightest cluster).
|
||||
|
||||
### keywords.py
|
||||
- `extract_keywords(text, top_n=30)` -- frequency-based keyword extraction with Swedish + English stopword removal.
|
||||
- `coverage(cv_text, posting_text)` -- computes keyword coverage of a CV against a job posting.
|
||||
|
||||
Multiword tech terms like "fast api" are collapsed to "fastapi" so they survive as single keywords.
|
||||
|
||||
## Installation (uv)
|
||||
|
||||
```bash
|
||||
cd packages/matching
|
||||
uv venv && source .venv/bin/activate
|
||||
uv pip install -e ".[dev]"
|
||||
pytest
|
||||
```
|
||||
|
||||
## License
|
||||
MIT
|
||||
|
|
@ -1,26 +0,0 @@
|
|||
[project]
|
||||
name = "matching"
|
||||
version = "0.1.0"
|
||||
description = "Job posting similarity, dedupe clustering, and keyword coverage for agency duplicate detection."
|
||||
requires-python = ">=3.13"
|
||||
dependencies = [
|
||||
"rapidfuzz>=3.6",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
dev = [
|
||||
"pytest>=8.0",
|
||||
"pytest-asyncio>=0.24",
|
||||
]
|
||||
|
||||
[build-system]
|
||||
requires = ["hatchling"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
[tool.hatch.build.targets.wheel]
|
||||
packages = ["src/matching"]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
pythonpath = ["src"]
|
||||
asyncio_mode = "auto"
|
||||
|
|
@ -1,28 +0,0 @@
|
|||
"""Job posting matching package.
|
||||
|
||||
Provides:
|
||||
- similarity: normalized text comparison (employer match, title/desc scores)
|
||||
- dedupe: cluster postings into duplicate groups
|
||||
- keywords: keyword extraction and CV-vs-posting coverage
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from matching.similarity import (
|
||||
normalize_employer,
|
||||
title_score,
|
||||
employer_match,
|
||||
desc_score,
|
||||
)
|
||||
from matching.dedupe import cluster
|
||||
from matching.keywords import extract_keywords, coverage
|
||||
|
||||
__all__ = [
|
||||
"normalize_employer",
|
||||
"title_score",
|
||||
"employer_match",
|
||||
"desc_score",
|
||||
"cluster",
|
||||
"extract_keywords",
|
||||
"coverage",
|
||||
]
|
||||
|
|
@ -1,136 +0,0 @@
|
|||
"""Duplicate clustering for job postings.
|
||||
|
||||
Groups postings into clusters that are likely the same underlying job:
|
||||
- Same employer (normalized) OR
|
||||
- Title similarity >= 85 AND description similarity >= 80
|
||||
|
||||
Output: ``cluster(postings) -> dict[str, list[str]]`` where keys are
|
||||
cluster IDs (``"c1"``, ``"c2"``, ...) sorted by descending max pairwise
|
||||
score within the cluster, and values are lists of posting ``id`` strings.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from matching.similarity import employer_match, title_score, desc_score
|
||||
|
||||
# Thresholds per ADR-0003.
|
||||
TITLE_THRESHOLD = 85.0
|
||||
DESC_THRESHOLD = 80.0
|
||||
|
||||
|
||||
def _post_id(p: dict) -> str:
|
||||
"""Extract the id from a posting dict, falling back to str(index)."""
|
||||
pid = p.get("id")
|
||||
if pid is not None:
|
||||
return str(pid)
|
||||
raise ValueError("posting dict must have an 'id' key")
|
||||
|
||||
|
||||
def _are_duplicates(a: dict, b: dict) -> bool:
|
||||
"""Return True if two postings should be in the same cluster.
|
||||
|
||||
Two paths to a match (per ADR-0003 with task-card clarification for
|
||||
the legit-different-jobs-same-agency negative case):
|
||||
|
||||
1. Same employer (normalized) AND some content overlap
|
||||
(title >= 85 OR desc >= 80).
|
||||
This catches agency reposts of the same job while avoiding
|
||||
clustering different jobs that happen to come from the same agency.
|
||||
|
||||
2. Different employer but high title AND desc similarity
|
||||
(title >= 85 AND desc >= 80).
|
||||
This catches cross-agency reposts of the same job.
|
||||
"""
|
||||
ts = title_score(a.get("title", ""), b.get("title", ""))
|
||||
ds = desc_score(a.get("description", ""), b.get("description", ""))
|
||||
same_employer = employer_match(a.get("employer", ""), b.get("employer", ""))
|
||||
|
||||
if same_employer:
|
||||
# Same employer + at least one content dimension similar.
|
||||
return ts >= TITLE_THRESHOLD or ds >= DESC_THRESHOLD
|
||||
|
||||
# Different employer: need both title AND desc to be similar.
|
||||
return ts >= TITLE_THRESHOLD and ds >= DESC_THRESHOLD
|
||||
|
||||
|
||||
def _pair_score(a: dict, b: dict) -> float:
|
||||
"""Compute a similarity score between two postings for sorting clusters."""
|
||||
ts = title_score(a.get("title", ""), b.get("title", ""))
|
||||
ds = desc_score(a.get("description", ""), b.get("description", ""))
|
||||
same_employer = employer_match(a.get("employer", ""), b.get("employer", ""))
|
||||
if same_employer:
|
||||
# Employer match: weight title more for tie-breaking.
|
||||
return 100.0 + ts
|
||||
return (ts + ds) / 2.0
|
||||
|
||||
|
||||
def cluster(postings: list[dict]) -> dict[str, list[str]]:
|
||||
"""Cluster job postings into duplicate groups.
|
||||
|
||||
Uses union-find so transitive duplicates (A~B, B~C => A~C) are grouped
|
||||
together.
|
||||
|
||||
Args:
|
||||
postings: list of dicts with keys ``id``, ``employer``, ``title``,
|
||||
``description``.
|
||||
|
||||
Returns:
|
||||
Dict mapping cluster_id (``"c1"``, ``"c2"``, ...) to a list of
|
||||
posting id strings. Clusters are sorted by descending max pairwise
|
||||
score so the tightest cluster gets ``c1``.
|
||||
"""
|
||||
n = len(postings)
|
||||
if n == 0:
|
||||
return {}
|
||||
|
||||
ids = [_post_id(p) for p in postings]
|
||||
|
||||
# Union-find (disjoint set).
|
||||
parent: list[int] = list(range(n))
|
||||
|
||||
def find(x: int) -> int:
|
||||
while parent[x] != x:
|
||||
parent[x] = parent[parent[x]]
|
||||
x = parent[x]
|
||||
return x
|
||||
|
||||
def union(a: int, b: int) -> None:
|
||||
ra, rb = find(a), find(b)
|
||||
if ra != rb:
|
||||
parent[ra] = rb
|
||||
|
||||
# O(n^2) pairwise comparison.
|
||||
for i in range(n):
|
||||
for j in range(i + 1, n):
|
||||
if _are_duplicates(postings[i], postings[j]):
|
||||
union(i, j)
|
||||
|
||||
# Collect groups.
|
||||
groups: dict[int, list[int]] = {}
|
||||
for i in range(n):
|
||||
root = find(i)
|
||||
groups.setdefault(root, []).append(i)
|
||||
|
||||
# Compute max pairwise score per group for sorting.
|
||||
group_scores: list[tuple[float, list[int]]] = []
|
||||
for indices in groups.values():
|
||||
if len(indices) < 2:
|
||||
max_sc = 0.0
|
||||
else:
|
||||
max_sc = 0.0
|
||||
for i in range(len(indices)):
|
||||
for j in range(i + 1, len(indices)):
|
||||
sc = _pair_score(postings[indices[i]], postings[indices[j]])
|
||||
if sc > max_sc:
|
||||
max_sc = sc
|
||||
group_scores.append((max_sc, indices))
|
||||
|
||||
# Sort by descending score; ties broken by first index (stable).
|
||||
group_scores.sort(key=lambda t: (-t[0], t[1][0]))
|
||||
|
||||
# Assign cluster IDs.
|
||||
result: dict[str, list[str]] = {}
|
||||
for idx, (_, indices) in enumerate(group_scores, start=1):
|
||||
result[f"c{idx}"] = [ids[i] for i in indices]
|
||||
|
||||
return result
|
||||
|
|
@ -1,162 +0,0 @@
|
|||
"""Keyword extraction and coverage analysis.
|
||||
|
||||
Deterministic, no LLM. Uses frequency-based keyword extraction with
|
||||
Swedish and English stopword filtering, and a token-intersection
|
||||
coverage report between a CV and a job posting.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stopwords (Swedish + English). Conservative lists.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SWEDISH_STOPWORDS: frozenset[str] = frozenset({
|
||||
"och", "eller", "som", "att", "den", "det", "de", "vi", "ni", "du",
|
||||
"han", "hon", "den", "en", "ett", "ar", "har", "var", "var", "inte",
|
||||
"med", "for", "fran", "till", "pa", "av", "i", "och", "men", "sa",
|
||||
"när", "då", "hur", "alla", "nagon", "nagot", "alla", "manga", "mycket",
|
||||
"skall", "ska", "kan", "kommer", "blir", "vore", "vill", "borde",
|
||||
"efter", "under", "over", "bAKom", "inom", "mellan", "genom", "utan",
|
||||
"mot", "ut", "fran", "frams", "igår", "idag", "imorgon",
|
||||
})
|
||||
|
||||
_ENGLISH_STOPWORDS: frozenset[str] = frozenset({
|
||||
"a", "an", "the", "and", "or", "but", "in", "on", "at", "to", "for",
|
||||
"of", "with", "by", "from", "is", "are", "was", "were", "be", "been",
|
||||
"being", "have", "has", "had", "do", "does", "did", "will", "would",
|
||||
"should", "could", "may", "might", "must", "can", "this", "that",
|
||||
"these", "those", "i", "you", "he", "she", "it", "we", "they", "me",
|
||||
"him", "her", "us", "them", "my", "your", "his", "its", "our", "their",
|
||||
"what", "which", "who", "whom", "where", "when", "why", "how", "all",
|
||||
"each", "every", "both", "few", "more", "most", "other", "some", "such",
|
||||
"no", "nor", "not", "only", "own", "same", "so", "than", "too", "very",
|
||||
"as", "if", "about", "against", "between", "into", "through", "during",
|
||||
"before", "after", "above", "below", "up", "down", "out", "off",
|
||||
"over", "under", "again", "further", "then", "once", "here", "there",
|
||||
"also", "etc", "e.g", "i.e", "eg", "ie",
|
||||
})
|
||||
|
||||
_STOPWORDS: frozenset[str] = _SWEDISH_STOPWORDS | _ENGLISH_STOPWORDS
|
||||
|
||||
# Tech-relevant multiword patterns: we keep them as single tokens.
|
||||
# e.g. "fast api" -> "fastapi" so it survives as a keyword.
|
||||
_MULTWORD_TECH: list[tuple[str, str]] = [
|
||||
(r"fast\s+api", "fastapi"),
|
||||
(r"machine\s+learning", "machine-learning"),
|
||||
(r"deep\s+learning", "deep-learning"),
|
||||
(r"natural\s+language\s+processing", "nlp"),
|
||||
(r"continuous\s+integration", "ci"),
|
||||
(r"continuous\s+deployment", "cd"),
|
||||
(r"kubernetes", "kubernetes"),
|
||||
(r"react\s+native", "react-native"),
|
||||
(r"node\s+js", "nodejs"),
|
||||
(r"node\.js", "nodejs"),
|
||||
(r"aws", "aws"),
|
||||
(r"gcp", "gcp"),
|
||||
(r"ci/cd", "ci-cd"),
|
||||
]
|
||||
|
||||
# Minimum token length for keywords.
|
||||
_MIN_TOKEN_LEN = 2
|
||||
|
||||
# Tokenizer: split on non-alphanumeric.
|
||||
_TOKEN_RE = re.compile(r"[a-z0-9+#.\-/]+")
|
||||
|
||||
|
||||
def _preprocess_multiwords(text: str) -> str:
|
||||
"""Replace known multiword tech terms with single tokens."""
|
||||
result = text.lower()
|
||||
for pattern, replacement in _MULTWORD_TECH:
|
||||
result = re.sub(pattern, replacement, result, flags=re.IGNORECASE)
|
||||
return result
|
||||
|
||||
|
||||
def _tokenize(text: str) -> list[str]:
|
||||
"""Tokenize text into lowercased tokens."""
|
||||
text = _preprocess_multiwords(text)
|
||||
raw_tokens = _TOKEN_RE.findall(text.lower())
|
||||
tokens: list[str] = []
|
||||
for t in raw_tokens:
|
||||
t = t.strip(".-/")
|
||||
if not t:
|
||||
continue
|
||||
if t in _STOPWORDS:
|
||||
continue
|
||||
if len(t) < _MIN_TOKEN_LEN:
|
||||
continue
|
||||
# Skip pure numbers (unless they look like versions).
|
||||
if t.isdigit() and len(t) > 4:
|
||||
continue
|
||||
tokens.append(t)
|
||||
return tokens
|
||||
|
||||
|
||||
def extract_keywords(text: str, top_n: int = 30) -> list[str]:
|
||||
"""Extract the top *top_n* keywords from *text* by frequency.
|
||||
|
||||
Keywords are lowercased tokens. Stopwords (Swedish + English) are
|
||||
removed. Multiword tech terms like "fast api" are collapsed to
|
||||
"fastapi".
|
||||
|
||||
Args:
|
||||
text: the text to analyze.
|
||||
top_n: maximum number of keywords to return.
|
||||
|
||||
Returns:
|
||||
List of keyword strings, most frequent first. Ties are broken
|
||||
alphabetically for determinism.
|
||||
"""
|
||||
if not text:
|
||||
return []
|
||||
tokens = _tokenize(text)
|
||||
if not tokens:
|
||||
return []
|
||||
counts: Counter[str] = Counter(tokens)
|
||||
# Sort by count desc, then alphabetically for deterministic order.
|
||||
ranked = sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
|
||||
return [word for word, _ in ranked[:top_n]]
|
||||
|
||||
|
||||
def coverage(cv_text: str, posting_text: str) -> dict[str, list[str] | float]:
|
||||
"""Compute keyword coverage of a CV against a job posting.
|
||||
|
||||
Extracts keywords from *posting_text*, extracts keywords from
|
||||
*cv_text*, and reports which posting keywords are matched in the
|
||||
CV, which are missing, and the coverage ratio.
|
||||
|
||||
Args:
|
||||
cv_text: the candidate's CV text.
|
||||
posting_text: the job posting text.
|
||||
|
||||
Returns:
|
||||
Dict with keys:
|
||||
- ``matched``: list of posting keywords found in CV.
|
||||
- ``missing``: list of posting keywords NOT found in CV.
|
||||
- ``ratio``: float (matched / total), 0.0 if no keywords.
|
||||
"""
|
||||
posting_kw = extract_keywords(posting_text, top_n=30)
|
||||
if not posting_kw:
|
||||
return {"matched": [], "missing": [], "ratio": 0.0}
|
||||
|
||||
cv_kw_set: set[str] = set(extract_keywords(cv_text, top_n=200))
|
||||
|
||||
matched: list[str] = []
|
||||
missing: list[str] = []
|
||||
for kw in posting_kw:
|
||||
if kw in cv_kw_set:
|
||||
matched.append(kw)
|
||||
else:
|
||||
missing.append(kw)
|
||||
|
||||
total = len(posting_kw)
|
||||
ratio = len(matched) / total if total > 0 else 0.0
|
||||
|
||||
return {
|
||||
"matched": matched,
|
||||
"missing": missing,
|
||||
"ratio": ratio,
|
||||
}
|
||||
|
|
@ -1,113 +0,0 @@
|
|||
"""Similarity helpers for job postings.
|
||||
|
||||
Uses rapidfuzz for fuzzy string matching. All functions are deterministic.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Agency-suffix / noise words to strip from employer names.
|
||||
# Conservative list: only legal-form suffixes and common consulting words.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_AGENCY_SUFFIXES: list[str] = [
|
||||
# Swedish legal forms
|
||||
"ab",
|
||||
"aktiebolag",
|
||||
"hb",
|
||||
"kb",
|
||||
"ekonomisk forening",
|
||||
# English legal forms
|
||||
"inc",
|
||||
"corp",
|
||||
"corporation",
|
||||
"ltd",
|
||||
"limited",
|
||||
"llc",
|
||||
"gmbh",
|
||||
"ag",
|
||||
"sas",
|
||||
"sarl",
|
||||
# Consulting / staffing suffixes (agency hints)
|
||||
"consulting",
|
||||
"consultancy",
|
||||
"consult",
|
||||
"recruitment",
|
||||
"staffing",
|
||||
"solutions",
|
||||
"services",
|
||||
"group",
|
||||
"partners",
|
||||
]
|
||||
|
||||
# Pre-compile regex for trailing suffix removal.
|
||||
_SUFFIX_RE = re.compile(
|
||||
r"\s+(" + "|".join(re.escape(s) for s in _AGENCY_SUFFIXES) + r")\.?\s*$",
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
|
||||
# Characters to collapse: punctuation -> space, then multi-space -> single.
|
||||
_PUNCT_RE = re.compile(r"[^\w\s]")
|
||||
_WS_RE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def normalize_employer(name: str) -> str:
|
||||
"""Normalize an employer/company name for comparison.
|
||||
|
||||
Steps:
|
||||
1. lowercase
|
||||
2. strip trailing agency/legal suffixes (ab, consulting, etc.)
|
||||
3. remove punctuation
|
||||
4. collapse whitespace
|
||||
|
||||
Examples:
|
||||
>>> normalize_employer("Acme Consulting AB")
|
||||
'acme'
|
||||
>>> normalize_employer("Acme AB")
|
||||
'acme'
|
||||
>>> normalize_employer(" Globex Corp. ")
|
||||
'globex'
|
||||
"""
|
||||
if not name:
|
||||
return ""
|
||||
s = name.strip().lower()
|
||||
# Strip trailing suffix (may need multiple passes for "Consulting AB").
|
||||
for _ in range(3):
|
||||
new = _SUFFIX_RE.sub("", s)
|
||||
if new == s:
|
||||
break
|
||||
s = new
|
||||
# Remove punctuation.
|
||||
s = _PUNCT_RE.sub(" ", s)
|
||||
s = _WS_RE.sub(" ", s).strip()
|
||||
return s
|
||||
|
||||
|
||||
def title_score(a: str, b: str) -> float:
|
||||
"""Token-set ratio score for two job titles (0-100).
|
||||
|
||||
Uses rapidfuzz ``fuzz.token_set_ratio`` which is order-independent
|
||||
and handles subsets well.
|
||||
"""
|
||||
if not a or not b:
|
||||
return 0.0
|
||||
return float(fuzz.token_set_ratio(a, b))
|
||||
|
||||
|
||||
def employer_match(a: str, b: str) -> bool:
|
||||
"""True if two employer names normalize to the same string."""
|
||||
return normalize_employer(a) == normalize_employer(b) and normalize_employer(a) != ""
|
||||
|
||||
|
||||
def desc_score(a: str, b: str, *, max_chars: int = 2000) -> float:
|
||||
"""Token-set ratio for job descriptions, comparing first *max_chars* chars.
|
||||
|
||||
Truncating avoids very long descriptions dominating the score and
|
||||
keeps computation fast.
|
||||
"""
|
||||
if not a or not b:
|
||||
return 0.0
|
||||
return float(fuzz.token_set_ratio(a[:max_chars], b[:max_chars]))
|
||||
|
|
@ -1,184 +0,0 @@
|
|||
"""Shared fixtures for matching tests.
|
||||
|
||||
Provides agency-repost fixture triples and a legit-different-jobs-same-agency
|
||||
negative case.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Agency-repost fixture triples.
|
||||
# Three realistic scenarios where agencies repost the same job.
|
||||
# Each triple is a list of 3 posting dicts that should all cluster together.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Triple 1: Same role, two agencies + direct employer posting.
|
||||
# All three have the same title and nearly identical description, but
|
||||
# different employer names (the two agencies vs the actual company).
|
||||
TRIPLE_1_AGENCY_REPOST = [
|
||||
{
|
||||
"id": "t1-a",
|
||||
"employer": "TechCorp AB",
|
||||
"title": "Senior Python Developer",
|
||||
"description": (
|
||||
"We are looking for a Senior Python Developer to join our backend team. "
|
||||
"You will work with Fast API, PostgreSQL, and Docker in a cloud-native "
|
||||
"environment. 5+ years of Python experience required. "
|
||||
"Experience with AWS and Kubernetes is a plus."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "t1-b",
|
||||
"employer": "Nordic IT Consulting AB",
|
||||
"title": "Senior Python Developer",
|
||||
"description": (
|
||||
"We are looking for a Senior Python Developer to join our backend team. "
|
||||
"You will work with Fast API, PostgreSQL, and Docker in a cloud-native "
|
||||
"environment. 5+ years of Python experience required. "
|
||||
"Experience with AWS and Kubernetes is a plus."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "t1-c",
|
||||
"employer": "Acme Recruitment Group",
|
||||
"title": "Senior Python Developer",
|
||||
"description": (
|
||||
"We are looking for a Senior Python Developer to join our backend team. "
|
||||
"You will work with Fast API, PostgreSQL, and Docker in a cloud-native "
|
||||
"environment. 5+ years of Python experience required. "
|
||||
"Experience with AWS and Kubernetes is a plus."
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
# Triple 2: Same role reposted by same agency with minor wording variations.
|
||||
TRIPLE_2_SAME_AGENCY_REPOST = [
|
||||
{
|
||||
"id": "t2-a",
|
||||
"employer": "Stockholm Tech Staffing AB",
|
||||
"title": "Fullstack Engineer",
|
||||
"description": (
|
||||
"Fullstack Engineer wanted for a fintech startup in Stockholm. "
|
||||
"Tech stack: React, TypeScript, Node.js, PostgreSQL. "
|
||||
"You will build customer-facing features and internal tools. "
|
||||
"Must have experience with CI/CD pipelines."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "t2-b",
|
||||
"employer": "Stockholm Tech Staffing AB",
|
||||
"title": "Fullstack Engineer",
|
||||
"description": (
|
||||
"Fullstack Engineer wanted for a fintech startup in Stockholm. "
|
||||
"Tech stack: React, TypeScript, Node.js, PostgreSQL. "
|
||||
"You will build customer-facing features and internal tools. "
|
||||
"Must have experience with CI/CD pipelines."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "t2-c",
|
||||
"employer": "Stockholm Tech Staffing",
|
||||
"title": "Fullstack Engineer",
|
||||
"description": (
|
||||
"Fullstack Engineer wanted for a fintech startup in Stockholm. "
|
||||
"Tech stack: React, TypeScript, Node.js, PostgreSQL. "
|
||||
"You will build customer-facing features and internal tools. "
|
||||
"Must have experience with CI/CD pipelines."
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
# Triple 3: Same role, slightly different title but same description body.
|
||||
# Different employers (agencies), high title + desc similarity.
|
||||
TRIPLE_3_CROSS_AGENCY = [
|
||||
{
|
||||
"id": "t3-a",
|
||||
"employer": "Data Recruiting Solutions",
|
||||
"title": "Data Engineer",
|
||||
"description": (
|
||||
"We seek a Data Engineer to build and maintain ETL pipelines using "
|
||||
"Python, Airflow, dbt, and Snowflake. You will design data models, "
|
||||
"optimize queries, and ensure data quality. Experience with "
|
||||
"distributed systems and Spark is required."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "t3-b",
|
||||
"employer": "Cloud Talent Partners",
|
||||
"title": "Data Engineer",
|
||||
"description": (
|
||||
"We seek a Data Engineer to build and maintain ETL pipelines using "
|
||||
"Python, Airflow, dbt, and Snowflake. You will design data models, "
|
||||
"optimize queries, and ensure data quality. Experience with "
|
||||
"distributed systems and Spark is required."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "t3-c",
|
||||
"employer": "Analytics Staffing Ltd",
|
||||
"title": "Data Engineer",
|
||||
"description": (
|
||||
"We seek a Data Engineer to build and maintain ETL pipelines using "
|
||||
"Python, Airflow, dbt, and Snowflake. You will design data models, "
|
||||
"optimize queries, and ensure data quality. Experience with "
|
||||
"distributed systems and Spark is required."
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# NEGATIVE case: legit different jobs at same agency -- must NOT cluster.
|
||||
# Same agency employer but different titles and different descriptions.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
NEGATIVE_DIFFERENT_JOBS_SAME_AGENCY = [
|
||||
{
|
||||
"id": "neg-a",
|
||||
"employer": "Nordic IT Consulting AB",
|
||||
"title": "Frontend Developer",
|
||||
"description": (
|
||||
"We are looking for a Frontend Developer with expertise in React "
|
||||
"and TypeScript. You will build responsive web applications and "
|
||||
"work closely with our design team. Experience with CSS-in-JS and "
|
||||
"accessibility standards is required."
|
||||
),
|
||||
},
|
||||
{
|
||||
"id": "neg-b",
|
||||
"employer": "Nordic IT Consulting AB",
|
||||
"title": "DevOps Engineer",
|
||||
"description": (
|
||||
"We need a DevOps Engineer to manage our Kubernetes clusters and "
|
||||
"CI/CD pipelines. You will work with Terraform, ArgoCD, and "
|
||||
"Prometheus monitoring. Strong Linux and networking background "
|
||||
"is required. AWS certification is a plus."
|
||||
),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def triple1():
|
||||
"""Agency repost triple 1: same role, two agencies + employer."""
|
||||
return [dict(p) for p in TRIPLE_1_AGENCY_REPOST]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def triple2():
|
||||
"""Agency repost triple 2: same agency reposts same job."""
|
||||
return [dict(p) for p in TRIPLE_2_SAME_AGENCY_REPOST]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def triple3():
|
||||
"""Agency repost triple 3: cross-agency same role same description."""
|
||||
return [dict(p) for p in TRIPLE_3_CROSS_AGENCY]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def negative_same_agency():
|
||||
"""Negative case: different jobs at same agency, must NOT cluster."""
|
||||
return [dict(p) for p in NEGATIVE_DIFFERENT_JOBS_SAME_AGENCY]
|
||||
|
|
@ -1,168 +0,0 @@
|
|||
"""Tests for dedupe clustering."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from matching.dedupe import cluster
|
||||
|
||||
|
||||
class TestClusterBasic:
|
||||
def test_empty_list(self):
|
||||
assert cluster([]) == {}
|
||||
|
||||
def test_single_posting(self):
|
||||
result = cluster([
|
||||
{"id": "a", "employer": "Acme AB", "title": "Dev", "description": "x"}
|
||||
])
|
||||
assert len(result) == 1
|
||||
assert "c1" in result
|
||||
assert result["c1"] == ["a"]
|
||||
|
||||
def test_no_duplicates_separate_clusters(self):
|
||||
postings = [
|
||||
{"id": "a", "employer": "Acme AB", "title": "Python Dev", "description": "Python backend"},
|
||||
{"id": "b", "employer": "Globex AB", "title": "React Dev", "description": "React frontend"},
|
||||
{"id": "c", "employer": "Foo Ltd", "title": "Data Scientist", "description": "ML pipelines"},
|
||||
]
|
||||
result = cluster(postings)
|
||||
# Each posting in its own cluster.
|
||||
total_ids = sum(len(v) for v in result.values())
|
||||
assert total_ids == 3
|
||||
# All cluster values are singletons.
|
||||
for ids in result.values():
|
||||
assert len(ids) == 1
|
||||
|
||||
|
||||
class TestClusterEmployerMatch:
|
||||
def test_same_employer_clusters(self):
|
||||
"""Same employer name (normalized) with similar content should cluster."""
|
||||
postings = [
|
||||
{"id": "a", "employer": "Acme AB", "title": "Python Developer", "description": "Python backend development"},
|
||||
{"id": "b", "employer": "Acme AB", "title": "Python Developer", "description": "Python backend development senior"},
|
||||
]
|
||||
result = cluster(postings)
|
||||
# Same employer + similar title -> same cluster.
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"a", "b"}
|
||||
|
||||
def test_same_employer_different_content_no_cluster(self):
|
||||
"""Same employer but completely different titles/descriptions should NOT cluster."""
|
||||
postings = [
|
||||
{"id": "a", "employer": "Acme AB", "title": "Python Developer", "description": "We need a Python developer for backend work."},
|
||||
{"id": "b", "employer": "Acme AB", "title": "Chef", "description": "Looking for a head chef for our restaurant kitchen."},
|
||||
]
|
||||
result = cluster(postings)
|
||||
# Same employer but different jobs -> no cluster.
|
||||
assert all(len(v) == 1 for v in result.values())
|
||||
|
||||
def test_employer_suffix_variations_cluster(self):
|
||||
"""Acme AB and Acme should cluster (normalized match)."""
|
||||
postings = [
|
||||
{"id": "a", "employer": "Acme AB", "title": "Dev", "description": "x"},
|
||||
{"id": "b", "employer": "Acme", "title": "Dev", "description": "y"},
|
||||
]
|
||||
result = cluster(postings)
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"a", "b"}
|
||||
|
||||
|
||||
class TestClusterTitleDescMatch:
|
||||
def test_title_desc_high_enough(self):
|
||||
"""Different employers but title >= 85 and desc >= 80 -> cluster."""
|
||||
desc = (
|
||||
"We are looking for a Senior Python Developer to join our backend "
|
||||
"team. You will work with Fast API, PostgreSQL, and Docker."
|
||||
)
|
||||
postings = [
|
||||
{"id": "a", "employer": "Agency One AB", "title": "Senior Python Developer", "description": desc},
|
||||
{"id": "b", "employer": "Agency Two AB", "title": "Senior Python Developer", "description": desc},
|
||||
]
|
||||
result = cluster(postings)
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"a", "b"}
|
||||
|
||||
def test_title_high_desc_low_no_cluster(self):
|
||||
"""Title similar but desc too different -> no cluster."""
|
||||
postings = [
|
||||
{"id": "a", "employer": "Agency A", "title": "Python Developer", "description": "We need a Python developer for backend work with Django."},
|
||||
{"id": "b", "employer": "Agency B", "title": "Python Developer", "description": "Looking for someone to teach Python to high school students."},
|
||||
]
|
||||
result = cluster(postings)
|
||||
# Should NOT cluster (different employers, low desc score).
|
||||
assert len(result) == 2 or all(len(v) == 1 for v in result.values())
|
||||
|
||||
|
||||
class TestClusterTransitive:
|
||||
def test_transitive_clustering(self):
|
||||
"""If A~B and B~C then A~C should be in same cluster."""
|
||||
# A and B same employer + similar title, B and C same employer + similar title.
|
||||
postings = [
|
||||
{"id": "a", "employer": "Acme AB", "title": "Python Developer", "description": "Python backend development"},
|
||||
{"id": "b", "employer": "Acme AB", "title": "Python Developer", "description": "Python backend development senior"},
|
||||
{"id": "c", "employer": "Acme AB", "title": "Python Developer", "description": "Python backend development lead"},
|
||||
]
|
||||
result = cluster(postings)
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"a", "b", "c"}
|
||||
|
||||
|
||||
class TestClusterSorting:
|
||||
def test_cluster_ids_sorted_by_score(self):
|
||||
"""Cluster with higher pairwise score should get c1."""
|
||||
# Tight cluster: identical titles and descriptions (different employers).
|
||||
tight_desc = "Python backend developer with Fast API and PostgreSQL and Docker and AWS and Kubernetes."
|
||||
# Looser cluster: different employers, high title but lower desc similarity.
|
||||
loose_desc_1 = "Python data engineering and pipelines with ETL tools."
|
||||
loose_desc_2 = "Python data engineering and ETL work with Airflow."
|
||||
postings = [
|
||||
# tight cluster (different employers, high title+desc)
|
||||
{"id": "t1", "employer": "Agency A", "title": "Python Developer", "description": tight_desc},
|
||||
{"id": "t2", "employer": "Agency B", "title": "Python Developer", "description": tight_desc},
|
||||
# loose cluster (different employers, high title but lower desc)
|
||||
{"id": "l1", "employer": "Agency C", "title": "Python Developer", "description": loose_desc_1},
|
||||
{"id": "l2", "employer": "Agency D", "title": "Python Developer", "description": loose_desc_2},
|
||||
]
|
||||
result = cluster(postings)
|
||||
# Both clusters should exist.
|
||||
all_ids = set()
|
||||
for ids in result.values():
|
||||
all_ids.update(ids)
|
||||
assert all_ids == {"t1", "t2", "l1", "l2"}
|
||||
# c1 should be the tight cluster (higher score: title+desc both 100).
|
||||
assert set(result["c1"]) == {"t1", "t2"}
|
||||
|
||||
|
||||
class TestAgencyRepostTriples:
|
||||
"""Test the 3 agency-repost fixture triples."""
|
||||
|
||||
def test_triple1_all_cluster(self, triple1):
|
||||
"""Triple 1: 3 postings of same role via different employers cluster."""
|
||||
result = cluster(triple1)
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"t1-a", "t1-b", "t1-c"}
|
||||
|
||||
def test_triple2_all_cluster(self, triple2):
|
||||
"""Triple 2: same agency reposts same job (suffix variations)."""
|
||||
result = cluster(triple2)
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"t2-a", "t2-b", "t2-c"}
|
||||
|
||||
def test_triple3_all_cluster(self, triple3):
|
||||
"""Triple 3: cross-agency same role same description."""
|
||||
result = cluster(triple3)
|
||||
assert len(result) == 1
|
||||
assert set(result["c1"]) == {"t3-a", "t3-b", "t3-c"}
|
||||
|
||||
|
||||
class TestNegativeSameAgencyDifferentJobs:
|
||||
"""Negative case: different jobs at same agency must NOT cluster."""
|
||||
|
||||
def test_different_jobs_same_agency_no_cluster(self, negative_same_agency):
|
||||
"""Different jobs at same agency must NOT cluster.
|
||||
|
||||
Same employer but completely different titles and descriptions.
|
||||
Per the clustering rule, same employer alone is not sufficient;
|
||||
some content overlap (title >= 85 OR desc >= 80) is also required.
|
||||
"""
|
||||
result = cluster(negative_same_agency)
|
||||
all_singletons = all(len(v) == 1 for v in result.values())
|
||||
assert all_singletons, "Different jobs at same agency should not cluster"
|
||||
|
|
@ -1,152 +0,0 @@
|
|||
"""Tests for keyword extraction and coverage."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from matching.keywords import extract_keywords, coverage
|
||||
|
||||
|
||||
class TestExtractKeywords:
|
||||
def test_basic_extraction(self):
|
||||
text = "Python developer with Fast API experience and PostgreSQL database skills."
|
||||
kws = extract_keywords(text)
|
||||
assert "python" in kws
|
||||
assert "fastapi" in kws
|
||||
assert "postgresql" in kws
|
||||
|
||||
def test_stopwords_removed(self):
|
||||
text = "We are looking for a developer with experience in Python."
|
||||
kws = extract_keywords(text)
|
||||
assert "we" not in kws
|
||||
assert "are" not in kws
|
||||
assert "for" not in kws
|
||||
assert "a" not in kws
|
||||
assert "in" not in kws
|
||||
assert "python" in kws
|
||||
assert "developer" in kws
|
||||
|
||||
def test_swedish_stopwords_removed(self):
|
||||
text = "Vi letar efter en Python utvecklare med erfarenhet av Docker."
|
||||
kws = extract_keywords(text)
|
||||
assert "vi" not in kws
|
||||
assert "en" not in kws
|
||||
assert "av" not in kws
|
||||
assert "python" in kws
|
||||
assert "docker" in kws
|
||||
|
||||
def test_top_n_limit(self):
|
||||
text = "python python python docker docker docker kubernetes kubernetes kubernetes react react react"
|
||||
kws = extract_keywords(text, top_n=2)
|
||||
assert len(kws) == 2
|
||||
|
||||
def test_empty_text(self):
|
||||
assert extract_keywords("") == []
|
||||
|
||||
def test_whitespace_only(self):
|
||||
assert extract_keywords(" ") == []
|
||||
|
||||
def test_multiword_fastapi(self):
|
||||
text = "Experience with fast api framework for building REST APIs."
|
||||
kws = extract_keywords(text)
|
||||
assert "fastapi" in kws
|
||||
|
||||
def test_multiword_machine_learning(self):
|
||||
text = "machine learning models for predictive analytics."
|
||||
kws = extract_keywords(text)
|
||||
assert "machine-learning" in kws
|
||||
|
||||
def test_frequency_ordering(self):
|
||||
text = "python python python docker docker kubernetes"
|
||||
kws = extract_keywords(text, top_n=3)
|
||||
assert kws[0] == "python"
|
||||
assert kws[1] == "docker"
|
||||
assert kws[2] == "kubernetes"
|
||||
|
||||
def test_deterministic_tie_breaking(self):
|
||||
"""Ties in frequency should be broken alphabetically."""
|
||||
text = "docker kubernetes"
|
||||
kws = extract_keywords(text)
|
||||
# Both have frequency 1, so alphabetical: docker < kubernetes
|
||||
assert kws[0] == "docker"
|
||||
assert kws[1] == "kubernetes"
|
||||
|
||||
def test_min_token_length(self):
|
||||
text = "x y z aa bb cc developer"
|
||||
kws = extract_keywords(text)
|
||||
assert "x" not in kws
|
||||
assert "y" not in kws
|
||||
assert "z" not in kws
|
||||
assert "developer" in kws
|
||||
|
||||
def test_tech_terms_preserved(self):
|
||||
text = "Node.js and React Native for mobile development."
|
||||
kws = extract_keywords(text)
|
||||
assert "nodejs" in kws
|
||||
assert "react-native" in kws
|
||||
|
||||
|
||||
class TestCoverage:
|
||||
def test_full_coverage(self):
|
||||
cv = "Python developer with Fast API PostgreSQL Docker AWS Kubernetes"
|
||||
posting = "Python developer with Fast API PostgreSQL Docker AWS Kubernetes"
|
||||
result = coverage(cv, posting)
|
||||
assert result["ratio"] == 1.0
|
||||
assert len(result["missing"]) == 0
|
||||
|
||||
def test_partial_coverage(self):
|
||||
cv = "Python developer with PostgreSQL and Docker experience"
|
||||
posting = "Python developer with Fast API PostgreSQL Docker AWS Kubernetes React"
|
||||
result = coverage(cv, posting)
|
||||
assert 0.0 < result["ratio"] < 1.0
|
||||
assert "python" in result["matched"]
|
||||
assert "postgresql" in result["matched"]
|
||||
assert "docker" in result["matched"]
|
||||
assert "fastapi" in result["missing"]
|
||||
assert "kubernetes" in result["missing"]
|
||||
|
||||
def test_zero_coverage(self):
|
||||
cv = "Chef with experience in French cuisine and menu planning"
|
||||
posting = "Python developer with Fast API PostgreSQL Docker"
|
||||
result = coverage(cv, posting)
|
||||
assert result["ratio"] == 0.0
|
||||
assert len(result["matched"]) == 0
|
||||
assert len(result["missing"]) > 0
|
||||
|
||||
def test_empty_posting(self):
|
||||
result = coverage("Python developer", "")
|
||||
assert result == {"matched": [], "missing": [], "ratio": 0.0}
|
||||
|
||||
def test_empty_cv(self):
|
||||
result = coverage("", "Python developer with Docker")
|
||||
assert result["ratio"] == 0.0
|
||||
assert len(result["matched"]) == 0
|
||||
assert len(result["missing"]) > 0
|
||||
|
||||
def test_both_empty(self):
|
||||
result = coverage("", "")
|
||||
assert result == {"matched": [], "missing": [], "ratio": 0.0}
|
||||
|
||||
def test_ratio_calculation(self):
|
||||
cv = "python docker postgresql"
|
||||
posting = "python docker postgresql kubernetes"
|
||||
result = coverage(cv, posting)
|
||||
# 3 of 4 matched (approx, depends on stopword filtering).
|
||||
assert result["ratio"] > 0.5
|
||||
assert result["ratio"] <= 1.0
|
||||
|
||||
def test_matched_and_missing_lists(self):
|
||||
cv = "python docker"
|
||||
posting = "python docker kubernetes react"
|
||||
result = coverage(cv, posting)
|
||||
assert "python" in result["matched"]
|
||||
assert "docker" in result["matched"]
|
||||
assert "kubernetes" in result["missing"]
|
||||
assert "react" in result["missing"]
|
||||
|
||||
def test_coverage_returns_dict_keys(self):
|
||||
result = coverage("python", "python docker")
|
||||
assert "matched" in result
|
||||
assert "missing" in result
|
||||
assert "ratio" in result
|
||||
assert isinstance(result["matched"], list)
|
||||
assert isinstance(result["missing"], list)
|
||||
assert isinstance(result["ratio"], float)
|
||||
|
|
@ -1,126 +0,0 @@
|
|||
"""Tests for similarity helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from matching.similarity import (
|
||||
normalize_employer,
|
||||
title_score,
|
||||
employer_match,
|
||||
desc_score,
|
||||
)
|
||||
|
||||
|
||||
class TestNormalizeEmployer:
|
||||
def test_simple_lowercase(self):
|
||||
assert normalize_employer("Acme") == "acme"
|
||||
|
||||
def test_strips_swedish_ab(self):
|
||||
assert normalize_employer("Acme AB") == "acme"
|
||||
|
||||
def test_strips_aktiebolag(self):
|
||||
assert normalize_employer("Acme Aktiebolag") == "acme"
|
||||
|
||||
def test_strips_consulting_suffix(self):
|
||||
assert normalize_employer("Nordic IT Consulting AB") == "nordic it"
|
||||
|
||||
def test_strips_corp_suffix(self):
|
||||
assert normalize_employer("Globex Corp.") == "globex"
|
||||
|
||||
def test_strips_ltd_suffix(self):
|
||||
assert normalize_employer("Foo Ltd") == "foo"
|
||||
|
||||
def test_strips_recruitment_suffix(self):
|
||||
assert normalize_employer("Acme Recruitment Group") == "acme"
|
||||
|
||||
def test_strips_multiple_suffixes(self):
|
||||
# "Consulting AB" should strip both "AB" then "Consulting"
|
||||
assert normalize_employer("Nordic Consulting AB") == "nordic"
|
||||
|
||||
def test_removes_punctuation(self):
|
||||
assert normalize_employer("Acme, Inc.") == "acme"
|
||||
|
||||
def test_empty_string(self):
|
||||
assert normalize_employer("") == ""
|
||||
|
||||
def test_whitespace_only(self):
|
||||
assert normalize_employer(" ") == ""
|
||||
|
||||
def test_preserves_core_name_with_special_chars(self):
|
||||
result = normalize_employer("Café Nu AB")
|
||||
assert "café" in result or "cafe" in result
|
||||
|
||||
def test_dots_in_name_preserved(self):
|
||||
# Punctuation (except & which gets stripped) is removed; H&M -> h m
|
||||
result = normalize_employer("H&M AB")
|
||||
assert result == "h m"
|
||||
|
||||
|
||||
class TestTitleScore:
|
||||
def test_identical_titles(self):
|
||||
assert title_score("Senior Python Developer", "Senior Python Developer") == 100.0
|
||||
|
||||
def test_similar_titles_high_score(self):
|
||||
score = title_score("Python Developer", "Senior Python Developer")
|
||||
assert score >= 85.0
|
||||
|
||||
def test_different_titles_low_score(self):
|
||||
score = title_score("Python Developer", "Frontend Designer")
|
||||
assert score < 50.0
|
||||
|
||||
def test_empty_title(self):
|
||||
assert title_score("", "Something") == 0.0
|
||||
|
||||
def test_both_empty(self):
|
||||
assert title_score("", "") == 0.0
|
||||
|
||||
def test_order_independent(self):
|
||||
# token_set_ratio is order-independent
|
||||
a = "Senior Python Developer"
|
||||
b = "Developer Python Senior"
|
||||
assert title_score(a, b) == 100.0
|
||||
|
||||
|
||||
class TestEmployerMatch:
|
||||
def test_same_name_matches(self):
|
||||
assert employer_match("Acme AB", "Acme AB") is True
|
||||
|
||||
def test_suffix_variation_matches(self):
|
||||
assert employer_match("Acme AB", "Acme") is True
|
||||
|
||||
def test_different_employers_no_match(self):
|
||||
assert employer_match("Acme AB", "Globex AB") is False
|
||||
|
||||
def test_consulting_variations_match(self):
|
||||
assert employer_match("Nordic IT Consulting AB", "Nordic IT") is True
|
||||
|
||||
def test_empty_no_match(self):
|
||||
assert employer_match("", "") is False
|
||||
|
||||
def test_one_empty_no_match(self):
|
||||
assert employer_match("Acme", "") is False
|
||||
|
||||
|
||||
class TestDescScore:
|
||||
def test_identical_descriptions(self):
|
||||
desc = "We are looking for a Python developer with 5 years experience."
|
||||
assert desc_score(desc, desc) == 100.0
|
||||
|
||||
def test_similar_descriptions_high(self):
|
||||
a = "We are looking for a Python developer with 5 years experience."
|
||||
b = "We are looking for a Python developer with 5 years experience in web."
|
||||
assert desc_score(a, b) >= 80.0
|
||||
|
||||
def test_different_descriptions_low(self):
|
||||
a = "We need a frontend developer skilled in React and CSS."
|
||||
b = "Looking for a data scientist with Python and SQL expertise."
|
||||
assert desc_score(a, b) < 50.0
|
||||
|
||||
def test_empty_desc(self):
|
||||
assert desc_score("", "something") == 0.0
|
||||
|
||||
def test_truncation(self):
|
||||
# Test that truncation to max_chars works.
|
||||
long_a = "Python " * 1000
|
||||
long_b = "Python " * 1000
|
||||
score = desc_score(long_a, long_b, max_chars=100)
|
||||
assert score == 100.0
|
||||
|
|
@ -1,5 +0,0 @@
|
|||
FROM mcr.microsoft.com/playwright/python:v1.55.0-jammy
|
||||
WORKDIR /shots
|
||||
RUN pip install --no-cache-dir playwright==1.55.0
|
||||
COPY scripts/screenshots.py ./
|
||||
CMD ["python", "screenshots.py"]
|
||||
|
|
@ -1,84 +0,0 @@
|
|||
"""Screenshot runner: drives the seed demo UI and saves PNGs to /out.
|
||||
|
||||
Runs inside the compose network; browser resolves 'web' and 'api' directly.
|
||||
"""
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
|
||||
from playwright.sync_api import sync_playwright
|
||||
|
||||
BASE = "http://web"
|
||||
API = "http://api:8000/api"
|
||||
OUT = "/out"
|
||||
|
||||
|
||||
def api(method, path, body=None):
|
||||
req = urllib.request.Request(
|
||||
API + path,
|
||||
data=json.dumps(body).encode() if body else None,
|
||||
method=method,
|
||||
headers={"Content-Type": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read())
|
||||
|
||||
|
||||
def main():
|
||||
# Idempotent: ensures full demo dataset exists
|
||||
seed = api("POST", "/concierge/seed-demo")
|
||||
print("seed:", json.dumps(seed)[:200])
|
||||
|
||||
apps = api("GET", "/applications")
|
||||
interview = next((a for a in apps if a["state"] == "interviewing"), apps[0])
|
||||
detail_id = interview["id"]
|
||||
print("detail id:", detail_id)
|
||||
|
||||
shots = [
|
||||
("welcome", "/", {}), # first-run wizard may redirect away; force below
|
||||
("welcome", "/welcome", {}),
|
||||
("today", "/today", {}),
|
||||
("cv", "/cv", {}),
|
||||
("research", "/research", {}),
|
||||
("applications", "/applications", {}),
|
||||
("detail", f"/applications/{detail_id}", {}),
|
||||
]
|
||||
|
||||
with sync_playwright() as p:
|
||||
browser = p.chromium.launch(args=["--disable-dev-shm-usage"])
|
||||
page = browser.new_page(viewport={"width": 1440, "height": 900},
|
||||
device_scale_factor=2)
|
||||
for name, path, _opts in shots[1:]: # skip the "/" duplicate
|
||||
try:
|
||||
page.goto(BASE + path, wait_until="networkidle", timeout=30000)
|
||||
page.wait_for_timeout(1200)
|
||||
# close possible wizard redirect back
|
||||
if name != "welcome" and page.url.endswith("/welcome"):
|
||||
page.goto(BASE + path, wait_until="networkidle", timeout=30000)
|
||||
page.wait_for_timeout(800)
|
||||
page.screenshot(path=f"{OUT}/{name}.png")
|
||||
print("shot:", name, "<-", page.url)
|
||||
except Exception as e:
|
||||
print("FAILED:", name, type(e).__name__, str(e)[:150])
|
||||
|
||||
# Interaction shot: open the Tailor CV panel on an approved application
|
||||
approved = next((a for a in apps if a["state"] in ("approved", "interviewing")), apps[0])
|
||||
try:
|
||||
page.goto(f"{BASE}/applications/{approved['id']}", wait_until="networkidle", timeout=30000)
|
||||
page.wait_for_timeout(1000)
|
||||
btn = page.locator('[data-testid="tailor-cv-btn"]')
|
||||
btn.click(timeout=8000)
|
||||
page.wait_for_selector('[data-testid="tailor-panel"]', timeout=20000)
|
||||
page.wait_for_timeout(800)
|
||||
page.screenshot(path=f"{OUT}/detail-tailor.png")
|
||||
print("shot: detail-tailor")
|
||||
except Exception as e:
|
||||
print("FAILED: detail-tailor", type(e).__name__, str(e)[:150])
|
||||
|
||||
browser.close()
|
||||
print("DONE")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Loading…
Reference in a new issue