173187dcbb
The structured model and StructuredCvProfileJson.FromSections already map
Projects/Certifications/Languages headings, but the AI normalize prompt
never emitted them, so on the benchmark CV the entire Projects section and
the in-summary languages (English Native, Norwegian B1) were silently
dropped. This closes that gap upstream — no backend schema or data change.
ai-service (tools/summarizer/app.py):
- /cv/normalize: added # Projects and # Certifications headings; a
languages-from-prose rule (pull "native English", "Norwegian at B1" out
of the summary even with no Languages section; ignore programming
languages); and skill-group prefix stripping ("Development:",
"DevOps & Infrastructure:", "Practices:" dropped, only the skills kept).
- /cv/classify-block: Projects and Certifications added to the section
enum + rules (fallback path).
Backend:
- LooksLikeNormalizedMarkdownCv now recognises # Projects / # Certifications
so those CVs still take the markdown assembly path.
Tests:
- CvExtractionCoverageTests (4) lock the C# mapping of Projects,
Certifications and Languages sections into the structured profile.
- ai-service test_classify_block_supports_projects_section (1).
426 backend tests, 17 ai-service tests pass; app.py compiles.
The LLM behaviour (prompt -> headings) needs Ollama to observe and was not
run here; the C# side that consumes the headings is proven and the prompt
change is additive. Merge-not-replace + the review screen are the next
increment (2.1-a, approved: always-review, conservative merge).
Deployment: these prompts live in the ai-service container, which
deploy.sh does not rebuild by default -- deploy with
DEPLOY_BUILD_AI_SERVICE=true or the change won't take effect.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
339 lines
12 KiB
Python
339 lines
12 KiB
Python
import importlib
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
from fastapi.testclient import TestClient
|
|
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
if str(ROOT) not in sys.path:
|
|
sys.path.insert(0, str(ROOT))
|
|
|
|
|
|
def load_app_module(monkeypatch, *, skip_model_load=True, ollama_model=None, service_token=None):
|
|
if skip_model_load:
|
|
monkeypatch.setenv("AI_SERVICE_SKIP_MODEL_LOAD", "1")
|
|
else:
|
|
monkeypatch.delenv("AI_SERVICE_SKIP_MODEL_LOAD", raising=False)
|
|
monkeypatch.delenv("AI_SERVICE_EAGER_MODEL_LOAD", raising=False)
|
|
# Default to keyless so the existing suite is unaffected by a token in the dev shell.
|
|
if service_token is None:
|
|
monkeypatch.delenv("AI_SERVICE_TOKEN", raising=False)
|
|
else:
|
|
monkeypatch.setenv("AI_SERVICE_TOKEN", service_token)
|
|
if ollama_model is None:
|
|
monkeypatch.delenv("OLLAMA_MODEL", raising=False)
|
|
else:
|
|
monkeypatch.setenv("OLLAMA_MODEL", ollama_model)
|
|
if "app" in sys.modules:
|
|
del sys.modules["app"]
|
|
module = importlib.import_module("app")
|
|
return importlib.reload(module)
|
|
|
|
|
|
def test_health_reports_runtime_without_ollama_and_without_forcing_model_load(monkeypatch):
|
|
module = load_app_module(monkeypatch)
|
|
client = TestClient(module.app)
|
|
|
|
response = client.get("/health")
|
|
|
|
assert response.status_code == 200
|
|
payload = response.json()
|
|
assert payload["ok"] is True
|
|
assert payload["device"] == "cpu"
|
|
assert payload["model_loaded"] is False
|
|
assert payload["model_disabled"] is True
|
|
assert payload["summarize_available"] is False
|
|
assert "disabled" in payload["model_load_error"].lower()
|
|
assert payload["ollama_configured"] is False
|
|
assert payload["ollama_model"] is None
|
|
assert payload["ollama_installed_models"] == []
|
|
assert payload["ollama_loaded_models"] == []
|
|
|
|
|
|
def test_summarize_returns_503_with_explicit_reason_when_model_loading_is_disabled(monkeypatch):
|
|
module = load_app_module(monkeypatch)
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post("/summarize", json={"text": "Platform engineering role with APIs and Python experience."})
|
|
|
|
assert response.status_code == 503
|
|
payload = response.json()
|
|
assert "disabled" in payload["detail"].lower()
|
|
|
|
|
|
def test_health_reports_ollama_unreachable_when_configured_but_not_available(monkeypatch):
|
|
module = load_app_module(monkeypatch, ollama_model="qwen2.5:7b")
|
|
|
|
def boom(path: str):
|
|
raise OSError("connection refused")
|
|
|
|
monkeypatch.setattr(module, "_ollama_json", boom)
|
|
client = TestClient(module.app)
|
|
|
|
response = client.get("/health")
|
|
|
|
assert response.status_code == 200
|
|
payload = response.json()
|
|
assert payload["ollama_configured"] is True
|
|
assert payload["ollama_reachable"] is False
|
|
assert payload["ollama_model"] == "qwen2.5:7b"
|
|
assert payload["ollama_model_available"] is False
|
|
|
|
|
|
def test_rewrite_cv_returns_plain_rewritten_text(monkeypatch):
|
|
module = load_app_module(monkeypatch, ollama_model="qwen2.5:7b")
|
|
monkeypatch.setattr(module, "_ollama_generate_text", lambda prompt: "# Professional Summary\nBuilt resilient backend systems.\n\n# Skills\n- C#\n- .NET")
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post("/cv/rewrite", json={
|
|
"instruction": "Rewrite this CV into a cleaner master CV.",
|
|
"text": "Professional Summary\nBuilt backend systems.",
|
|
"max_length": 220,
|
|
"min_length": 80,
|
|
})
|
|
|
|
assert response.status_code == 200
|
|
payload = response.json()
|
|
assert payload["rewritten_text"].startswith("# Professional Summary")
|
|
assert "Role summary:" not in payload["rewritten_text"]
|
|
|
|
|
|
def test_classify_block_returns_structured_json(monkeypatch):
|
|
module = load_app_module(monkeypatch)
|
|
|
|
def fake_generate_json(prompt: str):
|
|
assert "Senior Platform Engineer" in prompt
|
|
return {
|
|
"section": "Work Experience",
|
|
"confidence": 0.91,
|
|
"reason": "job block",
|
|
"title": "Senior Platform Engineer",
|
|
"company": "Atlas Systems",
|
|
"location": "Oslo",
|
|
"start": "2019",
|
|
"end": "Present",
|
|
"bullets": ["Built event-driven APIs and migration tooling."],
|
|
"summary": [],
|
|
"skills": ["Python", "SQL"],
|
|
}
|
|
|
|
monkeypatch.setattr(module, "_ollama_generate_json", fake_generate_json)
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post("/cv/classify-block", json={"block": "Senior Platform Engineer at Atlas Systems, Oslo, 2019 - Present. Built event-driven APIs and migration tooling."})
|
|
|
|
assert response.status_code == 200
|
|
payload = response.json()
|
|
assert payload["section"] == "Work Experience"
|
|
assert payload["title"] == "Senior Platform Engineer"
|
|
assert payload["company"] == "Atlas Systems"
|
|
assert payload["bullets"] == ["Built event-driven APIs and migration tooling."]
|
|
assert payload["summary"] == []
|
|
assert payload["skills"] == ["Python", "SQL"]
|
|
|
|
|
|
def test_classify_block_supports_projects_section(monkeypatch):
|
|
# Phase 2.1-b: Projects and Certifications are now valid classified sections so project blocks
|
|
# (e.g. the benchmark CV's JobTrack/InboxIntel) are no longer dropped into "Other".
|
|
module = load_app_module(monkeypatch)
|
|
|
|
def fake_generate_json(prompt: str):
|
|
assert "Projects" in prompt # the enum now advertises Projects to the model
|
|
return {
|
|
"section": "Projects",
|
|
"confidence": 0.83,
|
|
"reason": "project block",
|
|
"title": "JobTrack",
|
|
"company": None,
|
|
"location": None,
|
|
"start": None,
|
|
"end": None,
|
|
"bullets": ["Full-stack job-application tracker (React, ASP.NET Core, SQLite, Docker)."],
|
|
"summary": [],
|
|
"skills": [],
|
|
}
|
|
|
|
monkeypatch.setattr(module, "_ollama_generate_json", fake_generate_json)
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post("/cv/classify-block", json={"block": "JobTrack - Full-stack job-application tracker (React, ASP.NET Core, SQLite, Docker)."})
|
|
|
|
assert response.status_code == 200
|
|
payload = response.json()
|
|
assert payload["section"] == "Projects"
|
|
assert payload["title"] == "JobTrack"
|
|
|
|
|
|
def test_classify_block_defaults_missing_section_to_other(monkeypatch):
|
|
module = load_app_module(monkeypatch)
|
|
monkeypatch.setattr(module, "_ollama_generate_json", lambda prompt: {"bullets": []})
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post("/cv/classify-block", json={"block": "Miscellaneous profile text"})
|
|
|
|
assert response.status_code == 200
|
|
payload = response.json()
|
|
assert payload["section"] == "Other"
|
|
assert payload["bullets"] == []
|
|
assert payload["summary"] == []
|
|
assert payload["skills"] == []
|
|
|
|
|
|
# --- AI provider router -------------------------------------------------------
|
|
|
|
class _FakeResponse:
|
|
def __init__(self, payload):
|
|
self._data = json.dumps(payload).encode("utf-8")
|
|
|
|
def read(self):
|
|
return self._data
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *exc):
|
|
return False
|
|
|
|
|
|
def _install_fake_urlopen(monkeypatch, module, response_payload, captured):
|
|
def fake_urlopen(req, timeout=None):
|
|
captured["url"] = req.full_url
|
|
captured["headers"] = {k.lower(): v for k, v in req.header_items()}
|
|
captured["body"] = json.loads(req.data.decode("utf-8"))
|
|
return _FakeResponse(response_payload)
|
|
|
|
monkeypatch.setattr(module.urllib_request, "urlopen", fake_urlopen)
|
|
|
|
|
|
def test_provider_defaults_to_ollama_and_is_unchanged(monkeypatch):
|
|
monkeypatch.delenv("AI_PROVIDER", raising=False)
|
|
monkeypatch.setenv("OLLAMA_BASE_URL", "http://ollama-host:11434")
|
|
module = load_app_module(monkeypatch, ollama_model="qwen2.5:7b")
|
|
assert module.AI_PROVIDER == "ollama"
|
|
|
|
captured = {}
|
|
_install_fake_urlopen(monkeypatch, module, {"response": '{"score": 7}'}, captured)
|
|
|
|
assert module._ollama_generate_json("hi") == {"score": 7}
|
|
assert captured["url"] == "http://ollama-host:11434/api/generate"
|
|
assert captured["body"]["model"] == "qwen2.5:7b"
|
|
assert captured["body"]["format"] == "json"
|
|
assert captured["body"]["options"]["temperature"] == 0.1
|
|
|
|
|
|
def test_provider_gemini_dispatch(monkeypatch):
|
|
monkeypatch.setenv("AI_PROVIDER", "gemini")
|
|
monkeypatch.setenv("GEMINI_API_KEY", "test-key")
|
|
monkeypatch.setenv("GEMINI_MODEL", "gemini-2.0-flash")
|
|
module = load_app_module(monkeypatch)
|
|
|
|
captured = {}
|
|
payload = {"candidates": [{"content": {"parts": [{"text": '{"score": 9}'}]}}]}
|
|
_install_fake_urlopen(monkeypatch, module, payload, captured)
|
|
|
|
assert module._ollama_generate_json("hi") == {"score": 9}
|
|
assert "generativelanguage" in captured["url"]
|
|
assert "gemini-2.0-flash:generateContent" in captured["url"]
|
|
assert "key=" not in captured["url"] # key must not be in the URL
|
|
assert captured["headers"].get("x-goog-api-key") == "test-key"
|
|
assert captured["body"]["generationConfig"]["responseMimeType"] == "application/json"
|
|
|
|
|
|
def test_provider_groq_dispatch(monkeypatch):
|
|
monkeypatch.setenv("AI_PROVIDER", "groq")
|
|
monkeypatch.setenv("GROQ_API_KEY", "test-key")
|
|
module = load_app_module(monkeypatch)
|
|
|
|
captured = {}
|
|
payload = {"choices": [{"message": {"content": "rewritten CV text"}}]}
|
|
_install_fake_urlopen(monkeypatch, module, payload, captured)
|
|
|
|
assert module._ollama_generate_text("rewrite this") == "rewritten CV text"
|
|
assert captured["url"].endswith("/chat/completions")
|
|
assert captured["headers"].get("authorization") == "Bearer test-key"
|
|
assert captured["body"]["messages"][0]["content"] == "rewrite this"
|
|
|
|
|
|
def test_provider_missing_cloud_key_raises_503(monkeypatch):
|
|
monkeypatch.setenv("AI_PROVIDER", "gemini")
|
|
monkeypatch.delenv("GEMINI_API_KEY", raising=False)
|
|
module = load_app_module(monkeypatch)
|
|
|
|
from fastapi import HTTPException
|
|
|
|
try:
|
|
module._ollama_generate_json("hi")
|
|
except HTTPException as ex:
|
|
assert ex.status_code == 503
|
|
assert "GEMINI_API_KEY" in ex.detail
|
|
else:
|
|
raise AssertionError("expected HTTPException for missing GEMINI_API_KEY")
|
|
|
|
|
|
def test_health_reports_active_provider(monkeypatch):
|
|
monkeypatch.setenv("AI_PROVIDER", "gemini")
|
|
monkeypatch.setenv("GEMINI_API_KEY", "test-key")
|
|
module = load_app_module(monkeypatch)
|
|
client = TestClient(module.app)
|
|
|
|
payload = client.get("/health").json()
|
|
|
|
assert payload["ai_provider"] == "gemini"
|
|
assert payload["ai_provider_configured"] is True
|
|
|
|
|
|
def test_service_token_rejects_calls_without_the_header(monkeypatch):
|
|
module = load_app_module(monkeypatch, service_token="s3cret")
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post("/summarize", json={"text": "Platform engineering role."})
|
|
|
|
assert response.status_code == 401
|
|
assert "token" in response.json()["detail"].lower()
|
|
|
|
|
|
def test_service_token_rejects_a_wrong_header(monkeypatch):
|
|
module = load_app_module(monkeypatch, service_token="s3cret")
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post(
|
|
"/summarize",
|
|
json={"text": "Platform engineering role."},
|
|
headers={"X-Ai-Service-Token": "wrong"},
|
|
)
|
|
|
|
assert response.status_code == 401
|
|
|
|
|
|
def test_service_token_allows_the_correct_header(monkeypatch):
|
|
module = load_app_module(monkeypatch, service_token="s3cret")
|
|
client = TestClient(module.app)
|
|
|
|
response = client.post(
|
|
"/summarize",
|
|
json={"text": "Platform engineering role."},
|
|
headers={"X-Ai-Service-Token": "s3cret"},
|
|
)
|
|
|
|
# 503 = passed the token gate and reached the handler, which is model-disabled here.
|
|
assert response.status_code == 503
|
|
|
|
|
|
def test_health_stays_open_so_probes_and_healthchecks_work(monkeypatch):
|
|
module = load_app_module(monkeypatch, service_token="s3cret")
|
|
client = TestClient(module.app)
|
|
|
|
assert client.get("/health").status_code == 200
|
|
|
|
|
|
def test_endpoints_stay_open_when_no_token_is_configured(monkeypatch):
|
|
module = load_app_module(monkeypatch)
|
|
client = TestClient(module.app)
|
|
|
|
# Keyless local dev: reaches the handler (503 model-disabled), not a 401.
|
|
response = client.post("/summarize", json={"text": "Platform engineering role."})
|
|
|
|
assert response.status_code == 503
|