diff --git a/BUGFIX_LOG.md b/BUGFIX_LOG.md index c3d265b..c531b55 100644 --- a/BUGFIX_LOG.md +++ b/BUGFIX_LOG.md @@ -9,6 +9,20 @@ --- ## 2026-08-11 +### BUG · 默认职位列表混入大量当前 CV 未评分卡片 + +**错误** +修正 0 分显示后,页面仍出现大量灰色长横线。最新 200 条缓存中有 169 条没有当前 CV 的 `job_matches`,其中包含历史搜索结果和已经被相关性闸门拒绝的明显无关职位。 + +**原因** +`/api/jobs` 先读取全局最近 200 条原始缓存,再把没有 `match_score` 的职位默认视为可见。相关性闸门的拒绝只写运行审计事件,不按 CV 持久化,导致同一缓存职位在后续搜索中反复进入大模型闸门。 + +**解决方案** +默认 Web 列表改为直接查询最新 CV 下已有且未被 `skip` 的真实匹配;新增独立的 `job_relevance_rejections` 缓存,按职位、CV、JD 内容和 prompt 版本复用闸门拒绝。换 CV、JD 更新或 prompt 升级仍会重新判断,且不会为拒绝结果伪造 0 分评分。 + +**结果** +新增回归测试覆盖当前 CV 列表、无 CV 兼容、全新/缓存 JD 的拒绝落库、拒绝缓存作用域、重复搜索跳过和缓存清理;完整 pytest(255 tests)、Ruff、Python 编译和前端语法检查均通过。 + ### BUG · 未评分职位卡被错误显示为 0 分 **错误** diff --git a/CHANGELOG.md b/CHANGELOG.md index 3a00813..366d4f5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,9 @@ ### Bug Fixes +- **Current-CV job lists no longer mix in unassessed historical cards** (`server.py` / `cache.py` / `search_prefilter.py` / `search_assessment_stage.py`) + The default Web list now reads visible `job_matches` for the latest CV across the cache instead of taking the newest 200 raw cache rows and treating missing matches as relevant. JD relevance-gate rejections are cached separately by job, CV, JD content, language, and prompt version, so the same rejected cached job does not consume another model call on the next search. A different CV, changed JD, or updated gate prompt still triggers reassessment; no synthetic zero-score `MatchScore` is created. + - **Unassessed job cards no longer appear as zero-score matches** (`templates/index.html`) Job cards whose current-CV score is `null` now show a gray em dash instead of being coerced to `0` by JavaScript. Genuine numeric zero scores still display as zero, preserving the distinction between "not assessed" and "assessed with no match". diff --git a/README.md b/README.md index 32dff85..946bacb 100644 --- a/README.md +++ b/README.md @@ -132,6 +132,7 @@ DEFAULT_MODEL=gemini-3.5-flash-lite - **Dynamic title seniority gate**: uses CV eligible/stretch/blocked levels to remove obvious title-level mismatches before JD matching - **Experience-gap hard reject**: prefilter records `skip_exp`, and `jd_assessment` directly rejects roles whose explicit years-required exceeds the candidate's experience by more than 3 years - **Explainable matching**: job details include score breakdown, risks, skill matches, and recommendation tiers +- **Current-CV result list**: the default Web list shows only real, non-skipped matches for the latest CV; CV-scoped relevance rejections are reused until the CV, JD content, language, or gate prompt changes - **Risk-only relocation / office attendance handling**: same-country relocation and office-attendance requirements are recorded in `risks / risk_penalty` but are not treated as `location_score` penalties - **Artifact hub**: generate and reuse Interview Prep, Cover Letter, and CV Optimization from the job detail panel - **Log panel**: level filtering, keyword highlight, auto-refresh diff --git a/docs/RUNBOOK.md b/docs/RUNBOOK.md index 1718e0e..3681c2e 100644 --- a/docs/RUNBOOK.md +++ b/docs/RUNBOOK.md @@ -294,7 +294,7 @@ jobradar_email_sync_runs_total{status="failed",reason="auth"} | Indeed 抓取结果为 0 | JobSpy 被 Indeed 限流(常见,非故障) | 等几小时重试;查日志中 JobSpy WARNING | | Adzuna 返回 429 | 速率限制 | 调大 `_MIN_INTERVAL`(当前 1.2s);核对 `.env` 中 `ADZUNA_APP_ID` / `ADZUNA_APP_KEY` | | LLM 评估全部拒绝 | `cv_summary` / `cv_skills` 提取失败 | 检查 CV 解析结果;`uv run jobradar assess` 补跑评估 | -| 职位无模型评分 | 当前 `cv_hash` 下尚无 `job_matches` 记录(换 CV 后常见) | `uv run jobradar assess` 为当前 CV 补算 | +| 职位无模型评分 | 当前 `cv_hash` 下尚无 `job_matches` 记录(换 CV 后常见);默认 Web 列表不会展示这类历史卡片 | 仅在需要主动补算历史 JD 时运行 `uv run jobradar assess` | ### 安全注意 diff --git a/jobradar/assessment.py b/jobradar/assessment.py index 2819b4b..a6313b5 100644 --- a/jobradar/assessment.py +++ b/jobradar/assessment.py @@ -17,6 +17,7 @@ BATCH_SIZE = 8 # 每批 JD 数量,兼顾 context 长度与 token 节省 TITLE_RELEVANCE_PROMPT_VERSION = "title_relevance_v4" +JD_ASSESSMENT_PROMPT_VERSION = "jd_assessment_v1" _LANGUAGE_NAMES = {"zh": "中文", "en": "English", "es": "Español"} _TITLE_KEYWORD_STOPWORDS = { @@ -27,6 +28,10 @@ } +def jd_assessment_prompt_version(language: str = "zh") -> str: + return f"{JD_ASSESSMENT_PROMPT_VERSION}:{language}" + + def gate_worker_count(provider: str) -> int: """Bound independent gate calls without overloading local providers.""" return 1 if provider in {"ollama", "local"} else 2 diff --git a/jobradar/cache.py b/jobradar/cache.py index db901a3..de682d2 100644 --- a/jobradar/cache.py +++ b/jobradar/cache.py @@ -165,6 +165,19 @@ PRIMARY KEY (job_id, cv_hash) ); +CREATE TABLE IF NOT EXISTS job_relevance_rejections ( + job_id TEXT NOT NULL, + cv_hash TEXT NOT NULL, + description_hash TEXT NOT NULL, + reason TEXT NOT NULL DEFAULT '', + score INTEGER NOT NULL DEFAULT 0, + model_name TEXT NOT NULL DEFAULT '', + prompt_version TEXT NOT NULL DEFAULT '', + created_at TEXT NOT NULL, + updated_at TEXT NOT NULL, + PRIMARY KEY (job_id, cv_hash) +); + CREATE TABLE IF NOT EXISTS interview_preps ( job_id TEXT NOT NULL, cv_hash TEXT NOT NULL, @@ -465,12 +478,35 @@ def merge_job_raw_source(dedup_key: str, source_entry: dict) -> None: ) -def get_recent_jobs(limit: int = 50, language: str = "zh") -> list[JobResult]: - """按抓取时间倒序返回最近 limit 条未过期职位。""" +def get_recent_jobs( + limit: int = 50, + language: str = "zh", + *, + require_match: bool = False, +) -> list[JobResult]: + """按抓取时间倒序返回最近职位,可限定为当前 CV 下已有可见匹配。""" with _conn() as con: - rows = con.execute( - "SELECT * FROM job_cache ORDER BY fetched_at DESC LIMIT ?", (limit,) - ).fetchall() + if require_match: + cv_hash = get_latest_cv_hash() + if not cv_hash: + return [] + rows = con.execute( + """ + SELECT job_cache.* + FROM job_cache + JOIN job_matches + ON job_matches.job_id = job_cache.dedup_key + AND job_matches.cv_hash = ? + AND job_matches.recommendation != 'skip' + ORDER BY job_cache.fetched_at DESC + LIMIT ? + """, + (cv_hash, limit), + ).fetchall() + else: + rows = con.execute( + "SELECT * FROM job_cache ORDER BY fetched_at DESC LIMIT ?", (limit,) + ).fetchall() jobs = [_row_to_job(r) for r in rows] from jobradar.jd_profile import jd_profile_prompt_version @@ -702,6 +738,75 @@ def get_job_match(job_id: str, cv_hash: str, description: str = "", prompt_versi return MatchScore.model_validate_json(row["score_json"]) +def get_job_relevance_rejection( + job_id: str, + cv_hash: str, + description: str = "", + prompt_version: str = "", +) -> dict | None: + with _conn() as con: + row = con.execute( + """ + SELECT reason, score, model_name, description_hash, prompt_version + FROM job_relevance_rejections + WHERE job_id = ? AND cv_hash = ? + """, + (job_id, cv_hash), + ).fetchone() + if row is None: + return None + if description and row["description_hash"] != _description_hash(description): + return None + if prompt_version and row["prompt_version"] != prompt_version: + return None + return { + "reason": row["reason"], + "score": row["score"], + "model_name": row["model_name"], + "prompt_version": row["prompt_version"], + } + + +def save_job_relevance_rejection( + *, + job_id: str, + cv_hash: str, + description: str, + reason: str, + score: int, + model_name: str = "", + prompt_version: str = "", +) -> None: + now = datetime.utcnow().isoformat() + with _conn() as con: + con.execute( + """ + INSERT INTO job_relevance_rejections + (job_id, cv_hash, description_hash, reason, score, model_name, + prompt_version, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(job_id, cv_hash) DO UPDATE SET + description_hash = excluded.description_hash, + reason = excluded.reason, + score = excluded.score, + model_name = excluded.model_name, + prompt_version = excluded.prompt_version, + updated_at = excluded.updated_at + """, + ( + job_id, + cv_hash, + _description_hash(description), + reason, + score, + model_name, + prompt_version, + now, + now, + ), + ) + + def _attach_latest_match(job: JobResult, language: str = "zh") -> None: latest_cv_hash = get_latest_cv_hash() if not latest_cv_hash: @@ -1159,6 +1264,7 @@ def clear_all() -> None: con.execute("DELETE FROM jd_profiles") con.execute("DELETE FROM job_summaries") con.execute("DELETE FROM job_matches") + con.execute("DELETE FROM job_relevance_rejections") con.execute("DELETE FROM interview_preps") con.execute("DELETE FROM cover_letters") con.execute("DELETE FROM cv_optimizations") @@ -1189,6 +1295,10 @@ def delete_jobs(dedup_keys: list[str]) -> int: f"DELETE FROM job_matches WHERE job_id IN ({placeholders})", dedup_keys, ) + con.execute( + f"DELETE FROM job_relevance_rejections WHERE job_id IN ({placeholders})", + dedup_keys, + ) con.execute( f"DELETE FROM interview_preps WHERE job_id IN ({placeholders})", dedup_keys, @@ -1619,6 +1729,10 @@ def clean_expired() -> int: f"DELETE FROM job_summaries WHERE job_id IN ({placeholders})", expired_job_ids, ) + con.execute( + f"DELETE FROM job_relevance_rejections WHERE job_id IN ({placeholders})", + expired_job_ids, + ) # 删除有明确截止日期且已过期的 JD r1 = con.execute( "DELETE FROM job_cache WHERE expires_at IS NOT NULL AND expires_at < ?", diff --git a/jobradar/search_assessment_stage.py b/jobradar/search_assessment_stage.py index 2ef7aa5..32088f5 100644 --- a/jobradar/search_assessment_stage.py +++ b/jobradar/search_assessment_stage.py @@ -7,7 +7,11 @@ from typing import Callable from jobradar import cache -from jobradar.assessment import JDAssessment, batch_assess_jds +from jobradar.assessment import ( + JDAssessment, + batch_assess_jds, + jd_assessment_prompt_version, +) from jobradar.jd_profile import jd_profile_prompt_version from jobradar.job_evaluation import evaluate_job_once, job_evaluation_prompt_version from jobradar.logger import get_logger @@ -322,6 +326,15 @@ def _is_visible_job(job_obj) -> bool: details={"score": assessment.score, "cached": True}, ) llm_rejected += 1 + cache.save_job_relevance_rejection( + job_id=cached_job.dedup_key, + cv_hash=cv_hash, + description=cached_job.description_snippet, + reason=assessment.reason, + score=assessment.score, + model_name=f"{llm.provider}/{llm.model}", + prompt_version=jd_assessment_prompt_version(language), + ) write_cache( { "title": cached_job.title, @@ -443,6 +456,16 @@ def _is_visible_job(job_obj) -> bool: "raw_sources": raw_sources, } ) + if not assessment.relevant and cv_hash and llm is not None: + cache.save_job_relevance_rejection( + job_id=key, + cv_hash=cv_hash, + description=content, + reason=assessment.reason, + score=assessment.score, + model_name=f"{llm.provider}/{llm.model}", + prompt_version=jd_assessment_prompt_version(language), + ) if assessment.relevant: summary_job = JobResult( title=title, diff --git a/jobradar/search_prefilter.py b/jobradar/search_prefilter.py index fa24b4d..1cc892a 100644 --- a/jobradar/search_prefilter.py +++ b/jobradar/search_prefilter.py @@ -5,6 +5,7 @@ from typing import Callable from jobradar import cache +from jobradar.assessment import jd_assessment_prompt_version from jobradar.filters import infer_title_seniority, is_title_seniority_ok from jobradar.logger import get_logger from jobradar.matching import match_prompt_version @@ -95,7 +96,13 @@ def classify_cache_hit(cached_job: JobResult, cv_hash: str, language: str = "zh" prompt_version=match_prompt_version(language), ) if match is None: - return "reassess" + rejection = cache.get_job_relevance_rejection( + cached_job.dedup_key, + cv_hash, + cached_job.description_snippet, + prompt_version=jd_assessment_prompt_version(language), + ) + return "skip" if rejection is not None else "reassess" return "skip" if match.recommendation == "skip" else "reuse" diff --git a/jobradar/server.py b/jobradar/server.py index be01b31..6a6f57f 100644 --- a/jobradar/server.py +++ b/jobradar/server.py @@ -603,7 +603,10 @@ def delete_application(application_id: int) -> dict: @app.get("/api/jobs") def get_jobs(limit: int = 200, language: str = "zh") -> list[dict]: - jobs = cache.get_recent_jobs(limit, language=language) + require_match = bool(cache.get_latest_cv_hash()) + jobs = cache.get_recent_jobs(limit, language=language, require_match=require_match) + if require_match: + jobs = [j for j in jobs if j.match_score is not None] jobs = [j for j in jobs if j.is_effectively_relevant] jobs.sort(key=lambda j: (j.effective_score if j.effective_score is not None else -1), reverse=True) return [_job_to_dict(j) for j in jobs] diff --git a/tests/test_application_tracking.py b/tests/test_application_tracking.py index e9a1b0d..e41547f 100644 --- a/tests/test_application_tracking.py +++ b/tests/test_application_tracking.py @@ -226,6 +226,40 @@ def test_jobs_route_remains_bound_to_get_jobs(): assert route.endpoint.__name__ == "get_jobs" +def test_jobs_api_requests_only_current_cv_matches(monkeypatch): + from jobradar import server + + calls = {} + + def fake_recent_jobs(limit, language="zh", require_match=False): + calls["limit"] = limit + calls["language"] = language + calls["require_match"] = require_match + return [] + + monkeypatch.setattr(server.cache, "get_latest_cv_hash", lambda: "current-cv") + monkeypatch.setattr(server.cache, "get_recent_jobs", fake_recent_jobs) + + assert server.get_jobs() == [] + assert calls == {"limit": 200, "language": "zh", "require_match": True} + + +def test_jobs_api_preserves_cache_visibility_without_a_cv(monkeypatch): + from jobradar import server + + calls = {} + + def fake_recent_jobs(limit, language="zh", require_match=False): + calls["require_match"] = require_match + return [] + + monkeypatch.setattr(server.cache, "get_latest_cv_hash", lambda: "") + monkeypatch.setattr(server.cache, "get_recent_jobs", fake_recent_jobs) + + assert server.get_jobs() == [] + assert calls["require_match"] is False + + def test_application_nav_button_is_not_nested_in_config_button(): from pathlib import Path diff --git a/tests/test_cache.py b/tests/test_cache.py index 87ea804..7c2e9a9 100644 --- a/tests/test_cache.py +++ b/tests/test_cache.py @@ -489,7 +489,15 @@ def test_get_failed_urls_batch(self, temp_db): class TestCacheManagement: def test_clear_all(self, temp_db): - temp_db.save_job(make_job()) + job = make_job() + temp_db.save_job(job) + temp_db.save_job_relevance_rejection( + job_id=job.dedup_key, + cv_hash="cv123", + description=job.description_snippet, + reason="Unrelated", + score=1, + ) temp_db.record_failed_url("http://x.com", "reason") temp_db.save_search_candidates( "run-clear", @@ -498,6 +506,7 @@ def test_clear_all(self, temp_db): temp_db.clear_all() assert temp_db.get_job("google|software engineer") is None + assert temp_db.get_job_relevance_rejection(job.dedup_key, "cv123") is None assert not temp_db.is_failed_url("http://x.com") assert temp_db.get_search_candidates("run-clear") == [] @@ -542,10 +551,18 @@ def test_delete_jobs_removes_match_and_interview_prep(self, temp_db): prep = InterviewPrep(job_id=job.dedup_key, cv_hash="cv123", fit_summary="Strong fit") temp_db.save_job_match(match, "desc") temp_db.save_interview_prep(prep, "desc") + temp_db.save_job_relevance_rejection( + job_id=job.dedup_key, + cv_hash="cv123", + description="desc", + reason="Unrelated", + score=1, + ) temp_db.delete_jobs([job.dedup_key]) assert temp_db.get_job_match(job.dedup_key, "cv123", "desc") is None assert temp_db.get_interview_prep(job.dedup_key, "cv123", "desc") is None + assert temp_db.get_job_relevance_rejection(job.dedup_key, "cv123", "desc") is None def test_delete_jobs_removes_cover_letter(self, temp_db): job = make_job(description_snippet="desc") @@ -829,3 +846,83 @@ def test_no_conditions_is_a_noop(self, temp_db): assert result["total"] == 0 and result["deleted"] == 0 assert len(self._remaining(temp_db)) == 1 + + +def test_recent_jobs_can_require_visible_match_for_latest_cv(temp_db): + profile = CVProfile(summary="Python engineer", skills=["Python"], seniority="junior") + temp_db.save_cv_profile("current-cv", profile) + + visible = make_job(company="Visible", url="https://example.com/visible") + skipped = make_job(company="Skipped", url="https://example.com/skipped") + unassessed = make_job(company="Unassessed", url="https://example.com/unassessed") + old_cv_only = make_job(company="Old CV", url="https://example.com/old-cv") + for job in (visible, skipped, unassessed, old_cv_only): + temp_db.save_job(job) + + def save_match(job, cv_hash, recommendation, score): + temp_db.save_job_match( + MatchScore( + job_id=job.dedup_key, + cv_hash=cv_hash, + overall_score=score, + title_score=score, + seniority_score=score, + must_have_score=score, + nice_to_have_score=score, + domain_score=score, + location_score=100, + language_score=score, + risk_penalty=0, + recommendation=recommendation, + ), + job.description_snippet, + ) + + save_match(visible, "current-cv", "apply", 80) + save_match(skipped, "current-cv", "skip", 0) + save_match(old_cv_only, "old-cv", "strong_apply", 90) + + jobs = temp_db.get_recent_jobs(limit=10, require_match=True) + + assert [job.dedup_key for job in jobs] == [visible.dedup_key] + + +def test_relevance_rejection_is_scoped_by_cv_description_and_prompt(temp_db): + temp_db.save_job_relevance_rejection( + job_id="acme|sales agent", + cv_hash="current-cv", + description="Handle customer calls.", + reason="Unrelated customer-service role", + score=1, + model_name="gemini/test", + prompt_version="jd_assessment_v1:zh", + ) + + rejection = temp_db.get_job_relevance_rejection( + "acme|sales agent", + "current-cv", + "Handle customer calls.", + prompt_version="jd_assessment_v1:zh", + ) + + assert rejection is not None + assert rejection["reason"] == "Unrelated customer-service role" + assert rejection["score"] == 1 + assert temp_db.get_job_relevance_rejection( + "acme|sales agent", + "different-cv", + "Handle customer calls.", + prompt_version="jd_assessment_v1:zh", + ) is None + assert temp_db.get_job_relevance_rejection( + "acme|sales agent", + "current-cv", + "Updated job description.", + prompt_version="jd_assessment_v1:zh", + ) is None + assert temp_db.get_job_relevance_rejection( + "acme|sales agent", + "current-cv", + "Handle customer calls.", + prompt_version="jd_assessment_v2:zh", + ) is None diff --git a/tests/test_new_features.py b/tests/test_new_features.py index 9ca4352..c5301b6 100644 --- a/tests/test_new_features.py +++ b/tests/test_new_features.py @@ -970,3 +970,72 @@ def fake_open( assert Path(report_path) == REPORTS_DIR / "pipeline_stats_latest.json" assert created_dirs == [REPORTS_DIR] + + +@pytest.mark.parametrize("cached", [True, False]) +def test_gate_rejection_is_persisted_for_current_cv( + db, monkeypatch: pytest.MonkeyPatch, cached: bool +): + import jobradar.search_assessment_stage as stage + from jobradar.assessment import jd_assessment_prompt_version + from jobradar.search_prefilter import PrefilterResult + + cached_job = JobResult( + title="Customer Service Agent", + company="Acme", + url="https://example.com/customer-service", + description_snippet="Answer customer calls.", + ) + if cached: + db.save_job(cached_job) + monkeypatch.setattr( + stage, + "batch_assess_jds", + lambda jobs, profile, llm, language="zh": [ + JDAssessment( + relevant=False, + reason="Unrelated role", + score=1, + strengths=[], + weaknesses=[], + matched_keywords=[], + ) + ], + ) + + pending = PrefilterResult( + patch_pending=[(cached_job, cached_job.description_snippet)] if cached else [], + pending=[] if cached else [ + ( + { + "title": cached_job.title, + "company": cached_job.company, + "url": cached_job.url, + "source": "indeed.ie", + "description_snippet": cached_job.description_snippet, + }, + cached_job.description_snippet, + None, + ) + ], + ) + + stage.flush_assessments( + pending, + job_all_sources={}, + profile=_make_profile(), + llm=_make_llm(), + cv_hash="current-cv", + cb=lambda msg: None, + on_job=None, + language="zh", + ) + + rejection = db.get_job_relevance_rejection( + cached_job.dedup_key, + "current-cv", + cached_job.description_snippet, + prompt_version=jd_assessment_prompt_version("zh"), + ) + assert rejection is not None + assert rejection["reason"] == "Unrelated role" diff --git a/tests/test_search_prefilter.py b/tests/test_search_prefilter.py index dd71d5c..6a13b3b 100644 --- a/tests/test_search_prefilter.py +++ b/tests/test_search_prefilter.py @@ -133,6 +133,24 @@ def test_stale_prompt_version_triggers_reassessment(self, temp_db): assert classify_cache_hit(job, CURRENT_CV) == "reassess" + def test_current_cv_gate_rejection_is_skipped(self, temp_db): + from jobradar.assessment import jd_assessment_prompt_version + from jobradar.search_prefilter import classify_cache_hit + + job = make_job() + temp_db.save_job(job) + temp_db.save_job_relevance_rejection( + job_id=job.dedup_key, + cv_hash=CURRENT_CV, + description=job.description_snippet, + reason="Unrelated role", + score=1, + model_name="gemini/test", + prompt_version=jd_assessment_prompt_version("zh"), + ) + + assert classify_cache_hit(job, CURRENT_CV) == "skip" + def test_without_cv_hash_everything_is_reused(self, temp_db): """无 CV 时评分无从谈起,一律复用缓存内容。