diff --git a/app/core/config.py b/app/core/config.py index 4ddb271..7ac4cca 100644 --- a/app/core/config.py +++ b/app/core/config.py @@ -23,9 +23,11 @@ _BACKEND_ENV = Path(__file__).resolve().parents[2] / "backend" / ".env" # releases faster than this file can be updated, and rejecting a brand-new gpt-5.x # would be worse than the small risk of admitting one with a different feature set. # -# The gpt-4o family is excluded on purpose: snapshots before 2024-08-06 lack structured -# outputs, and distinguishing them by alias is not reliable. -SUPPORTED_MODEL_PREFIXES: tuple[str, ...] = ("gpt-5", "gpt-4.1", "o3", "o4","gpt-4o-mini") +# gpt-4o-mini is admitted: its only snapshot (2024-07-18) supports structured outputs +# and it is the production model. The wider gpt-4o family stays excluded because +# snapshots before 2024-08-06 lack structured outputs and aliases do not say which +# snapshot you get. It is not a reasoning model, so `reasoning` is omitted for it. +SUPPORTED_MODEL_PREFIXES: tuple[str, ...] = ("gpt-5", "gpt-4.1", "o3", "o4", "gpt-4o-mini") # "-chat-latest" variants track the ChatGPT product surface rather than the API model # line and do not expose reasoning effort. diff --git a/app/models/scoring.py b/app/models/scoring.py index 18e37ef..13fd553 100644 --- a/app/models/scoring.py +++ b/app/models/scoring.py @@ -62,7 +62,9 @@ class ATSScore(StrictModel): def _normalize(cls, value: Any) -> Any: return _normalize_keywords(value) - @field_validator("candidate_name", "job_title", "current_company", "professional_summary", mode="before") + @field_validator( + "candidate_name", "job_title", "current_company", "professional_summary", mode="before" + ) @classmethod def _blank_profile_text_to_none(cls, value: Any) -> Any: if isinstance(value, str): diff --git a/backend/.env.example b/backend/.env.example index 93c9a77..139f719 100644 --- a/backend/.env.example +++ b/backend/.env.example @@ -74,10 +74,13 @@ APIFY_PROFILE_MODE=Full APIFY_TIMEOUT=30 OPENAI_API_KEY= -OPENAI_MODEL=gpt-5.4-mini +# Production model. Not a reasoning model: OPENAI_EFFORT is accepted and ignored. +# Define every OPENAI_* name once; python-dotenv takes the LAST occurrence. +OPENAI_MODEL=gpt-4o-mini-2024-07-18 # Blank omits the parameter, for reasoning models that reject it. OPENAI_TEMPERATURE=0 -OPENAI_MAX_OUTPUT_TOKENS=4096 +# gpt-4o-mini rejects values above 16384 with a 400. +OPENAI_MAX_OUTPUT_TOKENS=4000 OPENAI_TIMEOUT=60 OPENAI_MAX_RETRIES=3 OPENAI_CONNECT_RETRIES=3 diff --git a/backend/agent/prompt.py b/backend/agent/prompt.py index 0484963..6ea28ef 100644 --- a/backend/agent/prompt.py +++ b/backend/agent/prompt.py @@ -33,11 +33,18 @@ Respond with JSON only: def user_prompt(state) -> str: + """The user turn as JSON. + + job_posts comes first on purpose: it is identical for every CV in a sync run, + and OpenAI prompt caching works on an exact token prefix. With the stable + block ahead of the per-candidate subject and resume, every CV after the first + reads the whole job list from cache at the discounted input rate. + """ return json.dumps( { + "job_posts": state.get("job_posts") or [], "subject": state.get("subject") or "", "resume_text": state.get("resume_text") or "", - "job_posts": state.get("job_posts") or [], }, ensure_ascii=False, ) diff --git a/backend/inbox/views.py b/backend/inbox/views.py index e2aea6a..b966a97 100644 --- a/backend/inbox/views.py +++ b/backend/inbox/views.py @@ -483,8 +483,10 @@ class Email: return serialize_inbox_rescan_run(row) async def plan_on_hold_pairs(self,channel,sheet=None): - """Build (candidate, job) pairs that have never been ATS-scored. + """Build (candidate, active job) pairs that have never been ATS-scored. + Only active openings are scored: a score against a closed role is never + shown for shortlisting, and each pair is a paid model call. When professional_summary is present, the summary-vs-job gradient runs before the already-scored pair skip: an obvious mismatch never reaches ATS. Skip a candidate who already has a score against an active job. @@ -496,8 +498,8 @@ class Email: from summary_gate.execute_agent import allow_ats from summary_gate.plugins import GATE_ENABLED - job_ids=[str(jid) for jid in await JobPosts.list_ids(self.session)] - active_ids={str(jid) for jid in await JobPosts.list_ids(self.session,active_only=True)} + job_ids=[str(jid) for jid in await JobPosts.list_ids(self.session,active_only=True)] + active_ids=set(job_ids) jd_by_id={} if GATE_ENABLED: jobs=await JobPosts.get_by_ids(self.session,job_ids,active_only=False) diff --git a/backend/llm_setup.py b/backend/llm_setup.py index 4b94063..e526bfd 100644 --- a/backend/llm_setup.py +++ b/backend/llm_setup.py @@ -31,7 +31,9 @@ OPENAI_BASE_URL = os.getenv("OPENAI_BASE_URL") or None OPENAI_ORGANIZATION = os.getenv("OPENAI_ORGANIZATION") or None OPENAI_PROJECT = os.getenv("OPENAI_PROJECT") or None OPENAI_MODEL = os.getenv("OPENAI_MODEL", "gpt-5.4-mini") -OPENAI_MAX_OUTPUT_TOKENS = int(os.getenv("OPENAI_MAX_OUTPUT_TOKENS") or 32768) +# Default kept under gpt-4o-mini's 16384 completion ceiling: a larger value is a 400 +# on every call, not a bigger budget. +OPENAI_MAX_OUTPUT_TOKENS = int(os.getenv("OPENAI_MAX_OUTPUT_TOKENS") or 4096) OPENAI_TIMEOUT = float(os.getenv("OPENAI_TIMEOUT") or 60) OPENAI_MAX_RETRIES = int(os.getenv("OPENAI_MAX_RETRIES") or 3) OPENAI_CONNECT_RETRIES = int(os.getenv("OPENAI_CONNECT_RETRIES") or 3) @@ -80,6 +82,7 @@ async def llm_call(system, user, *, model=None, temperature=None, json_mode=Fals kwargs["response_format"] = {"type": "json_object"} response = await get_client().chat.completions.create(**kwargs) + _log_usage(response, kwargs["model"]) content = (response.choices[0].message.content or "").strip() if not json_mode: return content @@ -89,6 +92,22 @@ async def llm_call(system, user, *, model=None, temperature=None, json_mode=Fals raise RuntimeError(f"model did not return valid JSON: {content[:200]}") from exc +def _log_usage(response, model) -> None: + """Per-call token and cache visibility. Never logs prompt or reply text.""" + usage = getattr(response, "usage", None) + if usage is None: + return + prompt_details = getattr(usage, "prompt_tokens_details", None) + logger.info( + "llm usage: model=%s request_id=%s prompt_tokens=%s completion_tokens=%s cached_tokens=%s", + model, + getattr(response, "_request_id", None), + getattr(usage, "prompt_tokens", None), + getattr(usage, "completion_tokens", None), + getattr(prompt_details, "cached_tokens", None), + ) + + async def check_connection(retries=None, delay=1.0): """Confirm the key works, retrying with a capped backoff.""" attempts = OPENAI_CONNECT_RETRIES if retries is None else retries