Merge pull request 'Implementing_limitListing' (#24) from Implementing_limitListing into main
Deploy to S3 / deploy (push) Successful in 35s
Details
Deploy to S3 / deploy (push) Successful in 35s
Details
Reviewed-on: #24pull/25/head
commit
2160059400
|
|
@ -31,6 +31,11 @@ frontend/
|
|||
# Candidate CVs live on the bind mount, not inside an image.
|
||||
backend/inbox/decoded_attachments/
|
||||
|
||||
# Alembic revision scripts stay out of images (gitignored; never ship to prod).
|
||||
# Schema drift is applied filelessly at API boot when DB_AUTOGENERATE=true.
|
||||
backend/migrations/versions/*.py
|
||||
!backend/migrations/versions/.gitkeep
|
||||
|
||||
docs/
|
||||
tests/
|
||||
scripts/
|
||||
|
|
|
|||
36
.env.example
36
.env.example
|
|
@ -1,36 +0,0 @@
|
|||
# Copy to .env and fill in. Never commit .env.
|
||||
|
||||
OPENAI_API_KEY=
|
||||
|
||||
# Must be a structured-outputs model family: gpt-5*, gpt-4.1*, o3*, o4*.
|
||||
# "-chat-latest" variants are rejected -- they track the ChatGPT product surface and
|
||||
# do not expose reasoning effort.
|
||||
# Note gpt-4.1 is allowed but is not a reasoning model, so OPENAI_EFFORT is ignored
|
||||
# for it (the adapter omits the parameter rather than sending a 400).
|
||||
OPENAI_MODEL=gpt-5.4-mini
|
||||
|
||||
# Covers reasoning tokens AND the visible response on a reasoning model. Too low and
|
||||
# the JSON truncates mid-object, failing the candidate with MODEL_RESPONSE_INVALID.
|
||||
# Enforced floor is 2048. Do not lower this to save cost -- lower OPENAI_EFFORT.
|
||||
OPENAI_MAX_OUTPUT_TOKENS=4000
|
||||
|
||||
# none | minimal | low | medium | high | xhigh
|
||||
# Per-model support varies; the API rejects a level the model does not implement.
|
||||
OPENAI_EFFORT=low
|
||||
|
||||
OPENAI_MAX_RETRIES=3
|
||||
OPENAI_TIMEOUT_SECONDS=120
|
||||
|
||||
# OpenAI prompt caching is automatic and cannot be turned off. This only controls
|
||||
# whether a prompt_cache_key routing hint is sent to raise the cache hit rate.
|
||||
OPENAI_ENABLE_PROMPT_CACHE=true
|
||||
|
||||
SCORING_CONCURRENCY=5
|
||||
MAX_RESUMES_PER_REQUEST=50
|
||||
MAX_PDF_SIZE_MB=10
|
||||
MAX_JD_CHARS=30000
|
||||
MAX_RESUME_CHARS=60000
|
||||
|
||||
# text | json
|
||||
LOG_FORMAT=json
|
||||
LOG_LEVEL=INFO
|
||||
|
|
@ -59,11 +59,7 @@ frontend/dist/
|
|||
# also swallowed any module with two underscores (e.g. test_talent_plugins.py).
|
||||
backend/migrations/versions/**_**_**.py
|
||||
Utopia-ai-hr-ats-portal 1.pem
|
||||
db_setup.py
|
||||
tests/
|
||||
tests/**
|
||||
*/tests/**
|
||||
*/tests/**/*
|
||||
# Paper form source documents (Annexure A/E/J) — reference material, not code.
|
||||
# Root-anchored: backend/candidate_forms/ is the forms domain package and IS tracked.
|
||||
/candidate_forms/
|
||||
|
||||
# Local-only Compose overrides (never deployed)
|
||||
docker.local.env
|
||||
tests/**
|
||||
|
|
@ -0,0 +1,145 @@
|
|||
# Docker
|
||||
|
||||
## How the browser reaches the API
|
||||
|
||||
The SPA is on **http://127.0.0.1:5173** (nginx → `backend-api` on the Compose
|
||||
network). The API is also published on **http://127.0.0.1:8000** for host tools
|
||||
and `npm run dev` (`VITE_API_BASE=http://127.0.0.1:8000`).
|
||||
|
||||
```env
|
||||
FRONTEND_PORT=5173
|
||||
BACKEND_PORT=8000
|
||||
FRONTEND_URL=http://127.0.0.1:5173
|
||||
```
|
||||
|
||||
Sole env file: **`backend/.env`** (no repo-root `.env`). Always pass it for
|
||||
Compose variable substitution:
|
||||
|
||||
```bash
|
||||
docker compose --env-file ./backend/.env up -d --build
|
||||
```
|
||||
|
||||
## Local (host Postgres)
|
||||
|
||||
In `backend/.env`:
|
||||
|
||||
```env
|
||||
PROD_ENV=false
|
||||
DB_USERNAME=...
|
||||
DB_PASSWORD=...
|
||||
DB_HOST=localhost
|
||||
DB_PORT=5432
|
||||
DB_NAME=hrms
|
||||
DB_SSLMODE=
|
||||
FRONTEND_PORT=8080
|
||||
```
|
||||
|
||||
Containers set `IN_DOCKER=1`. With `PROD_ENV=false`, `db_setup` rewrites
|
||||
`localhost` / `127.0.0.1` → `host.docker.internal` for the connection URL only
|
||||
(SSL off unless `DB_SSLMODE` is set). Host Postgres must accept Docker-bridge
|
||||
clients (`listen_addresses`, `pg_hba`).
|
||||
|
||||
```bash
|
||||
cp backend/.env.example backend/.env # set JWT, OpenAI, DB_*, PROD_ENV=false
|
||||
docker compose --env-file ./backend/.env up -d --build
|
||||
```
|
||||
|
||||
| Service | Host access |
|
||||
|---|---|
|
||||
| `frontend` | `${FRONTEND_PORT:-80}` (all interfaces) |
|
||||
| `backend-api` | Compose network only (`backend-api:8000`); nginx proxies |
|
||||
| `ats-engine` | Compose network only (`ats-engine:8100`) |
|
||||
| `redis` | Compose network only (`redis:6379`) |
|
||||
| `postgres` | not started (optional `--profile postgres`) |
|
||||
|
||||
Optional loopback publishes for host tools (curl / redis-cli / Postman):
|
||||
|
||||
```bash
|
||||
docker compose --env-file ./backend/.env -f docker-compose.yml -f docker-compose.host-ports.yml up -d
|
||||
```
|
||||
|
||||
If bind fails on Windows because Cursor/VS Code still holds `:80` / `:8100` /
|
||||
`:6379` after a previous run, clear **Ports** in the IDE or set free values in
|
||||
`backend/.env` (`FRONTEND_PORT`, and with the overlay `ATS_PORT` / `REDIS_PORT` /
|
||||
`BACKEND_PORT`).
|
||||
|
||||
Optional live-reload / bind mounts:
|
||||
|
||||
```bash
|
||||
docker compose --env-file ./backend/.env -f docker-compose.yml -f docker-compose.dev.yml up -d --build
|
||||
```
|
||||
|
||||
Optional Compose Postgres (empty volume — not host data):
|
||||
|
||||
```bash
|
||||
docker compose --env-file ./backend/.env --profile postgres up -d postgres
|
||||
# set DB_HOST=postgres in backend/.env, then recreate backend services
|
||||
```
|
||||
|
||||
## Production (RDS)
|
||||
|
||||
In `backend/.env`, set `PROD_ENV=true` and point plain `DB_*` at RDS (no
|
||||
prefixed credential sets). Blank `DB_SSLMODE` → SSL `require`. Host is never
|
||||
rewritten.
|
||||
|
||||
```bash
|
||||
cp backend/.env.example backend/.env
|
||||
# Edit backend/.env: PROD_ENV=true, DB_* = RDS, JWT_SECRET_KEY, FRONTEND_PORT=80, …
|
||||
docker compose --env-file ./backend/.env up -d --build
|
||||
docker compose --env-file ./backend/.env ps
|
||||
```
|
||||
|
||||
Browser → `http://<host>/` → nginx (same-origin) → `backend-api:8000`.
|
||||
CV files live in the `attachments-data` named volume (shared by API + workers).
|
||||
|
||||
### Schema / migrations (automatic)
|
||||
|
||||
On every `backend-api` start:
|
||||
|
||||
1. Fresh empty Postgres → create all tables from models and stamp a marker.
|
||||
2. Otherwise → `alembic upgrade head` if any revision files exist in the image
|
||||
(they normally do not — versions stay gitignored and are excluded from builds).
|
||||
3. If `DB_AUTOGENERATE=true` → detect ORM drift and apply DDL **in-memory**.
|
||||
4. Apply any pending `backend/migrations/manual/*.sql` (seed/RBAC batches only).
|
||||
|
||||
Toggle in `backend/.env`: `DB_AUTO_MIGRATE` / `DB_AUTOGENERATE` (default `true`).
|
||||
|
||||
### Verify
|
||||
|
||||
```bash
|
||||
docker compose --env-file ./backend/.env config
|
||||
curl -sf http://127.0.0.1:${FRONTEND_PORT:-8080}/health
|
||||
curl -sf -o /dev/null -w "%{http_code}\n" http://127.0.0.1:${FRONTEND_PORT:-8080}/
|
||||
docker compose --env-file ./backend/.env logs -f backend-api
|
||||
```
|
||||
|
||||
### Secrets
|
||||
|
||||
- Never bake `.env` into images (`.dockerignore` already excludes them).
|
||||
- Require a strong `DB_PASSWORD` and `JWT_SECRET_KEY` before any real deploy.
|
||||
- Only `backend/.env` holds app + Compose substitution values.
|
||||
- Do not put `DB_HOST` under Compose `environment:` (empty override blanks RDS).
|
||||
|
||||
### TLS
|
||||
|
||||
This stack serves HTTP on the frontend port. Terminate TLS at a reverse proxy or
|
||||
cloud load balancer in front of that port.
|
||||
|
||||
## Data migration
|
||||
|
||||
The optional Compose Postgres volume starts empty. To move an existing host database:
|
||||
|
||||
```bash
|
||||
pg_dump -Fc hrms > hrms.dump
|
||||
pg_restore -h 127.0.0.1 -p 5433 -U postgres -d hrms --clean --if-exists hrms.dump
|
||||
```
|
||||
|
||||
## Useful commands
|
||||
|
||||
```bash
|
||||
docker compose --env-file ./backend/.env logs -f backend-api
|
||||
docker compose --env-file ./backend/.env logs -f taskiq-worker
|
||||
docker compose --env-file ./backend/.env restart backend-api
|
||||
docker compose --env-file ./backend/.env down # keep volumes
|
||||
docker compose --env-file ./backend/.env down -v # wipe volumes
|
||||
```
|
||||
102
README.md
102
README.md
|
|
@ -45,29 +45,19 @@ job and persists an evidence-based score.
|
|||
**Prerequisites:** Python 3.11+, Node 18+, PostgreSQL, an OpenAI API key.
|
||||
Optional: Redis + Docker (only for background inbox sync / taskiq workers).
|
||||
|
||||
### 1. Environment files
|
||||
### 1. Environment file
|
||||
|
||||
Root `.env` (engine + scoring settings — see [.env.example](.env.example)):
|
||||
|
||||
```
|
||||
OPENAI_API_KEY=sk-...
|
||||
OPENAI_MODEL=gpt-5.4-mini
|
||||
OPENAI_MAX_OUTPUT_TOKENS=4000
|
||||
OPENAI_EFFORT=low
|
||||
SCORING_CONCURRENCY=5
|
||||
MAX_RESUMES_PER_REQUEST=50
|
||||
MAX_PDF_SIZE_MB=10
|
||||
```
|
||||
|
||||
`backend/.env` (everything in `backend/.env.example`; the must-haves):
|
||||
Sole file: `backend/.env` (see [backend/.env.example](backend/.env.example)):
|
||||
|
||||
```
|
||||
PROD_ENV=false
|
||||
DB_USERNAME=... DB_PASSWORD=... DB_HOST=localhost DB_PORT=5432 DB_NAME=hrms
|
||||
JWT_SECRET_KEY=...
|
||||
OPENAI_API_KEY=sk-... # shared names with the root .env
|
||||
OPENAI_API_KEY=sk-...
|
||||
FRONTEND_PORT=8080
|
||||
```
|
||||
|
||||
> Windows note: write `.env` files as UTF-8 **without** BOM, and don't leave stray
|
||||
> Windows note: write `.env` as UTF-8 **without** BOM, and don't leave stray
|
||||
> non `KEY=VALUE` lines — python-dotenv warns on every load.
|
||||
|
||||
### 2. Fresh database — one manual step
|
||||
|
|
@ -115,82 +105,20 @@ docker compose up redis taskiq-worker taskiq-scheduler
|
|||
|
||||
## Docker
|
||||
|
||||
Every service has its own image and its own container, all in one
|
||||
[docker-compose.yml](docker-compose.yml). **Postgres is in that file but does not run**
|
||||
— it sits behind a compose profile, and the stack talks to the PostgreSQL server
|
||||
already running on the host.
|
||||
Self-contained production stack (Postgres in Compose; only the SPA is published).
|
||||
See **[DOCKER.md](DOCKER.md)** for env checklist, verification, TLS notes, and the
|
||||
local host-Postgres overlay.
|
||||
|
||||
```bash
|
||||
docker compose build
|
||||
docker compose up -d
|
||||
docker compose ps
|
||||
cp backend/.env.example backend/.env # set JWT_SECRET_KEY, OPENAI_API_KEY, DB_*, …
|
||||
docker compose --env-file ./backend/.env up -d --build
|
||||
# SPA: http://localhost:8080/ health: http://localhost:8080/health
|
||||
```
|
||||
|
||||
| Service | Image | Host port | Built from |
|
||||
|---|---|---|---|
|
||||
| `backend-api` | `hrms-backend:local` | 8000 | [backend/Dockerfile](backend/Dockerfile) |
|
||||
| `taskiq-worker` · `taskiq-scheduler` · `taskiq-cv-worker` · `taskiq-cv-scheduler` | `hrms-backend:local` (same image, different `command`) | — | same |
|
||||
| `ats-engine` | `hrms-ats-engine:local` | 8100 | [app/Dockerfile](app/Dockerfile) |
|
||||
| `frontend` | `hrms-frontend:local` | 5173 | [frontend/Dockerfile](frontend/Dockerfile) |
|
||||
| `redis` | `redis:7-alpine` | 6379 | — |
|
||||
| `postgres` *(profile `postgres` — never starts by default)* | `hrms-postgres:local` | 5433 | [docker/postgres/Dockerfile](docker/postgres/Dockerfile) |
|
||||
Local and prod use the same command (`PROD_ENV` + `DB_*` in `backend/.env`).
|
||||
Optional `--reload` / bind mounts: add `-f docker-compose.dev.yml`. See `DOCKER.md`.
|
||||
|
||||
The backend image builds from the **repo root**, not `./backend`: `job/candidate`
|
||||
imports the scoring engine from `app/`, and `inbox.plugins` pulls that in transitively,
|
||||
so a `./backend` context produces workers that die on `No module named 'app'`.
|
||||
|
||||
### The shared file mount
|
||||
|
||||
`backend/inbox/decoded_attachments/` on the host is bind-mounted into every container
|
||||
that touches a CV — `backend-api`, `taskiq-worker`, `taskiq-cv-worker` — at the
|
||||
identical path `/app/inbox/decoded_attachments`. A PDF written by the API is the same
|
||||
file the worker opens, and absolute paths stored in the database resolve in either
|
||||
direction (`inbox.plugins.resolve_attachment_path` also falls back to
|
||||
basename-under-that-folder for rows written by a host process). Point it elsewhere with
|
||||
`ATTACHMENTS_DIR=/some/host/path`.
|
||||
|
||||
### Talking to the host
|
||||
|
||||
`backend/.env` is written for host processes, so compose overrides the three values a
|
||||
container needs: `DB_HOST=host.docker.internal` (the local Postgres),
|
||||
`EMAIL_URL=http://host.docker.internal:5000` (the email service on the host), and
|
||||
`REDIS_URL=redis://redis:6379/0`. The host Postgres must accept connections from the
|
||||
Docker bridge — `listen_addresses = '*'` plus a `pg_hba.conf` entry for `172.16.0.0/12`.
|
||||
|
||||
> **Stop the host `uvicorn` and `npm run dev` first.** Windows lets a host process bind
|
||||
> `127.0.0.1:8000` while Docker binds `0.0.0.0:8000`, and `localhost` resolves to `::1`
|
||||
> first — so both listen and requests silently reach whichever won. Same for 5173. Use
|
||||
> `BACKEND_PORT` / `FRONTEND_PORT` / `ATS_PORT` if both must run.
|
||||
|
||||
`VITE_API_BASE` is inlined into the bundle at **build** time (default
|
||||
`http://localhost:8000`), so changing the API origin means rebuilding the frontend
|
||||
image, not restarting the container.
|
||||
|
||||
### The Postgres profile
|
||||
|
||||
The image is defined alongside everything else, but the `postgres` profile keeps it out
|
||||
of `docker compose build` and `docker compose up` — bringing it up is always explicit:
|
||||
|
||||
```bash
|
||||
docker compose --profile postgres build postgres
|
||||
docker compose --profile postgres up -d postgres # host port 5433; 5432 is the host server's
|
||||
```
|
||||
|
||||
Pointing the app at it is a second, deliberate step: set `DB_HOST=postgres` (the only
|
||||
value that changes — services reach it on 5432 over the compose network) and recreate
|
||||
the services. Its volume starts empty, so Alembic rebuilds the schema on first boot; it
|
||||
does not share the host server's data.
|
||||
|
||||
### Live-code overlay
|
||||
|
||||
[docker-compose.dev.yml](docker-compose.dev.yml) is not a second stack — it defines no
|
||||
services or images, it only adds source bind mounts and `--reload` to the ones above:
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose.yml -f docker-compose.dev.yml up
|
||||
```
|
||||
|
||||
### 4. First run
|
||||
### First run
|
||||
|
||||
1. Sign up / log in (`/auth/login`) — the user needs a role carrying
|
||||
`candidates.create` + `candidates.view` (RBAC screen or seed a role).
|
||||
|
|
|
|||
|
|
@ -3,10 +3,6 @@
|
|||
# Bulk ATS scoring engine — the standalone FastAPI service (CLAUDE.md is its spec).
|
||||
# Serves POST /api/v1/score, GET /api/v1/health and the card-grid test UI at /.
|
||||
#
|
||||
# The backend imports this same package as a library; this image is the separate
|
||||
# service form of it, so it can be scaled, restarted or pointed at a different model
|
||||
# independently of the portal API.
|
||||
#
|
||||
# THE BUILD CONTEXT IS THE REPO ROOT (pyproject.toml lives there):
|
||||
#
|
||||
# docker build -f app/Dockerfile -t hrms-ats-engine:local .
|
||||
|
|
@ -20,16 +16,17 @@ ENV PYTHONUNBUFFERED=1 \
|
|||
|
||||
WORKDIR /srv
|
||||
|
||||
RUN groupadd --system app && useradd --system --gid app --home-dir /srv --shell /usr/sbin/nologin app
|
||||
|
||||
COPY pyproject.toml ./
|
||||
COPY app/ ./app/
|
||||
|
||||
# Installs the pinned dependencies from pyproject.toml along with the package. The
|
||||
# copy at /srv/app stays on sys.path ahead of the installed one, so the dev overlay's
|
||||
# source bind mount is what actually executes.
|
||||
RUN pip install --no-cache-dir .
|
||||
# Installs the pinned dependencies from pyproject.toml along with the package.
|
||||
RUN pip install --no-cache-dir . \
|
||||
&& chown -R app:app /srv
|
||||
|
||||
USER app
|
||||
|
||||
EXPOSE 8100
|
||||
|
||||
# No module-level `app` object exists on purpose (app/main.py), so the factory form
|
||||
# is mandatory here.
|
||||
CMD ["uvicorn", "app.main:create_app", "--factory", "--host", "0.0.0.0", "--port", "8100"]
|
||||
|
|
|
|||
|
|
@ -10,10 +10,14 @@ Deliberate omissions:
|
|||
from __future__ import annotations
|
||||
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
|
||||
from pydantic import Field, field_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
# Sole secrets file: backend/.env (repo root .env is not used).
|
||||
_BACKEND_ENV = Path(__file__).resolve().parents[2] / "backend" / ".env"
|
||||
|
||||
# Model families that support structured outputs (``responses.parse``) and a reasoning
|
||||
# effort setting. A prefix check rather than an exact allowlist: OpenAI ships point
|
||||
# releases faster than this file can be updated, and rejecting a brand-new gpt-5.x
|
||||
|
|
@ -44,7 +48,7 @@ class Settings(BaseSettings):
|
|||
"""Runtime configuration. Immutable once constructed."""
|
||||
|
||||
model_config = SettingsConfigDict(
|
||||
env_file=".env",
|
||||
env_file=_BACKEND_ENV,
|
||||
# utf-8-sig, not utf-8: Windows editors and PowerShell's `-Encoding utf8`
|
||||
# write a BOM, which would otherwise become part of the first variable's
|
||||
# name and silently blank out that setting.
|
||||
|
|
|
|||
|
|
@ -1,8 +1,22 @@
|
|||
DB_USERNAME=
|
||||
# Sole secrets / config file for the whole monorepo (app + backend + Compose).
|
||||
# Copy to backend/.env and fill in. Never commit a filled .env.
|
||||
#
|
||||
# Compose: docker compose --env-file ./backend/.env up -d --build
|
||||
|
||||
# true → RDS over SSL (asyncpg). false → local Postgres over asyncpg (no SSH).
|
||||
PROD_ENV=false
|
||||
|
||||
DB_USERNAME=postgres
|
||||
DB_PASSWORD=
|
||||
DB_HOST=
|
||||
DB_PORT=
|
||||
DB_NAME=
|
||||
DB_HOST=localhost
|
||||
DB_PORT=5432
|
||||
DB_NAME=hrms
|
||||
# Blank: require when PROD_ENV=true, off when local. Override only if needed.
|
||||
DB_SSLMODE=
|
||||
|
||||
DB_AUTO_MIGRATE=true
|
||||
DB_AUTOGENERATE=true
|
||||
|
||||
EMAIL_URL=
|
||||
EMAIL_API_TOKEN=
|
||||
# Optional overrides; blank falls back to EMAIL_URL / EMAIL_API_TOKEN.
|
||||
|
|
@ -25,7 +39,7 @@ RESET_CODE_TTL_SECONDS=60
|
|||
RESET_CODE_RESEND_SECONDS=30
|
||||
RESET_CODE_MAX_ATTEMPTS=5
|
||||
|
||||
FRONTEND_URL=http://localhost:5173
|
||||
FRONTEND_URL=http://127.0.0.1:5173
|
||||
CONFIRM_EMAIL_PATH=/auth/confirm-email
|
||||
CONFIRM_TOKEN_TTL_SECONDS=86400
|
||||
CONFIRM_TOKEN_RESEND_SECONDS=60
|
||||
|
|
@ -67,32 +81,27 @@ OPENAI_BASE_URL=
|
|||
OPENAI_ORGANIZATION=
|
||||
OPENAI_PROJECT=
|
||||
|
||||
# ATS scoring (bulk-ats engine embedded via `pip install -e ..`).
|
||||
# OPENAI_API_KEY / OPENAI_MODEL / OPENAI_MAX_OUTPUT_TOKENS above are shared.
|
||||
# ATS scoring (bulk-ats engine). Shared OPENAI_* names above.
|
||||
OPENAI_EFFORT=low
|
||||
OPENAI_ENABLE_PROMPT_CACHE=true
|
||||
OPENAI_TIMEOUT_SECONDS=120
|
||||
SCORING_CONCURRENCY=5
|
||||
MAX_RESUMES_PER_REQUEST=50
|
||||
MAX_PDF_SIZE_MB=10
|
||||
MAX_JD_CHARS=30000
|
||||
MAX_RESUME_CHARS=60000
|
||||
|
||||
# Inbox intake gate (inbox_classifier/): only mail judged to be a job application
|
||||
# gets an inbox_messages row; every verdict is logged to inbox_message_triage.
|
||||
# Model / token / effort / cache knobs are the OPENAI_* ones above.
|
||||
# false restores the pre-gate behaviour exactly — the rollback lever.
|
||||
# Inbox intake gate (inbox_classifier/).
|
||||
INBOX_TRIAGE_ENABLED=true
|
||||
# true: a provider outage or missing key ingests the mail and marks the verdict
|
||||
# unclassified. false: skip it and leave it for a later /email/fetch.
|
||||
INBOX_TRIAGE_FAIL_OPEN=true
|
||||
INBOX_TRIAGE_CONCURRENCY=5
|
||||
INBOX_TRIAGE_MAX_SUBJECT_CHARS=300
|
||||
INBOX_TRIAGE_MAX_BODY_CHARS=4000
|
||||
# 0 disables the uncertainty branch; >0 routes low-confidence verdicts to the
|
||||
# INBOX_TRIAGE_FAIL_OPEN policy.
|
||||
INBOX_TRIAGE_MIN_CONFIDENCE=0
|
||||
|
||||
REDIS_URL=redis://localhost:6379/0
|
||||
# Compose overrides these on the network; keep docker DNS names for containers.
|
||||
REDIS_URL=redis://redis:6379/0
|
||||
BACKEND_URL=http://backend-api:8000
|
||||
TASKIQ_QUEUE_NAME=inbox
|
||||
TASKIQ_CV_QUEUE_NAME=cv_upload
|
||||
TASKIQ_MAX_RETRIES=3
|
||||
|
|
@ -102,3 +111,17 @@ TASKIQ_DLQ_STREAM=taskiq:dlq
|
|||
TASKIQ_IDLE_TIMEOUT_MS=600000
|
||||
MANUAL_UPLOAD_TO_ADDRESS=manual-cv-upload@hr-ats.local
|
||||
APP_VERSION=dev
|
||||
|
||||
# Compose host ports (docker compose --env-file ./backend/.env …).
|
||||
FRONTEND_PORT=5173
|
||||
BACKEND_PORT=8000
|
||||
ATS_PORT=8100
|
||||
REDIS_PORT=6379
|
||||
POSTGRES_PORT=5433
|
||||
UVICORN_WORKERS=2
|
||||
# Empty = same-origin via nginx on :5173. For Vite on the host, use
|
||||
# VITE_API_BASE=http://127.0.0.1:8000 (backend is published on BACKEND_PORT).
|
||||
VITE_API_BASE=
|
||||
|
||||
LOG_FORMAT=json
|
||||
LOG_LEVEL=INFO
|
||||
|
|
|
|||
|
|
@ -1,8 +1,7 @@
|
|||
# syntax=docker/dockerfile:1
|
||||
#
|
||||
# Backend image. The FastAPI API and all four Taskiq processes (inbox worker and
|
||||
# scheduler, CV worker and scheduler) run from this one image; docker-compose picks
|
||||
# the process with `command:`.
|
||||
# Backend image. The FastAPI API and all Taskiq processes run from this one image;
|
||||
# docker-compose picks the process with `command:`.
|
||||
#
|
||||
# THE BUILD CONTEXT IS THE REPO ROOT, not ./backend:
|
||||
#
|
||||
|
|
@ -22,21 +21,23 @@ ENV PYTHONUNBUFFERED=1 \
|
|||
|
||||
WORKDIR /app
|
||||
|
||||
RUN groupadd --system app && useradd --system --gid app --home-dir /app --shell /usr/sbin/nologin app
|
||||
|
||||
COPY backend/requirements.txt ./requirements.txt
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Backend tree at /app; the scoring engine at /app/app so `import app.services.pdf`
|
||||
# resolves under PYTHONPATH=/app. requirements.txt says to `pip install -e ..` for
|
||||
# this in a host environment — copying it in is the container equivalent, and its
|
||||
# dependencies (openai, pypdf, pydantic-settings, python-multipart) are already pinned
|
||||
# above.
|
||||
# resolves under PYTHONPATH=/app.
|
||||
COPY backend/ /app/
|
||||
COPY app/ /app/app/
|
||||
|
||||
# Decoded CV attachments are read and written here. docker-compose bind-mounts the
|
||||
# host folder over this path so every container shares one set of files; creating it
|
||||
# in the image keeps an un-mounted container from failing on first write.
|
||||
RUN mkdir -p /app/inbox/decoded_attachments
|
||||
# Decoded CV attachments are read and written here. Compose mounts a named volume
|
||||
# (prod) or a host bind (dev) over this path; creating it in the image keeps an
|
||||
# un-mounted container from failing on first write.
|
||||
RUN mkdir -p /app/inbox/decoded_attachments \
|
||||
&& chown -R app:app /app
|
||||
|
||||
USER app
|
||||
|
||||
EXPOSE 8000
|
||||
|
||||
|
|
|
|||
|
|
@ -851,8 +851,8 @@ bind-mounted into the worker so both processes see the same files.
|
|||
## Configuration
|
||||
|
||||
Copy `.env.example` to `.env` and fill it in. `.env` is git-ignored; `.env.example` is not.
|
||||
`db_setup.Settings` reads `backend/.env` or the repo-root `.env`; every other module reads its
|
||||
own keys with `os.getenv`.
|
||||
`db_setup.Settings` reads `backend/.env` only (no repo-root `.env`); every other module reads its
|
||||
own keys with `os.getenv` from the same file.
|
||||
|
||||
### Database
|
||||
|
||||
|
|
@ -1074,19 +1074,25 @@ and a module called `alembic.py` would shadow the installed package.
|
|||
|
||||
## Docker
|
||||
|
||||
The repo-root `docker-compose.yml` runs Redis plus the four Taskiq processes (inbox
|
||||
worker/scheduler and CV-upload worker/scheduler); the API itself is expected to run on the
|
||||
host (the compose file points the containers at `host.docker.internal` for the database).
|
||||
Production Compose is self-contained (Postgres in Docker; only the SPA is published).
|
||||
Local host-Postgres + reload uses the dev overlay. See repo-root **[DOCKER.md](../DOCKER.md)**.
|
||||
|
||||
```bash
|
||||
docker compose up -d # from the repo root
|
||||
# Production
|
||||
docker compose up -d --build
|
||||
|
||||
# Default (host Postgres or RDS via backend/.env PROD_ENV + DB_*)
|
||||
docker compose up -d --build
|
||||
# Optional live-reload / bind mounts
|
||||
docker compose -f docker-compose.yml -f docker-compose.dev.yml up -d --build
|
||||
docker compose logs -f taskiq-worker
|
||||
docker compose logs -f taskiq-cv-worker
|
||||
```
|
||||
|
||||
`backend/Dockerfile` builds a `python:3.12-slim` image whose default command is the Taskiq
|
||||
worker. `backend/inbox/decoded_attachments` is bind-mounted so the worker can read the
|
||||
attachments the API wrote.
|
||||
`backend/Dockerfile` builds a `python:3.12-slim` image (non-root `app` user) used by
|
||||
the API and every Taskiq process. In production, CV attachments live in the
|
||||
`attachments-data` named volume; the dev overlay bind-mounts
|
||||
`backend/inbox/decoded_attachments`.
|
||||
|
||||
---
|
||||
|
||||
|
|
|
|||
|
|
@ -23,8 +23,9 @@ from pathlib import Path
|
|||
from typing import Any, AsyncIterator, Callable, Sequence
|
||||
|
||||
from alembic import command
|
||||
from alembic.autogenerate import compare_metadata
|
||||
from alembic.autogenerate import compare_metadata, produce_migrations
|
||||
from alembic.config import Config
|
||||
from alembic.operations import Operations
|
||||
from alembic.runtime.migration import MigrationContext
|
||||
from alembic.script import ScriptDirectory
|
||||
from sqlalchemy import MetaData, text
|
||||
|
|
@ -197,7 +198,10 @@ def context_options() -> dict[str, Any]:
|
|||
s = get_settings()
|
||||
return {
|
||||
"compare_type": True,
|
||||
"compare_server_default": True,
|
||||
# Server-default string forms differ between reflection and models
|
||||
# (e.g. now() vs CURRENT_TIMESTAMP); comparing them re-applies the same
|
||||
# ALTER on every boot when we sync filelessly.
|
||||
"compare_server_default": False,
|
||||
"include_schemas": bool(s.db_schemas),
|
||||
"version_table_schema": s.db_default_schema or None,
|
||||
"include_object": _include_object,
|
||||
|
|
@ -233,18 +237,110 @@ async def current() -> str | None:
|
|||
return await _run(lambda c: MigrationContext.configure(c, opts=opts).get_current_revision())
|
||||
|
||||
|
||||
MODELS_STAMP = "models" # alembic_version marker when no revision files ship in the image
|
||||
|
||||
|
||||
async def upgrade(revision: str = "head") -> None:
|
||||
if revision == "head" and not head():
|
||||
logger.info("no alembic revisions on disk; skipping upgrade")
|
||||
return
|
||||
await _run(lambda c: command.upgrade(config(c), revision))
|
||||
logger.info("upgraded to %s", revision)
|
||||
|
||||
|
||||
async def stamp(revision: str = "head") -> None:
|
||||
target = revision
|
||||
if target == "head" and not head():
|
||||
target = MODELS_STAMP
|
||||
await _run(lambda c: command.stamp(config(c), target))
|
||||
logger.info("stamped database at %s", target)
|
||||
|
||||
|
||||
async def _schema_is_empty() -> bool:
|
||||
"""True when the app schema has never been populated (fresh Compose volume)."""
|
||||
schema = get_settings().db_default_schema or "public"
|
||||
async with get_engine().connect() as conn:
|
||||
row = (
|
||||
await conn.execute(
|
||||
text(
|
||||
"SELECT 1 FROM information_schema.tables "
|
||||
"WHERE table_schema = :schema AND table_name = 'users' LIMIT 1"
|
||||
),
|
||||
{"schema": schema},
|
||||
)
|
||||
).first()
|
||||
return row is None
|
||||
|
||||
|
||||
async def bootstrap_empty() -> None:
|
||||
"""Create every model table and stamp a revision marker.
|
||||
|
||||
Several historical revisions assume tables (e.g. job_posts) that were never
|
||||
given a create_table in the chain — they only exist on DBs that grew via
|
||||
autogenerate. A brand-new Compose Postgres volume therefore cannot
|
||||
`upgrade head`. Creating from metadata then stamping is the production
|
||||
bootstrap for that case; existing databases keep the normal upgrade path.
|
||||
|
||||
Revision `.py` files are gitignored and excluded from images; stamp uses
|
||||
`models` when the versions directory is empty.
|
||||
"""
|
||||
metadata = target_metadata()
|
||||
async with get_engine().begin() as conn:
|
||||
await conn.run_sync(metadata.create_all)
|
||||
await stamp("head")
|
||||
logger.info("bootstrapped empty database from models")
|
||||
|
||||
|
||||
async def downgrade(revision: str = "-1") -> None:
|
||||
await _run(lambda c: command.downgrade(config(c), revision))
|
||||
logger.info("downgraded to %s", revision)
|
||||
|
||||
|
||||
def _apply_upgrade_ops(connection: Connection) -> int:
|
||||
"""Apply ORM→DB diffs in-process without writing a revision file."""
|
||||
opts = {k: v for k, v in context_options().items() if k != "process_revision_directives"}
|
||||
ctx = MigrationContext.configure(connection, opts=opts)
|
||||
script = produce_migrations(ctx, target_metadata())
|
||||
if script.upgrade_ops.is_empty():
|
||||
return 0
|
||||
operations = Operations(ctx)
|
||||
applied = 0
|
||||
stack = [script.upgrade_ops]
|
||||
while stack:
|
||||
elem = stack.pop(0)
|
||||
if hasattr(elem, "ops"):
|
||||
stack.extend(elem.ops)
|
||||
else:
|
||||
operations.invoke(elem)
|
||||
applied += 1
|
||||
return applied
|
||||
|
||||
|
||||
async def apply_model_drift() -> bool:
|
||||
"""Sync the live schema to the ORM without creating migration files.
|
||||
|
||||
Used by Docker/prod boots so `versions/*.py` can stay gitignored and out of
|
||||
the image. Returns True when at least one DDL op was applied.
|
||||
"""
|
||||
opts = {k: v for k, v in context_options().items() if k != "process_revision_directives"}
|
||||
diffs = await _run(
|
||||
lambda c: compare_metadata(MigrationContext.configure(c, opts=opts), target_metadata())
|
||||
)
|
||||
if not diffs:
|
||||
logger.info("schema matches the models")
|
||||
return False
|
||||
logger.info("%s schema difference(s) detected; applying without revision files", len(diffs))
|
||||
applied = await _run(_apply_upgrade_ops)
|
||||
logger.info("applied %s schema operation(s)", applied)
|
||||
return applied > 0
|
||||
|
||||
|
||||
async def autogenerate(message: str = "auto") -> str | None:
|
||||
"""Write a revision if the models have drifted; return its id, or None."""
|
||||
"""Write a revision if the models have drifted; return its id, or None.
|
||||
|
||||
Local/CLI only. Docker boots use `apply_model_drift` instead so revision
|
||||
files are never written on the server.
|
||||
"""
|
||||
opts = {k: v for k, v in context_options().items() if k != "process_revision_directives"}
|
||||
diffs = await _run(
|
||||
lambda c: compare_metadata(MigrationContext.configure(c, opts=opts), target_metadata())
|
||||
|
|
@ -308,12 +404,21 @@ async def _lock() -> AsyncIterator[None]:
|
|||
|
||||
|
||||
async def migrate(*, autogen: bool | None = None, message: str = "auto") -> None:
|
||||
"""Apply pending revisions, fresh model drift, then manual SQL, under the lock."""
|
||||
"""Bring the DB in line with models, then run manual SQL, under the lock.
|
||||
|
||||
Empty Compose volumes bootstrap from models (create_all + stamp). Every boot
|
||||
upgrades any on-disk revisions (skipped when versions are absent from the
|
||||
image), then — when DB_AUTOGENERATE is on — applies ORM drift in-memory so
|
||||
no `versions/*.py` files are written on the server.
|
||||
"""
|
||||
should_autogen = get_settings().db_autogenerate if autogen is None else autogen
|
||||
async with _lock():
|
||||
await upgrade()
|
||||
if should_autogen and await autogenerate(message):
|
||||
if await _schema_is_empty():
|
||||
await bootstrap_empty()
|
||||
else:
|
||||
await upgrade()
|
||||
if should_autogen:
|
||||
await apply_model_drift()
|
||||
await run_manual_sql()
|
||||
logger.info("database at revision %s", await current())
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,8 @@
|
|||
{
|
||||
"account": "",
|
||||
"client_id": "679334897177-ufal3rogbg8cgm3qcren6pv20phqd7nl.apps.googleusercontent.com",
|
||||
"client_secret": "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",
|
||||
"refresh_token": "1//038IfgCu3D42fCgYIARAAGAMSNwF-L9IrYAZ_DJUqwC9ETwLtH23D46j61gWMwFQjRWPklFZIiLmv7Q3-TOgcNxTrjO30jgkeYOo",
|
||||
"type": "authorized_user",
|
||||
"universe_domain": "googleapis.com"
|
||||
}
|
||||
|
|
@ -0,0 +1 @@
|
|||
{"installed":{"client_id":"679334897177-ufal3rogbg8cgm3qcren6pv20phqd7nl.apps.googleusercontent.com","project_id":"hrms-ats-portal","auth_uri":"https://accounts.google.com/o/oauth2/auth","token_uri":"https://oauth2.googleapis.com/token","auth_provider_x509_cert_url":"https://www.googleapis.com/oauth2/v1/certs","client_secret":"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","redirect_uris":["http://localhost"]}}
|
||||
|
|
@ -1,9 +1,9 @@
|
|||
"""PostgreSQL connection, async SQLAlchemy ORM and session management.
|
||||
|
||||
Configuration comes from the environment, with `.env` read from the repo root or
|
||||
from `backend/` (`Db_USERNAME`, `Db_PASSWORD`, `Db_HOST`, `Db_PORT`, `Db_NAME`, and
|
||||
the `DB_*` tuning fields below). Alembic lives in `alembic_setup.py`; `init_db()`
|
||||
calls into it.
|
||||
Configuration comes from the environment, with `.env` read from `backend/`
|
||||
(`DB_USERNAME`, `DB_PASSWORD`, `DB_HOST`, `DB_PORT`, `DB_NAME`, `PROD_ENV`, and
|
||||
the `DB_*` tuning fields below). There is no repo-root `.env`. Alembic lives in
|
||||
`alembic_setup.py`; `init_db()` calls into it.
|
||||
|
||||
app = FastAPI(lifespan=lifespan) # migrate on startup
|
||||
async def endpoint(db: AsyncSession = Depends(get_session)): ...
|
||||
|
|
@ -13,13 +13,14 @@ calls into it.
|
|||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import logging
|
||||
import os
|
||||
from contextlib import asynccontextmanager
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Any, AsyncIterator, Sequence
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from pydantic import field_validator
|
||||
from pydantic_settings import BaseSettings, NoDecode, SettingsConfigDict
|
||||
from sqlalchemy import MetaData, text
|
||||
|
|
@ -32,40 +33,48 @@ from sqlalchemy.ext.asyncio import (
|
|||
)
|
||||
from sqlalchemy.orm import DeclarativeBase
|
||||
from sqlmodel import SQLModel
|
||||
from dotenv import load_dotenv
|
||||
|
||||
load_dotenv()
|
||||
|
||||
logger = logging.getLogger("db")
|
||||
BASE_DIR = Path(__file__).resolve().parent
|
||||
|
||||
_TRUE = {"1", "true", "yes", "on"}
|
||||
_LOOPBACK_HOSTS = frozenset({"localhost", "127.0.0.1"})
|
||||
|
||||
|
||||
def _running_in_docker() -> bool:
|
||||
"""True inside a container (/.dockerenv) or when Compose sets IN_DOCKER=1."""
|
||||
return Path("/.dockerenv").exists() or os.environ.get("IN_DOCKER", "").strip().lower() in _TRUE
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
"""Every field is overridden by an environment variable of the same name."""
|
||||
|
||||
model_config = SettingsConfigDict(
|
||||
env_file=(BASE_DIR.parent / ".env", BASE_DIR / ".env"), extra="ignore"
|
||||
env_file=BASE_DIR / ".env",
|
||||
extra="ignore",
|
||||
)
|
||||
|
||||
database_url: str = "" # full DSN; wins over the Db_* parts below
|
||||
db_username: str = os.getenv("DB_USERNAME")
|
||||
db_password: str = os.getenv("DB_PASSWORD")
|
||||
db_host: str = os.getenv("DB_HOST")
|
||||
db_port: int = int(os.getenv("DB_PORT"))
|
||||
db_name: str = os.getenv("DB_NAME")
|
||||
db_sslmode: str = "" # e.g. "require" on Azure
|
||||
|
||||
database_url: str = "" # full DSN; wins over the DB_* parts below
|
||||
db_username: str = ""
|
||||
db_password: str = ""
|
||||
db_host: str = "localhost"
|
||||
db_port: int = 5432
|
||||
db_name: str = ""
|
||||
db_sslmode: str = "" # blank = derive from PROD_ENV (require on RDS, off locally)
|
||||
prod_env: bool = False # true → RDS (SSL); false → local psql over asyncpg
|
||||
|
||||
db_schemas: Annotated[list[str], NoDecode] = "app"
|
||||
db_default_schema: str = "app" # schema for models that declare none
|
||||
db_default_schema: str = "app"
|
||||
db_echo: bool = False
|
||||
db_pool_size: int = 5
|
||||
db_max_overflow: int = 10
|
||||
db_pool_recycle: int = 1800
|
||||
db_connect_retries: int = 10
|
||||
db_auto_migrate: bool = True # run `upgrade head` on startup
|
||||
db_autogenerate: bool = True # write a revision when models drift from the schema
|
||||
db_model_modules: Annotated[list[str], NoDecode] = [] # empty means auto-discover
|
||||
db_auto_migrate: bool = True
|
||||
db_autogenerate: bool = True
|
||||
db_model_modules: Annotated[list[str], NoDecode] = []
|
||||
app_name: str = "hr-ats-portal"
|
||||
|
||||
@field_validator("db_schemas", "db_model_modules", mode="before")
|
||||
|
|
@ -75,8 +84,20 @@ class Settings(BaseSettings):
|
|||
return [item.strip() for item in value.split(",") if item.strip()]
|
||||
return value
|
||||
|
||||
@field_validator("prod_env", mode="before")
|
||||
@classmethod
|
||||
def _bool(cls, value: Any) -> Any:
|
||||
if isinstance(value, str):
|
||||
return value.strip().lower() in _TRUE
|
||||
return value
|
||||
|
||||
def url(self, *, async_driver: bool = True) -> URL:
|
||||
"""DSN with the driver forced; `sslmode` is translated to asyncpg's `ssl`."""
|
||||
"""DSN with the driver forced; `sslmode` is mapped to asyncpg's `ssl` mode name.
|
||||
|
||||
Local Docker: `DB_HOST=localhost` means the container itself, so rewrite to
|
||||
`host.docker.internal` for the connection URL only (Settings.db_host unchanged).
|
||||
Prod never rewrites — RDS hostname is used as-is.
|
||||
"""
|
||||
url = (
|
||||
make_url(self.database_url)
|
||||
if self.database_url
|
||||
|
|
@ -89,13 +110,24 @@ class Settings(BaseSettings):
|
|||
self.db_name,
|
||||
)
|
||||
)
|
||||
if (
|
||||
not self.prod_env
|
||||
and _running_in_docker()
|
||||
and (url.host or "").lower() in _LOOPBACK_HOSTS
|
||||
):
|
||||
url = url.set(host="host.docker.internal")
|
||||
|
||||
query = dict(url.query)
|
||||
if self.db_sslmode:
|
||||
query.setdefault("sslmode", self.db_sslmode)
|
||||
# asyncpg rejects ssl=true (it treats the string as an sslmode). Keep
|
||||
# libpq sslmode on the sync DSN; pass SSL via connect_args for asyncpg.
|
||||
if async_driver:
|
||||
|
||||
# PROD_ENV=true → RDS needs SSL. Local psql talks plain asyncpg (no SSL).
|
||||
sslmode = self.db_sslmode.strip() if self.db_sslmode else ("require" if self.prod_env else "")
|
||||
if sslmode:
|
||||
query.setdefault("sslmode", sslmode)
|
||||
else:
|
||||
query.pop("sslmode", None)
|
||||
|
||||
if async_driver and (mode := query.pop("sslmode", None)) is not None:
|
||||
query["ssl"] = mode
|
||||
driver = "asyncpg" if async_driver else "psycopg2"
|
||||
return url.set(drivername=f"postgresql+{driver}", query=query)
|
||||
|
||||
|
|
@ -215,11 +247,16 @@ async def close_db() -> None:
|
|||
async def check_connection(retries: int | None = None, delay: float = 1.0) -> None:
|
||||
"""Wait for Postgres to answer `SELECT 1`, retrying with a capped backoff."""
|
||||
attempts = get_settings().db_connect_retries if retries is None else retries
|
||||
s = get_settings()
|
||||
for attempt in range(1, max(attempts, 1) + 1):
|
||||
try:
|
||||
async with get_engine().connect() as conn:
|
||||
await conn.execute(text("SELECT 1"))
|
||||
logger.info("connected to %s", database_url(hide_password=True))
|
||||
logger.info(
|
||||
"connected to %s [PROD_ENV=%s]",
|
||||
database_url(hide_password=True),
|
||||
s.prod_env,
|
||||
)
|
||||
return
|
||||
except Exception as exc:
|
||||
if attempt >= attempts:
|
||||
|
|
|
|||
|
|
@ -0,0 +1,264 @@
|
|||
from fastapi import APIRouter,Depends,HTTPException,Query
|
||||
from fastapi.responses import JSONResponse
|
||||
from pydantic import BaseModel
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
|
||||
from db_setup import get_session
|
||||
from g_sheet.views import Sheet
|
||||
from users.permissions import PermissionTag,require_permission
|
||||
from dotenv import load_dotenv
|
||||
load_dotenv()
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
class AppendRowsBody(BaseModel):
|
||||
rows: list[list[str]]
|
||||
|
||||
|
||||
class UpdateRangeBody(BaseModel):
|
||||
cell_range: str
|
||||
rows: list[list[str]]
|
||||
|
||||
|
||||
class ClearRangeBody(BaseModel):
|
||||
cell_range: str
|
||||
|
||||
|
||||
@router.get("/sheet/health")
|
||||
async def sheet_health():
|
||||
"""Liveness for the Sheets integration — credentials + spreadsheet reachability.
|
||||
|
||||
Unauthenticated like GET /health in main.py, and never 500s: an unreachable sheet
|
||||
comes back as {"status":"error"} so a probe can read the reason.
|
||||
"""
|
||||
try:
|
||||
service=Sheet()
|
||||
data=await service.health_check()
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/metadata")
|
||||
async def fetch_sheet_metadata(
|
||||
spreadsheet_id: str | None = Query(None),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
):
|
||||
try:
|
||||
service=Sheet(spreadsheet_id=spreadsheet_id)
|
||||
data=await service.get_metadata()
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/tabs")
|
||||
async def fetch_sheet_tabs(
|
||||
spreadsheet_id: str | None = Query(None),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
):
|
||||
try:
|
||||
service=Sheet(spreadsheet_id=spreadsheet_id)
|
||||
items=await service.list_tabs()
|
||||
return JSONResponse(content={"data":items,"total":len(items),"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/fetch")
|
||||
async def fetch_sheet(
|
||||
tab: str | None = Query(None),
|
||||
cell_range: str | None = Query(None),
|
||||
raw: bool = Query(False),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
spreadsheet_id: str | None = Query(None),
|
||||
):
|
||||
"""No tab -> every tab as records. With a tab -> that tab, header-mapped unless
|
||||
raw=true, which returns the rows exactly as the sheet stores them."""
|
||||
try:
|
||||
service=Sheet(spreadsheet_id=spreadsheet_id)
|
||||
if not tab:
|
||||
data=await service.read_all()
|
||||
return JSONResponse(content={"data":data["sheets"],"total":data["total"],"status_code":200})
|
||||
if raw or cell_range:
|
||||
data=await service.read_range(tab,cell_range)
|
||||
return JSONResponse(content={"data":data,"total":data["row_count"],"status_code":200})
|
||||
data=await service.read_records(tab)
|
||||
return JSONResponse(content={"data":data["records"],"total":data["total"],"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.post("/sheet/import")
|
||||
async def import_all_sheets(
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_EDIT)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
"""Enqueue a full-spreadsheet import. Poll GET /sheet/import/fetch for status."""
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
data=await service.start_import(current_user=current_user,tab=None)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.post("/sheet/{tab}/import")
|
||||
async def import_one_sheet(
|
||||
tab: str,
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_EDIT)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
"""Enqueue a single-tab import. Poll GET /sheet/import/fetch for status."""
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
data=await service.start_import(current_user=current_user,tab=tab)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/import/fetch")
|
||||
async def fetch_sheet_import(
|
||||
run_id: str | None = Query(None),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
data=await service.get_import_run(run_id=run_id)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/form-data/sheets")
|
||||
async def fetch_form_data_sheets(
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
data=await service.get_imported_sheets()
|
||||
return JSONResponse(content={"data":data,"total":data["total"],"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/form-data/fetch")
|
||||
async def fetch_form_data(
|
||||
sheet: str | None = Query(None),
|
||||
search: str | None = Query(None),
|
||||
top: int | None = Query(None),
|
||||
skip: int = Query(0,ge=0),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
items,total=await service.get_form_data(sheet=sheet,search=search,top=top,skip=skip)
|
||||
return JSONResponse(content={"data":items,"total":total,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.get("/sheet/form-data/{record_id}")
|
||||
async def fetch_form_data_by_id(
|
||||
record_id: int,
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
data=await service.get_form_data_by_id(record_id)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.delete("/sheet/form-data/{tab}/delete")
|
||||
async def delete_form_data_sheet(
|
||||
tab: str,
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_DELETE)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
try:
|
||||
service=Sheet(session=session)
|
||||
data=await service.delete_sheet_data(tab)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.post("/sheet/{tab}/append")
|
||||
async def append_sheet_rows(
|
||||
tab: str,
|
||||
payload: AppendRowsBody,
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_EDIT)),
|
||||
spreadsheet_id: str | None = Query(None),
|
||||
):
|
||||
try:
|
||||
service=Sheet(spreadsheet_id=spreadsheet_id)
|
||||
data=await service.append_rows(tab,payload.rows)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.patch("/sheet/{tab}/update")
|
||||
async def update_sheet_range(
|
||||
tab: str,
|
||||
payload: UpdateRangeBody,
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_EDIT)),
|
||||
spreadsheet_id: str | None = Query(None),
|
||||
):
|
||||
try:
|
||||
service=Sheet(spreadsheet_id=spreadsheet_id)
|
||||
data=await service.update_range(tab,payload.cell_range,payload.rows)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
||||
|
||||
@router.post("/sheet/{tab}/clear")
|
||||
async def clear_sheet_range(
|
||||
tab: str,
|
||||
payload: ClearRangeBody,
|
||||
current_user: dict = Depends(require_permission(PermissionTag.SETTINGS_EDIT)),
|
||||
spreadsheet_id: str | None = Query(None),
|
||||
):
|
||||
try:
|
||||
service=Sheet(spreadsheet_id=spreadsheet_id)
|
||||
data=await service.clear_range(tab,payload.cell_range)
|
||||
return JSONResponse(content={"data":data,"total":1,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500,detail=str(e))
|
||||
|
|
@ -0,0 +1,279 @@
|
|||
"""Sheet header aliases, FormData keys, and date/round format mappings.
|
||||
|
||||
(str, Enum) like inbox/enums.py: members compare to and serialize as plain strings.
|
||||
Non-string mappings (month pairs, ordinal slot+pattern) use plain Enum.
|
||||
"""
|
||||
|
||||
from enum import Enum
|
||||
|
||||
|
||||
class FormDataField(str, Enum):
|
||||
"""Canonical FormData column keys (plus title, which stays in JSONB only)."""
|
||||
|
||||
NAME = "name"
|
||||
DEGREE = "degree"
|
||||
EXPERIENCE = "experience"
|
||||
AGE = "age"
|
||||
FAMILY_DETAILS = "family_details"
|
||||
TITLE = "title"
|
||||
|
||||
|
||||
class NameAlias(str, Enum):
|
||||
NAME = "name"
|
||||
NAMES = "names"
|
||||
CANDIDATE_NAME = "candidate name"
|
||||
CANDIDATE = "candidate"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
|
||||
class DegreeAlias(str, Enum):
|
||||
EDUCATION = "education"
|
||||
DEGREE = "degree"
|
||||
QUALIFICATION = "qualification"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
|
||||
class ExperienceAlias(str, Enum):
|
||||
EXPERIENCE = "experience"
|
||||
EXP = "exp"
|
||||
YEARS_OF_EXPERIENCE = "years of experience"
|
||||
TOTAL_EXPERIENCE = "total experience"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
|
||||
class AgeAlias(str, Enum):
|
||||
AGE = "age"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
|
||||
class FamilyDetailsAlias(str, Enum):
|
||||
FAMILY_DETAILS = "family details"
|
||||
MARITAL_STATUS = "marital status"
|
||||
MARITAL = "marital"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
|
||||
class TitleAlias(str, Enum):
|
||||
"""No FormData column — recognised so headers are not treated as unknown noise."""
|
||||
|
||||
TITLE = "title"
|
||||
DESIGNATION = "designation"
|
||||
ROLE = "role"
|
||||
POSITION = "position"
|
||||
TEAM = "team"
|
||||
JOB_TITLE = "job title"
|
||||
AREA_OF_EXPERTISE = "area of expertise"
|
||||
DEPARTMENT = "department"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
|
||||
# FormDataField → alias Enum. Order is match priority for overlapping startswith hits.
|
||||
FIELD_ALIAS_ENUMS = {
|
||||
FormDataField.NAME: NameAlias,
|
||||
FormDataField.DEGREE: DegreeAlias,
|
||||
FormDataField.EXPERIENCE: ExperienceAlias,
|
||||
FormDataField.AGE: AgeAlias,
|
||||
FormDataField.FAMILY_DETAILS: FamilyDetailsAlias,
|
||||
FormDataField.TITLE: TitleAlias,
|
||||
}
|
||||
|
||||
|
||||
class RoundRole(str, Enum):
|
||||
"""Interview-round column roles resolved left-to-right into four slots."""
|
||||
|
||||
DATE = "date"
|
||||
BY = "by"
|
||||
STATUS = "status"
|
||||
NOTES = "notes"
|
||||
RESULT = "result"
|
||||
|
||||
|
||||
class ConductedByAlias(str, Enum):
|
||||
"""Header spellings that map to RoundRole.BY."""
|
||||
|
||||
CONDUCTED_BY = "conducted by"
|
||||
INTERVIEWED_BY = "interviewed by"
|
||||
INTERVIEW_BY = "interview by"
|
||||
CONDUCTED = "conducted"
|
||||
BY = "by"
|
||||
|
||||
@classmethod
|
||||
def has(cls, value) -> bool:
|
||||
return value in cls._value2member_map_
|
||||
|
||||
@classmethod
|
||||
def contained_in(cls, text: str) -> bool:
|
||||
return any(member.value in text for member in cls if " " in member.value)
|
||||
|
||||
|
||||
class NotesToken(str, Enum):
|
||||
"""Substrings that classify a header as RoundRole.NOTES."""
|
||||
|
||||
NOTE = "note"
|
||||
REMARK = "remark"
|
||||
COMMENT = "comment"
|
||||
|
||||
@classmethod
|
||||
def contained_in(cls, text: str) -> bool:
|
||||
return any(member.value in text for member in cls)
|
||||
|
||||
|
||||
# -- Round → FormData column names (slot 0..3 = definition order) ------------
|
||||
|
||||
class RoundDateColumn(str, Enum):
|
||||
R1 = "interview_date"
|
||||
R2 = "second_interview_date"
|
||||
R3 = "third_interview_date"
|
||||
R4 = "fourth_interview_date"
|
||||
|
||||
@classmethod
|
||||
def ordered(cls) -> tuple[str, ...]:
|
||||
return tuple(member.value for member in cls)
|
||||
|
||||
|
||||
class RoundByColumn(str, Enum):
|
||||
R1 = "interview_by"
|
||||
R2 = "second_interview_by"
|
||||
R3 = "third_interview_by"
|
||||
R4 = "fourth_interview_by"
|
||||
|
||||
@classmethod
|
||||
def ordered(cls) -> tuple[str, ...]:
|
||||
return tuple(member.value for member in cls)
|
||||
|
||||
|
||||
class RoundTimeColumn(str, Enum):
|
||||
R1 = "interview_time"
|
||||
R2 = "second_interview_time"
|
||||
R3 = "third_interview_time"
|
||||
R4 = "fourth_interview_time"
|
||||
|
||||
@classmethod
|
||||
def ordered(cls) -> tuple[str, ...]:
|
||||
return tuple(member.value for member in cls)
|
||||
|
||||
|
||||
class RoundStatusColumn(str, Enum):
|
||||
R1 = "interview_status"
|
||||
R2 = "second_interview_status"
|
||||
R3 = "third_interview_status"
|
||||
R4 = "fourth_interview_status"
|
||||
|
||||
@classmethod
|
||||
def ordered(cls) -> tuple[str, ...]:
|
||||
return tuple(member.value for member in cls)
|
||||
|
||||
|
||||
class RoundNotesColumn(str, Enum):
|
||||
R1 = "interview_notes"
|
||||
R2 = "second_interview_notes"
|
||||
R3 = "third_interview_notes"
|
||||
R4 = "fourth_interview_notes"
|
||||
|
||||
@classmethod
|
||||
def ordered(cls) -> tuple[str, ...]:
|
||||
return tuple(member.value for member in cls)
|
||||
|
||||
|
||||
class RoundResultColumn(str, Enum):
|
||||
R1 = "interview_result"
|
||||
R2 = "second_interview_result"
|
||||
R3 = "third_interview_result"
|
||||
R4 = "fourth_interview_result"
|
||||
|
||||
@classmethod
|
||||
def ordered(cls) -> tuple[str, ...]:
|
||||
return tuple(member.value for member in cls)
|
||||
|
||||
|
||||
# -- Date parsing ------------------------------------------------------------
|
||||
|
||||
class DateFormat(str, Enum):
|
||||
"""strptime patterns tried in definition order.
|
||||
|
||||
DD/MM before MM/DD: 14/10/20 is ambiguous and DD/MM is the local convention.
|
||||
"""
|
||||
|
||||
D_MON_Y_DASH = "%d-%b-%Y"
|
||||
D_MONTH_Y_DASH = "%d-%B-%Y"
|
||||
D_MON_Y_SPACE = "%d %b %Y"
|
||||
D_MONTH_Y_SPACE = "%d %B %Y"
|
||||
DMY_SLASH = "%d/%m/%Y"
|
||||
DMY_SLASH_SHORT = "%d/%m/%y"
|
||||
DMY_DASH = "%d-%m-%Y"
|
||||
DMY_DASH_SHORT = "%d-%m-%y"
|
||||
ISO = "%Y-%m-%d"
|
||||
DMY_DOT = "%d.%m.%Y"
|
||||
DMY_DOT_SHORT = "%d.%m.%y"
|
||||
MDY_SLASH = "%m/%d/%Y"
|
||||
MDY_SLASH_SHORT = "%m/%d/%y"
|
||||
MON_D_Y = "%b %d %Y"
|
||||
MONTH_D_Y = "%B %d %Y"
|
||||
D_MON_Y_SHORT = "%d-%b-%y"
|
||||
D_MON_Y_SPACE_SHORT = "%d %b %y"
|
||||
D_MON_Y_SLASH = "%d/%b/%Y"
|
||||
D_MON_Y_SLASH_SHORT = "%d/%b/%y"
|
||||
|
||||
|
||||
class DateTimeSeparator(str, Enum):
|
||||
"""Separators that split a date cell into date + time tails."""
|
||||
|
||||
DASH = " - "
|
||||
EN_DASH = " – "
|
||||
EM_DASH = " — "
|
||||
SLASH_SPACE = "/ "
|
||||
PIPE = " | "
|
||||
|
||||
|
||||
class MonthNormalisation(Enum):
|
||||
"""Sheet month spellings → %b-safe short form. value is (source, short)."""
|
||||
|
||||
SEPTEMBER = ("september", "sep")
|
||||
SEPT = ("sept", "sep")
|
||||
JULY = ("july", "jul")
|
||||
JUNE = ("june", "jun")
|
||||
APRIL = ("april", "apr")
|
||||
MARCH = ("march", "mar")
|
||||
|
||||
@property
|
||||
def source(self) -> str:
|
||||
return self.value[0]
|
||||
|
||||
@property
|
||||
def short(self) -> str:
|
||||
return self.value[1]
|
||||
|
||||
|
||||
class RoundOrdinal(Enum):
|
||||
"""Interview-round ordinal in a header → slot index 0..3. value is (slot, regex)."""
|
||||
|
||||
FIRST = (0, r"(?:1st|first|01st)")
|
||||
SECOND = (1, r"(?:2nd|second|02nd)")
|
||||
THIRD = (2, r"(?:3rd|third|03rd)")
|
||||
FOURTH = (3, r"(?:4th|fourth|04th)")
|
||||
|
||||
@property
|
||||
def slot(self) -> int:
|
||||
return self.value[0]
|
||||
|
||||
@property
|
||||
def pattern(self) -> str:
|
||||
return self.value[1]
|
||||
|
|
@ -0,0 +1,210 @@
|
|||
"""FormData + SheetImportRun — spreadsheet mirror and background import runs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from sqlalchemy import Column, DateTime, Index, delete, func, or_
|
||||
from sqlalchemy.dialects.postgresql import JSONB
|
||||
from sqlalchemy.ext.asyncio import AsyncSession
|
||||
from sqlmodel import Field, SQLModel, select
|
||||
|
||||
|
||||
def _now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
class FormData(SQLModel, table=True):
|
||||
"""One spreadsheet data row. raw_record keeps the full original header→value map."""
|
||||
|
||||
__tablename__ = "form_data"
|
||||
__table_args__ = (
|
||||
Index("ix_form_data_sheet_row_number", "sheet", "row_number", unique=True),
|
||||
)
|
||||
|
||||
id: int | None = Field(default=None, primary_key=True)
|
||||
sheet: str = Field(nullable=False, index=True)
|
||||
name: str | None = Field(default=None, index=True)
|
||||
degree: str | None = Field(default=None)
|
||||
experience: str | None = Field(default=None)
|
||||
age: int | None = Field(default=None)
|
||||
age_raw: str | None = Field(default=None)
|
||||
family_details: str | None = Field(default=None)
|
||||
|
||||
interview_date: datetime | None = Field(default=None, sa_type=DateTime(timezone=True))
|
||||
interview_by: str | None = Field(default=None)
|
||||
interview_time: str | None = Field(default=None)
|
||||
interview_status: str | None = Field(default=None)
|
||||
interview_notes: str | None = Field(default=None)
|
||||
interview_result: str | None = Field(default=None)
|
||||
|
||||
second_interview_date: datetime | None = Field(default=None, sa_type=DateTime(timezone=True))
|
||||
second_interview_by: str | None = Field(default=None)
|
||||
second_interview_time: str | None = Field(default=None)
|
||||
second_interview_status: str | None = Field(default=None)
|
||||
second_interview_notes: str | None = Field(default=None)
|
||||
second_interview_result: str | None = Field(default=None)
|
||||
|
||||
third_interview_date: datetime | None = Field(default=None, sa_type=DateTime(timezone=True))
|
||||
third_interview_by: str | None = Field(default=None)
|
||||
third_interview_time: str | None = Field(default=None)
|
||||
third_interview_status: str | None = Field(default=None)
|
||||
third_interview_notes: str | None = Field(default=None)
|
||||
third_interview_result: str | None = Field(default=None)
|
||||
|
||||
fourth_interview_date: datetime | None = Field(default=None, sa_type=DateTime(timezone=True))
|
||||
fourth_interview_by: str | None = Field(default=None)
|
||||
fourth_interview_time: str | None = Field(default=None)
|
||||
fourth_interview_status: str | None = Field(default=None)
|
||||
fourth_interview_notes: str | None = Field(default=None)
|
||||
fourth_interview_result: str | None = Field(default=None)
|
||||
|
||||
raw_record: dict | None = Field(default=None, sa_column=Column(JSONB))
|
||||
row_number: int | None = Field(default=None)
|
||||
imported_at: datetime = Field(default_factory=_now, sa_type=DateTime(timezone=True))
|
||||
created_at: datetime = Field(default_factory=_now, sa_type=DateTime(timezone=True))
|
||||
updated_at: datetime = Field(default_factory=_now, sa_type=DateTime(timezone=True))
|
||||
|
||||
@classmethod
|
||||
def _filters(cls, *, sheet=None, search=None):
|
||||
filters = []
|
||||
if sheet:
|
||||
filters.append(cls.sheet == sheet)
|
||||
if search:
|
||||
pattern = f"%{search}%"
|
||||
filters.append(or_(
|
||||
cls.name.ilike(pattern),
|
||||
cls.degree.ilike(pattern),
|
||||
cls.experience.ilike(pattern),
|
||||
cls.interview_by.ilike(pattern),
|
||||
))
|
||||
return filters
|
||||
|
||||
@classmethod
|
||||
async def get_form_data_by_id(cls, session: AsyncSession, record_id):
|
||||
try:
|
||||
rid = int(record_id)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
result = await session.execute(select(cls).where(cls.id == rid))
|
||||
return result.scalars().first()
|
||||
|
||||
@classmethod
|
||||
async def fetch_form_data(cls, session: AsyncSession, *, sheet=None, search=None, top=None, skip=None):
|
||||
statement = select(cls).order_by(cls.sheet, cls.row_number)
|
||||
for clause in cls._filters(sheet=sheet, search=search):
|
||||
statement = statement.where(clause)
|
||||
if skip:
|
||||
statement = statement.offset(skip)
|
||||
if top is not None:
|
||||
statement = statement.limit(top)
|
||||
result = await session.execute(statement)
|
||||
return result.scalars().all()
|
||||
|
||||
@classmethod
|
||||
async def count_form_data(cls, session: AsyncSession, *, sheet=None, search=None):
|
||||
statement = select(func.count()).select_from(cls)
|
||||
for clause in cls._filters(sheet=sheet, search=search):
|
||||
statement = statement.where(clause)
|
||||
result = await session.execute(statement)
|
||||
return result.scalar_one()
|
||||
|
||||
@classmethod
|
||||
async def get_sheet_names(cls, session: AsyncSession):
|
||||
result = await session.execute(
|
||||
select(cls.sheet).distinct().order_by(cls.sheet)
|
||||
)
|
||||
return list(result.scalars().all())
|
||||
|
||||
@classmethod
|
||||
async def delete_by_sheet(cls, session: AsyncSession, sheet: str, *, commit: bool = True):
|
||||
count_result = await session.execute(
|
||||
select(func.count()).select_from(cls).where(cls.sheet == sheet)
|
||||
)
|
||||
deleted = count_result.scalar_one()
|
||||
await session.execute(delete(cls).where(cls.sheet == sheet))
|
||||
if commit:
|
||||
await session.commit()
|
||||
return deleted
|
||||
|
||||
@classmethod
|
||||
async def insert_form_data_bulk(cls, session: AsyncSession, records: list[dict], *, commit: bool = True):
|
||||
rows = [cls(**fields) for fields in records]
|
||||
session.add_all(rows)
|
||||
if commit:
|
||||
await session.commit()
|
||||
return len(rows)
|
||||
|
||||
@classmethod
|
||||
async def replace_sheet(cls, session: AsyncSession, sheet: str, records: list[dict]):
|
||||
"""Delete + insert in one transaction so a mid-insert failure keeps prior rows."""
|
||||
deleted = await cls.delete_by_sheet(session, sheet, commit=False)
|
||||
inserted = await cls.insert_form_data_bulk(session, records, commit=False)
|
||||
await session.commit()
|
||||
return {"deleted": deleted, "inserted": inserted}
|
||||
|
||||
|
||||
class SheetImportRun(SQLModel, table=True):
|
||||
"""One Google Sheet → FormData import job (Taskiq). Survives tab close."""
|
||||
|
||||
__tablename__ = "sheet_import_runs"
|
||||
|
||||
id: uuid.UUID = Field(default_factory=uuid.uuid4, primary_key=True)
|
||||
status: str = Field(default="queued", index=True) # queued|running|completed|failed
|
||||
task_id: str | None = Field(default=None)
|
||||
created_by: uuid.UUID | None = Field(default=None, foreign_key="users.id")
|
||||
tab: str | None = Field(default=None) # None = import all tabs
|
||||
report: dict | None = Field(default=None, sa_column=Column(JSONB))
|
||||
error: str | None = Field(default=None)
|
||||
created_at: datetime = Field(default_factory=_now, sa_type=DateTime(timezone=True))
|
||||
started_at: datetime | None = Field(default=None, sa_type=DateTime(timezone=True))
|
||||
finished_at: datetime | None = Field(default=None, sa_type=DateTime(timezone=True))
|
||||
|
||||
@staticmethod
|
||||
def _as_uuid(record_id) -> uuid.UUID | None:
|
||||
if record_id in (None, ""):
|
||||
return None
|
||||
try:
|
||||
return uuid.UUID(str(record_id))
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
@classmethod
|
||||
async def get_by_id(cls, session: AsyncSession, record_id):
|
||||
uid = cls._as_uuid(record_id)
|
||||
if uid is None:
|
||||
return None
|
||||
result = await session.execute(select(cls).where(cls.id == uid))
|
||||
return result.scalars().first()
|
||||
|
||||
@classmethod
|
||||
async def get_active(cls, session: AsyncSession):
|
||||
result = await session.execute(
|
||||
select(cls)
|
||||
.where(cls.status.in_(("queued", "running")))
|
||||
.order_by(cls.created_at.desc())
|
||||
)
|
||||
return result.scalars().first()
|
||||
|
||||
@classmethod
|
||||
async def insert_run(cls, session: AsyncSession, fields: dict, *, commit: bool = True):
|
||||
row = cls(**fields)
|
||||
session.add(row)
|
||||
if commit:
|
||||
await session.commit()
|
||||
await session.refresh(row)
|
||||
return row
|
||||
|
||||
@classmethod
|
||||
async def update_run(cls, session: AsyncSession, record_id, fields: dict, *, commit: bool = True):
|
||||
row = await cls.get_by_id(session, record_id)
|
||||
if not row:
|
||||
return None
|
||||
for key, value in fields.items():
|
||||
setattr(row, key, value)
|
||||
session.add(row)
|
||||
if commit:
|
||||
await session.commit()
|
||||
await session.refresh(row)
|
||||
return row
|
||||
|
|
@ -0,0 +1,536 @@
|
|||
"""Google Sheets helpers — credential loading, retrying API calls, row/record shaping.
|
||||
|
||||
No FastAPI imports here by house rule: this module raises its own SheetsServiceError
|
||||
family and lets g_sheet/views.py translate that into HTTPException.
|
||||
|
||||
Auth reuses the credentials already on disk (authorized_user ADC + a valid refresh
|
||||
token). Nothing here launches a browser, runs InstalledAppFlow, or reads stdin.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import random
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from google.auth import default as google_auth_default
|
||||
from google.auth.transport.requests import Request
|
||||
from googleapiclient.discovery import build
|
||||
from googleapiclient.errors import HttpError
|
||||
|
||||
from g_sheet.enums import (
|
||||
ConductedByAlias,
|
||||
DateFormat,
|
||||
DateTimeSeparator,
|
||||
FIELD_ALIAS_ENUMS,
|
||||
FormDataField,
|
||||
MonthNormalisation,
|
||||
NotesToken,
|
||||
RoundByColumn,
|
||||
RoundDateColumn,
|
||||
RoundNotesColumn,
|
||||
RoundOrdinal,
|
||||
RoundResultColumn,
|
||||
RoundRole,
|
||||
RoundStatusColumn,
|
||||
RoundTimeColumn,
|
||||
)
|
||||
|
||||
load_dotenv()
|
||||
|
||||
logger=logging.getLogger("g_sheet.plugins")
|
||||
|
||||
# backend/ — GOOGLE_APPLICATION_CREDENTIALS is stored relative to it ("credentials/...").
|
||||
ROOT=Path(__file__).resolve().parent.parent
|
||||
|
||||
SCOPES=[
|
||||
"https://www.googleapis.com/auth/spreadsheets",
|
||||
"https://www.googleapis.com/auth/drive",
|
||||
]
|
||||
|
||||
SPREADSHEET_ID=os.getenv("SPREADSHEET_ID")
|
||||
SPREADSHEET_NAME=os.getenv("SPREADSHEET_NAME")
|
||||
SPREADSHEET_URL=os.getenv("SPREADSHEET_URL")
|
||||
GOOGLE_APPLICATION_CREDENTIALS=os.getenv("GOOGLE_APPLICATION_CREDENTIALS")
|
||||
|
||||
# 429 and 5xx are transient; every other 4xx is a bad request that a retry repeats.
|
||||
RETRY_ATTEMPTS=3
|
||||
RETRY_BASE_DELAY=0.5
|
||||
RETRY_MAX_DELAY=8.0
|
||||
RETRYABLE_STATUSES={429,500,502,503,504}
|
||||
|
||||
|
||||
class SheetsServiceError(Exception):
|
||||
"""Base for every failure this domain raises. Carries an HTTP-ish status code."""
|
||||
|
||||
status_code=500
|
||||
|
||||
def __init__(self,message,status_code=None):
|
||||
super().__init__(message)
|
||||
self.message=message
|
||||
if status_code is not None:
|
||||
self.status_code=status_code
|
||||
|
||||
|
||||
class SheetsAuthError(SheetsServiceError):
|
||||
"""Credentials missing, unreadable, or rejected by Google."""
|
||||
|
||||
status_code=401
|
||||
|
||||
|
||||
class SheetsApiError(SheetsServiceError):
|
||||
"""The Sheets API answered with an error. status_code is Google's own."""
|
||||
|
||||
status_code=502
|
||||
|
||||
|
||||
def resolve_credentials_path(credentials_path=None):
|
||||
"""Absolute path to the ADC json. Relative values resolve against backend/.
|
||||
|
||||
The service may be imported from any working directory, so a bare
|
||||
"credentials/application_default_credentials.json" must not depend on cwd.
|
||||
"""
|
||||
raw=credentials_path or GOOGLE_APPLICATION_CREDENTIALS
|
||||
if not raw:
|
||||
return None
|
||||
path=Path(raw)
|
||||
if not path.is_absolute():
|
||||
path=ROOT/path
|
||||
return path
|
||||
|
||||
|
||||
def load_credentials(credentials_path=None,scopes=None):
|
||||
"""Build scoped ADC credentials and refresh them once. Never prompts."""
|
||||
path=resolve_credentials_path(credentials_path)
|
||||
if path is not None:
|
||||
if not path.exists():
|
||||
raise SheetsAuthError(f"Google credentials file not found: {path.name}")
|
||||
os.environ["GOOGLE_APPLICATION_CREDENTIALS"]=str(path)
|
||||
try:
|
||||
credentials,_=google_auth_default(scopes=scopes or SCOPES)
|
||||
credentials.refresh(Request())
|
||||
except SheetsServiceError:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise SheetsAuthError(f"Google credential refresh failed: {e}")
|
||||
return credentials
|
||||
|
||||
|
||||
def ensure_fresh(credentials):
|
||||
"""Refresh only when the token has actually gone stale — not on every call."""
|
||||
if credentials is None:
|
||||
raise SheetsAuthError("Google credentials are not initialised")
|
||||
if credentials.valid and not credentials.expired:
|
||||
return credentials
|
||||
try:
|
||||
credentials.refresh(Request())
|
||||
except Exception as e:
|
||||
raise SheetsAuthError(f"Google credential refresh failed: {e}")
|
||||
return credentials
|
||||
|
||||
|
||||
def build_sheets_client(credentials):
|
||||
"""Sheets v4 client. cache_discovery=False — the file cache warns under threads."""
|
||||
try:
|
||||
return build("sheets","v4",credentials=credentials,cache_discovery=False)
|
||||
except Exception as e:
|
||||
raise SheetsApiError(f"Could not build the Sheets client: {e}")
|
||||
|
||||
|
||||
def _status_of(error):
|
||||
status=getattr(getattr(error,"resp",None),"status",None)
|
||||
if status is None:
|
||||
status=getattr(error,"status_code",None)
|
||||
try:
|
||||
return int(status)
|
||||
except (TypeError,ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _reason_of(error):
|
||||
"""Google's message without the response body, so nothing sensitive leaks out."""
|
||||
try:
|
||||
return error._get_reason().strip()
|
||||
except Exception:
|
||||
return str(error)
|
||||
|
||||
|
||||
def execute(request,description="sheets request"):
|
||||
"""Run a googleapiclient request with jittered exponential backoff.
|
||||
|
||||
Retries 429 and 5xx up to RETRY_ATTEMPTS; every other HttpError raises straight
|
||||
away as SheetsApiError carrying Google's status code.
|
||||
"""
|
||||
delay=RETRY_BASE_DELAY
|
||||
last_error=None
|
||||
for attempt in range(1,RETRY_ATTEMPTS+1):
|
||||
try:
|
||||
return request.execute()
|
||||
except HttpError as e:
|
||||
status=_status_of(e)
|
||||
reason=_reason_of(e)
|
||||
last_error=SheetsApiError(f"{description} failed: {reason}",status or 502)
|
||||
if status not in RETRYABLE_STATUSES or attempt==RETRY_ATTEMPTS:
|
||||
raise last_error
|
||||
sleep_for=min(delay,RETRY_MAX_DELAY)+random.uniform(0,RETRY_BASE_DELAY)
|
||||
logger.warning(
|
||||
"%s got %s, retry %s/%s in %.2fs",
|
||||
description,status,attempt,RETRY_ATTEMPTS,sleep_for,
|
||||
)
|
||||
time.sleep(sleep_for)
|
||||
delay*=2
|
||||
except SheetsServiceError:
|
||||
raise
|
||||
except Exception as e:
|
||||
raise SheetsApiError(f"{description} failed: {e}")
|
||||
raise last_error
|
||||
|
||||
|
||||
def quote_tab(tab,cell_range=None):
|
||||
"""A1 target for a tab whose name may contain spaces or quotes."""
|
||||
safe=str(tab).replace("'","''")
|
||||
if cell_range:
|
||||
return f"'{safe}'!{cell_range}"
|
||||
return f"'{safe}'"
|
||||
|
||||
|
||||
def normalise_headers(header_row):
|
||||
"""First row -> unique, non-empty column keys.
|
||||
|
||||
Blank cells become column_{i}; a repeated header keeps its first spelling and the
|
||||
later ones get _1, _2 so no key silently overwrites another.
|
||||
"""
|
||||
headers=[]
|
||||
seen={}
|
||||
for index,raw in enumerate(header_row):
|
||||
name=str(raw).strip() if raw is not None else ""
|
||||
if not name:
|
||||
name=f"column_{index}"
|
||||
count=seen.get(name,0)
|
||||
seen[name]=count+1
|
||||
headers.append(name if count==0 else f"{name}_{count}")
|
||||
return headers
|
||||
|
||||
|
||||
def rows_to_records(rows):
|
||||
"""Sheet rows -> list of dicts keyed by the header row.
|
||||
|
||||
Sheets truncates trailing empties, so short rows are padded to header width.
|
||||
Fully blank rows are dropped rather than emitted as all-empty records.
|
||||
"""
|
||||
if not rows:
|
||||
return []
|
||||
headers=normalise_headers(rows[0])
|
||||
records=[]
|
||||
for row in rows[1:]:
|
||||
values=[str(cell) if cell is not None else "" for cell in row]
|
||||
if not any(value.strip() for value in values):
|
||||
continue
|
||||
if len(values)<len(headers):
|
||||
values=values+[""]*(len(headers)-len(values))
|
||||
records.append(dict(zip(headers,values[:len(headers)])))
|
||||
return records
|
||||
|
||||
|
||||
def stringify_rows(rows):
|
||||
"""Normalise raw values() output into list[list[str]] with no None holes."""
|
||||
return [[str(cell) if cell is not None else "" for cell in row] for row in rows or []]
|
||||
|
||||
|
||||
# -- FormData mapping ------------------------------------------------------
|
||||
|
||||
BY_FIELDS=RoundByColumn.ordered()
|
||||
DATE_FIELDS=RoundDateColumn.ordered()
|
||||
TIME_FIELDS=RoundTimeColumn.ordered()
|
||||
STATUS_FIELDS=RoundStatusColumn.ordered()
|
||||
NOTES_FIELDS=RoundNotesColumn.ordered()
|
||||
RESULT_FIELDS=RoundResultColumn.ordered()
|
||||
|
||||
_ORDINAL_PATTERNS=tuple(
|
||||
(member.slot,re.compile(rf"\b{member.pattern}\b",re.I))
|
||||
for member in RoundOrdinal
|
||||
)
|
||||
|
||||
_TIME_RE=re.compile(r"(\d{1,2}:\d{2}\s*(?:[AaPp][Mm])?)")
|
||||
_DAY_ORDINAL_RE=re.compile(r"\b(\d+)(st|nd|rd|th)\b",re.I)
|
||||
_DIGIT_RE=re.compile(r"\d")
|
||||
_AGE_RE=re.compile(r"\d+")
|
||||
|
||||
|
||||
def canonical_header(h):
|
||||
"""Lower, collapse whitespace (incl. embedded newlines), strip _N and (tails)."""
|
||||
text=str(h or "").replace("\n"," ").replace("\r"," ")
|
||||
text=re.sub(r"\s+"," ",text).strip().lower()
|
||||
text=re.sub(r"_\d+$","",text)
|
||||
text=re.sub(r"\s*\([^)]*\)\s*$","",text).strip()
|
||||
return text
|
||||
|
||||
|
||||
def match_field(h):
|
||||
"""Map a sheet header to a FormDataField, or None.
|
||||
|
||||
Exact alias first, then startswith. No fuzzy matching — dirty headers mislabel
|
||||
more often than they rescue, and a miss is non-fatal (value stays in JSONB).
|
||||
"""
|
||||
canon=canonical_header(h)
|
||||
if not canon:
|
||||
return None
|
||||
for field,alias_enum in FIELD_ALIAS_ENUMS.items():
|
||||
if alias_enum.has(canon):
|
||||
return field
|
||||
for field,alias_enum in FIELD_ALIAS_ENUMS.items():
|
||||
for alias in alias_enum:
|
||||
if canon.startswith(alias.value):
|
||||
return field
|
||||
return None
|
||||
|
||||
|
||||
def resolve_name(record,headers):
|
||||
"""Candidate name: alias match, else column A (headers[0]) — always the name."""
|
||||
for header in headers:
|
||||
if match_field(header)==FormDataField.NAME:
|
||||
value=record.get(header)
|
||||
if value is not None and str(value).strip():
|
||||
return str(value).strip()
|
||||
if headers:
|
||||
value=record.get(headers[0])
|
||||
if value is not None and str(value).strip():
|
||||
return str(value).strip()
|
||||
return None
|
||||
|
||||
|
||||
def _classify_round_role(canon):
|
||||
"""RoundRole for a canonical header, or None for unrecognised headers."""
|
||||
if not canon:
|
||||
return None
|
||||
if ConductedByAlias.contained_in(canon) or ConductedByAlias.has(canon):
|
||||
return RoundRole.BY
|
||||
if canon.startswith(ConductedByAlias.CONDUCTED.value):
|
||||
return RoundRole.BY
|
||||
if RoundRole.RESULT.value in canon:
|
||||
return RoundRole.RESULT
|
||||
if RoundRole.STATUS.value in canon:
|
||||
return RoundRole.STATUS
|
||||
if NotesToken.contained_in(canon):
|
||||
return RoundRole.NOTES
|
||||
if RoundRole.DATE.value in canon:
|
||||
return RoundRole.DATE
|
||||
return None
|
||||
|
||||
|
||||
def _extract_ordinal(canon):
|
||||
for slot,pattern in _ORDINAL_PATTERNS:
|
||||
if pattern.search(canon):
|
||||
return slot
|
||||
return None
|
||||
|
||||
|
||||
def resolve_round_columns(headers):
|
||||
"""Positional interview-round map: scan left→right into four slots.
|
||||
|
||||
Ordinal in the header (`2nd`, `second`) pins the slot; otherwise the first free
|
||||
slot for that role is taken, never moving backwards. A fifth Results_4 stays
|
||||
unmapped (JSONB). Literal-date headers like `19-Feb-2026` classify as nothing.
|
||||
"""
|
||||
slots=[{role:None for role in RoundRole} for _ in range(4)]
|
||||
cursor={role:0 for role in RoundRole}
|
||||
|
||||
for header in headers:
|
||||
canon=canonical_header(header)
|
||||
role=_classify_round_role(canon)
|
||||
if role is None:
|
||||
continue
|
||||
ordinal=_extract_ordinal(canon)
|
||||
if ordinal is not None:
|
||||
if slots[ordinal][role] is None:
|
||||
slots[ordinal][role]=header
|
||||
continue
|
||||
start=cursor[role]
|
||||
chosen=None
|
||||
for index in range(start,4):
|
||||
if slots[index][role] is None:
|
||||
chosen=index
|
||||
break
|
||||
if chosen is None:
|
||||
continue
|
||||
slots[chosen][role]=header
|
||||
cursor[role]=chosen+1
|
||||
return slots
|
||||
|
||||
|
||||
def _normalise_month_spellings(text):
|
||||
"""strptime %b rejects `Sept`; expand common sheet spellings first."""
|
||||
lowered=text.lower()
|
||||
for member in MonthNormalisation:
|
||||
if member.source in lowered:
|
||||
text=re.sub(member.source,member.short,text,flags=re.I)
|
||||
lowered=text.lower()
|
||||
return text
|
||||
|
||||
|
||||
def parse_date(value):
|
||||
"""Tolerant date parse → aware UTC datetime, or None. Never raises."""
|
||||
if value is None:
|
||||
return None
|
||||
text=str(value).strip()
|
||||
if not text or not _DIGIT_RE.search(text):
|
||||
return None
|
||||
|
||||
date_part=text
|
||||
for sep in DateTimeSeparator:
|
||||
if sep.value in text:
|
||||
date_part=text.split(sep.value,1)[0].strip()
|
||||
break
|
||||
# Drop a trailing time when joined without a dash: "6th Nov 2025 7:30 PM"
|
||||
time_match=_TIME_RE.search(date_part)
|
||||
if time_match and time_match.start()>0:
|
||||
date_part=date_part[:time_match.start()].strip(" ,;-")
|
||||
|
||||
date_part=_DAY_ORDINAL_RE.sub(r"\1",date_part)
|
||||
date_part=_normalise_month_spellings(date_part)
|
||||
date_part=re.sub(r"\s+"," ",date_part).strip(" ,;")
|
||||
|
||||
for fmt in DateFormat:
|
||||
try:
|
||||
return datetime.strptime(date_part,fmt.value).replace(tzinfo=timezone.utc)
|
||||
except ValueError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def parse_date_time(value):
|
||||
"""(datetime|None, time_string|None) — fills *_time for the cells that carry one."""
|
||||
parsed=parse_date(value)
|
||||
if value is None:
|
||||
return parsed,None
|
||||
text=str(value).strip()
|
||||
match=_TIME_RE.search(text)
|
||||
time_str=match.group(1).strip() if match else None
|
||||
return parsed,time_str
|
||||
|
||||
|
||||
def parse_age(value):
|
||||
"""(int|None, raw|None) — first digit run if 0 < n < 100, always keep the raw."""
|
||||
if value is None:
|
||||
return None,None
|
||||
raw=str(value).strip()
|
||||
if not raw:
|
||||
return None,None
|
||||
match=_AGE_RE.search(raw)
|
||||
if not match:
|
||||
return None,raw
|
||||
number=int(match.group())
|
||||
if 0<number<100:
|
||||
return number,raw
|
||||
return None,raw
|
||||
|
||||
|
||||
def _blank_to_none(value):
|
||||
if value is None:
|
||||
return None
|
||||
text=str(value).strip()
|
||||
return text if text else None
|
||||
|
||||
|
||||
def map_record_to_form_data(sheet,record,headers,row_number):
|
||||
"""Pure row mapper → kwargs dict for FormData(**...)."""
|
||||
rounds=resolve_round_columns(headers)
|
||||
mapped={
|
||||
"sheet":sheet,
|
||||
"row_number":row_number,
|
||||
"raw_record":dict(record),
|
||||
"name":_blank_to_none(resolve_name(record,headers)),
|
||||
"degree":None,
|
||||
"experience":None,
|
||||
"age":None,
|
||||
"age_raw":None,
|
||||
"family_details":None,
|
||||
}
|
||||
for field in BY_FIELDS+TIME_FIELDS+STATUS_FIELDS+NOTES_FIELDS+RESULT_FIELDS:
|
||||
mapped[field]=None
|
||||
for field in DATE_FIELDS:
|
||||
mapped[field]=None
|
||||
|
||||
for header,value in record.items():
|
||||
field=match_field(header)
|
||||
if field==FormDataField.DEGREE:
|
||||
mapped["degree"]=_blank_to_none(value)
|
||||
elif field==FormDataField.EXPERIENCE:
|
||||
mapped["experience"]=_blank_to_none(value)
|
||||
elif field==FormDataField.AGE:
|
||||
age,age_raw=parse_age(value)
|
||||
mapped["age"]=age
|
||||
mapped["age_raw"]=age_raw
|
||||
elif field==FormDataField.FAMILY_DETAILS:
|
||||
mapped["family_details"]=_blank_to_none(value)
|
||||
|
||||
for index,slot in enumerate(rounds):
|
||||
if slot.get(RoundRole.DATE):
|
||||
dt,tm=parse_date_time(record.get(slot[RoundRole.DATE]))
|
||||
mapped[DATE_FIELDS[index]]=dt
|
||||
mapped[TIME_FIELDS[index]]=tm
|
||||
if slot.get(RoundRole.BY):
|
||||
mapped[BY_FIELDS[index]]=_blank_to_none(record.get(slot[RoundRole.BY]))
|
||||
if slot.get(RoundRole.STATUS):
|
||||
mapped[STATUS_FIELDS[index]]=_blank_to_none(record.get(slot[RoundRole.STATUS]))
|
||||
if slot.get(RoundRole.NOTES):
|
||||
mapped[NOTES_FIELDS[index]]=_blank_to_none(record.get(slot[RoundRole.NOTES]))
|
||||
if slot.get(RoundRole.RESULT):
|
||||
mapped[RESULT_FIELDS[index]]=_blank_to_none(record.get(slot[RoundRole.RESULT]))
|
||||
|
||||
return mapped
|
||||
|
||||
|
||||
def collect_unmapped_headers(headers):
|
||||
"""Headers that are neither a typed alias nor claimed by a round slot.
|
||||
|
||||
`title` aliases are included — they have no FormData column and live in JSONB.
|
||||
"""
|
||||
rounds=resolve_round_columns(headers)
|
||||
claimed=set()
|
||||
for slot in rounds:
|
||||
for role in RoundRole:
|
||||
if slot.get(role):
|
||||
claimed.add(slot[role])
|
||||
unmapped=[]
|
||||
for header in headers:
|
||||
if header in claimed:
|
||||
continue
|
||||
field=match_field(header)
|
||||
if field is None or field==FormDataField.TITLE:
|
||||
unmapped.append(header)
|
||||
return unmapped
|
||||
|
||||
|
||||
def import_row_stats(mapped_rows,headers):
|
||||
"""Aggregate parse diagnostics for an import report."""
|
||||
dates_parsed=0
|
||||
dates_unparsed=0
|
||||
ages_parsed=0
|
||||
for row in mapped_rows:
|
||||
raw=row.get("raw_record") or {}
|
||||
rounds=resolve_round_columns(headers)
|
||||
for index,slot in enumerate(rounds):
|
||||
header=slot.get(RoundRole.DATE)
|
||||
if not header:
|
||||
continue
|
||||
cell=raw.get(header)
|
||||
if cell is None or not str(cell).strip():
|
||||
continue
|
||||
if row.get(DATE_FIELDS[index]) is not None:
|
||||
dates_parsed+=1
|
||||
elif _DIGIT_RE.search(str(cell)):
|
||||
dates_unparsed+=1
|
||||
if row.get("age") is not None:
|
||||
ages_parsed+=1
|
||||
return {
|
||||
"dates_parsed":dates_parsed,
|
||||
"dates_unparsed":dates_unparsed,
|
||||
"ages_parsed":ages_parsed,
|
||||
"unmapped_headers":collect_unmapped_headers(headers),
|
||||
}
|
||||
|
|
@ -0,0 +1,188 @@
|
|||
"""Google Sheets response shapes. Plain dicts only — no DB, no Depends."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def serialize_metadata(payload: dict) -> dict:
|
||||
"""spreadsheets.get response -> the spreadsheet header the UI renders."""
|
||||
properties = payload.get("properties") or {}
|
||||
return {
|
||||
"spreadsheet_id": payload.get("spreadsheetId"),
|
||||
"title": properties.get("title"),
|
||||
"locale": properties.get("locale"),
|
||||
"time_zone": properties.get("timeZone"),
|
||||
"url": payload.get("spreadsheetUrl"),
|
||||
"tabs": [serialize_tab(sheet) for sheet in payload.get("sheets") or []],
|
||||
}
|
||||
|
||||
|
||||
def serialize_tab(sheet: dict) -> dict:
|
||||
"""One entry of spreadsheets.get -> tab name plus its grid size."""
|
||||
properties = sheet.get("properties") or {}
|
||||
grid = properties.get("gridProperties") or {}
|
||||
return {
|
||||
"title": properties.get("title"),
|
||||
"sheet_id": properties.get("sheetId"),
|
||||
"index": properties.get("index"),
|
||||
"row_count": grid.get("rowCount"),
|
||||
"column_count": grid.get("columnCount"),
|
||||
}
|
||||
|
||||
|
||||
def serialize_values(tab: str, cell_range: str | None, rows: list[list[str]]) -> dict:
|
||||
"""Raw rows -> the read_range envelope."""
|
||||
return {
|
||||
"tab": tab,
|
||||
"range": cell_range,
|
||||
"rows": rows,
|
||||
"row_count": len(rows),
|
||||
}
|
||||
|
||||
|
||||
def serialize_records(tab: str, records: list[dict]) -> dict:
|
||||
"""Header-mapped rows -> the read_records envelope."""
|
||||
return {
|
||||
"tab": tab,
|
||||
"records": records,
|
||||
"total": len(records),
|
||||
"headers": list(records[0].keys()) if records else [],
|
||||
}
|
||||
|
||||
|
||||
def serialize_append(tab: str, payload: dict) -> dict:
|
||||
"""values.append response -> what was written and where."""
|
||||
updates = payload.get("updates") or {}
|
||||
return {
|
||||
"tab": tab,
|
||||
"spreadsheet_id": payload.get("spreadsheetId"),
|
||||
"updated_range": updates.get("updatedRange"),
|
||||
"updated_rows": updates.get("updatedRows", 0),
|
||||
"updated_columns": updates.get("updatedColumns", 0),
|
||||
"updated_cells": updates.get("updatedCells", 0),
|
||||
}
|
||||
|
||||
|
||||
def serialize_update(tab: str, payload: dict) -> dict:
|
||||
"""values.update response -> the same shape as an append result."""
|
||||
return {
|
||||
"tab": tab,
|
||||
"spreadsheet_id": payload.get("spreadsheetId"),
|
||||
"updated_range": payload.get("updatedRange"),
|
||||
"updated_rows": payload.get("updatedRows", 0),
|
||||
"updated_columns": payload.get("updatedColumns", 0),
|
||||
"updated_cells": payload.get("updatedCells", 0),
|
||||
}
|
||||
|
||||
|
||||
def serialize_clear(tab: str, payload: dict) -> dict:
|
||||
"""values.clear response -> the cleared range."""
|
||||
return {
|
||||
"tab": tab,
|
||||
"spreadsheet_id": payload.get("spreadsheetId"),
|
||||
"cleared_range": payload.get("clearedRange"),
|
||||
}
|
||||
|
||||
|
||||
def serialize_health(ok: bool, detail: str, tabs: list[str] | None = None) -> dict:
|
||||
"""health_check result. Returned on failure too — this one never raises."""
|
||||
return {
|
||||
"status": "ok" if ok else "error",
|
||||
"detail": detail,
|
||||
"tabs": tabs or [],
|
||||
"tab_count": len(tabs or []),
|
||||
}
|
||||
|
||||
|
||||
def _iso(value):
|
||||
return value.isoformat() if value is not None else None
|
||||
|
||||
|
||||
def serialize_form_data(row) -> dict:
|
||||
"""FormData ORM row → API dict, including raw_record."""
|
||||
return {
|
||||
"id": row.id,
|
||||
"sheet": row.sheet,
|
||||
"name": row.name,
|
||||
"degree": row.degree,
|
||||
"experience": row.experience,
|
||||
"age": row.age,
|
||||
"age_raw": row.age_raw,
|
||||
"family_details": row.family_details,
|
||||
"interview_date": _iso(row.interview_date),
|
||||
"interview_by": row.interview_by,
|
||||
"interview_time": row.interview_time,
|
||||
"interview_status": row.interview_status,
|
||||
"interview_notes": row.interview_notes,
|
||||
"interview_result": row.interview_result,
|
||||
"second_interview_date": _iso(row.second_interview_date),
|
||||
"second_interview_by": row.second_interview_by,
|
||||
"second_interview_time": row.second_interview_time,
|
||||
"second_interview_status": row.second_interview_status,
|
||||
"second_interview_notes": row.second_interview_notes,
|
||||
"second_interview_result": row.second_interview_result,
|
||||
"third_interview_date": _iso(row.third_interview_date),
|
||||
"third_interview_by": row.third_interview_by,
|
||||
"third_interview_time": row.third_interview_time,
|
||||
"third_interview_status": row.third_interview_status,
|
||||
"third_interview_notes": row.third_interview_notes,
|
||||
"third_interview_result": row.third_interview_result,
|
||||
"fourth_interview_date": _iso(row.fourth_interview_date),
|
||||
"fourth_interview_by": row.fourth_interview_by,
|
||||
"fourth_interview_time": row.fourth_interview_time,
|
||||
"fourth_interview_status": row.fourth_interview_status,
|
||||
"fourth_interview_notes": row.fourth_interview_notes,
|
||||
"fourth_interview_result": row.fourth_interview_result,
|
||||
"raw_record": row.raw_record,
|
||||
"row_number": row.row_number,
|
||||
"imported_at": _iso(row.imported_at),
|
||||
"created_at": _iso(row.created_at),
|
||||
"updated_at": _iso(row.updated_at),
|
||||
}
|
||||
|
||||
|
||||
def serialize_import(report: dict) -> dict:
|
||||
"""Per-tab import report."""
|
||||
return {
|
||||
"tab": report.get("tab"),
|
||||
"rows_read": report.get("rows_read", 0),
|
||||
"inserted": report.get("inserted", 0),
|
||||
"deleted": report.get("deleted", 0),
|
||||
"dates_parsed": report.get("dates_parsed", 0),
|
||||
"dates_unparsed": report.get("dates_unparsed", 0),
|
||||
"ages_parsed": report.get("ages_parsed", 0),
|
||||
"unmapped_headers": report.get("unmapped_headers") or [],
|
||||
"error": report.get("error"),
|
||||
}
|
||||
|
||||
|
||||
def serialize_import_all(reports: list[dict]) -> dict:
|
||||
"""Aggregate of per-tab reports from import_all."""
|
||||
ok=[r for r in reports if not r.get("error")]
|
||||
failed=[r for r in reports if r.get("error")]
|
||||
return {
|
||||
"tabs": len(reports),
|
||||
"succeeded": len(ok),
|
||||
"failed": len(failed),
|
||||
"inserted": sum(r.get("inserted", 0) for r in ok),
|
||||
"deleted": sum(r.get("deleted", 0) for r in ok),
|
||||
"reports": [serialize_import(r) for r in reports],
|
||||
}
|
||||
|
||||
|
||||
def serialize_sheet_summary(sheets: list[str]) -> dict:
|
||||
return {"sheets": sheets, "total": len(sheets)}
|
||||
|
||||
|
||||
def serialize_import_run(row) -> dict:
|
||||
return {
|
||||
"id": str(row.id),
|
||||
"status": row.status,
|
||||
"task_id": row.task_id,
|
||||
"created_by": str(row.created_by) if row.created_by else None,
|
||||
"tab": row.tab,
|
||||
"report": row.report,
|
||||
"error": row.error,
|
||||
"created_at": _iso(row.created_at),
|
||||
"started_at": _iso(row.started_at),
|
||||
"finished_at": _iso(row.finished_at),
|
||||
}
|
||||
|
|
@ -0,0 +1,93 @@
|
|||
"""Google Sheet → FormData import Taskiq tasks (shared inbox worker stream)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime,timezone
|
||||
|
||||
import redis.asyncio as redis
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from db_setup import session_scope
|
||||
from g_sheet.models import SheetImportRun
|
||||
from g_sheet.views import Sheet
|
||||
from taskiq_management.broker_setup import MAX_RETRIES,RETRY_DELAY,broker
|
||||
from taskiq_management.middleware import PermanentTaskError
|
||||
|
||||
load_dotenv()
|
||||
|
||||
logger=logging.getLogger("g_sheet.tasks")
|
||||
REDIS_URL=os.getenv("REDIS_URL","redis://localhost:6379/0")
|
||||
_LOCK_KEY="g_sheet:import:lock"
|
||||
_LOCK_TTL=3600
|
||||
|
||||
|
||||
async def _fail(run_id:str,error:str) -> dict:
|
||||
async with session_scope() as session:
|
||||
await SheetImportRun.update_run(session,run_id,{
|
||||
"status":"failed",
|
||||
"error":error,
|
||||
"finished_at":datetime.now(timezone.utc),
|
||||
})
|
||||
return {"status":"failed","error":error}
|
||||
|
||||
|
||||
@broker.task(
|
||||
task_name="g_sheet.import_sheets",
|
||||
retry_on_error=True,
|
||||
max_retries=MAX_RETRIES,
|
||||
delay=RETRY_DELAY,
|
||||
)
|
||||
async def import_sheets(run_id:str) -> dict:
|
||||
if not run_id or not str(run_id).strip():
|
||||
raise PermanentTaskError("run_id is required")
|
||||
run_id=str(run_id).strip()
|
||||
|
||||
client=redis.from_url(REDIS_URL,decode_responses=True)
|
||||
try:
|
||||
acquired=await client.set(_LOCK_KEY,run_id,nx=True,ex=_LOCK_TTL)
|
||||
if not acquired:
|
||||
return await _fail(run_id,"another sheet import is already running")
|
||||
|
||||
try:
|
||||
async with session_scope() as session:
|
||||
row=await SheetImportRun.get_by_id(session,run_id)
|
||||
if not row:
|
||||
raise PermanentTaskError(f"import run {run_id} not found")
|
||||
await SheetImportRun.update_run(session,run_id,{
|
||||
"status":"running",
|
||||
"started_at":datetime.now(timezone.utc),
|
||||
"error":None,
|
||||
})
|
||||
tab=row.tab
|
||||
|
||||
async with session_scope() as session:
|
||||
service=Sheet(session=session)
|
||||
try:
|
||||
if tab:
|
||||
report=await service.import_sheet(tab)
|
||||
else:
|
||||
report=await service.import_all()
|
||||
except Exception as e:
|
||||
logger.exception("sheet import failed for run %s",run_id)
|
||||
# Bad tab names and permanent Sheets 4xx — do not burn retries.
|
||||
from fastapi import HTTPException
|
||||
if isinstance(e,HTTPException) and e.status_code in (400,404,422):
|
||||
await _fail(run_id,str(e.detail))
|
||||
raise PermanentTaskError(str(e.detail)) from e
|
||||
return await _fail(run_id,str(e))
|
||||
|
||||
await SheetImportRun.update_run(session,run_id,{
|
||||
"status":"completed",
|
||||
"report":report,
|
||||
"error":None,
|
||||
"finished_at":datetime.now(timezone.utc),
|
||||
})
|
||||
return {"status":"completed","report":report}
|
||||
finally:
|
||||
current=await client.get(_LOCK_KEY)
|
||||
if current==run_id:
|
||||
await client.delete(_LOCK_KEY)
|
||||
finally:
|
||||
await client.aclose()
|
||||
|
|
@ -0,0 +1,334 @@
|
|||
"""Google Sheets service — business logic for the g_sheet domain.
|
||||
|
||||
The Google client is blocking, so every call goes through asyncio.to_thread rather
|
||||
than stalling the event loop. Client construction is lazy and guarded by a lock so
|
||||
concurrent requests build it exactly once.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import threading
|
||||
from datetime import datetime,timezone
|
||||
|
||||
from fastapi import HTTPException
|
||||
|
||||
from g_sheet.plugins import (
|
||||
SCOPES,
|
||||
SPREADSHEET_ID,
|
||||
SPREADSHEET_NAME,
|
||||
SPREADSHEET_URL,
|
||||
SheetsServiceError,
|
||||
build_sheets_client,
|
||||
ensure_fresh,
|
||||
execute,
|
||||
import_row_stats,
|
||||
load_credentials,
|
||||
map_record_to_form_data,
|
||||
quote_tab,
|
||||
rows_to_records,
|
||||
stringify_rows,
|
||||
)
|
||||
from g_sheet.models import FormData,SheetImportRun
|
||||
from g_sheet.serializers import (
|
||||
serialize_append,
|
||||
serialize_clear,
|
||||
serialize_form_data,
|
||||
serialize_health,
|
||||
serialize_import,
|
||||
serialize_import_all,
|
||||
serialize_import_run,
|
||||
serialize_metadata,
|
||||
serialize_records,
|
||||
serialize_sheet_summary,
|
||||
serialize_update,
|
||||
serialize_values,
|
||||
)
|
||||
|
||||
logger=logging.getLogger("g_sheet.views")
|
||||
|
||||
|
||||
class Sheet:
|
||||
def __init__(self,session=None,spreadsheet_id=None,credentials_path=None,scopes=None):
|
||||
self.session=session
|
||||
self.spreadsheet_id=spreadsheet_id or SPREADSHEET_ID
|
||||
self.spreadsheet_name=SPREADSHEET_NAME
|
||||
self.spreadsheet_url=SPREADSHEET_URL
|
||||
self.credentials_path=credentials_path
|
||||
self.scopes=scopes or SCOPES
|
||||
self.credentials=None
|
||||
self.client=None
|
||||
self._lock=threading.Lock()
|
||||
|
||||
def _require_session(self):
|
||||
if self.session is None:
|
||||
raise HTTPException(status_code=500,detail="Database session is required")
|
||||
return self.session
|
||||
|
||||
# -- client ------------------------------------------------------------
|
||||
|
||||
def _connect(self):
|
||||
"""Build credentials + client once, then keep refreshing the same token.
|
||||
|
||||
Double-checked under the lock: two requests racing here must not each build
|
||||
their own client.
|
||||
"""
|
||||
if self.client is not None:
|
||||
return ensure_fresh(self.credentials) and self.client
|
||||
with self._lock:
|
||||
if self.client is None:
|
||||
self.credentials=load_credentials(self.credentials_path,self.scopes)
|
||||
self.client=build_sheets_client(self.credentials)
|
||||
else:
|
||||
ensure_fresh(self.credentials)
|
||||
return self.client
|
||||
|
||||
async def _values(self):
|
||||
if not self.spreadsheet_id:
|
||||
raise HTTPException(status_code=500,detail="SPREADSHEET_ID is not configured")
|
||||
client=await asyncio.to_thread(self._connect)
|
||||
return client.spreadsheets().values()
|
||||
|
||||
async def _spreadsheets(self):
|
||||
if not self.spreadsheet_id:
|
||||
raise HTTPException(status_code=500,detail="SPREADSHEET_ID is not configured")
|
||||
client=await asyncio.to_thread(self._connect)
|
||||
return client.spreadsheets()
|
||||
|
||||
# -- reads -------------------------------------------------------------
|
||||
|
||||
async def get_metadata(self):
|
||||
"""Spreadsheet title, id, url and every tab with its row/column counts."""
|
||||
try:
|
||||
spreadsheets=await self._spreadsheets()
|
||||
request=spreadsheets.get(spreadsheetId=self.spreadsheet_id,fields=(
|
||||
"spreadsheetId,spreadsheetUrl,properties(title,locale,timeZone),"
|
||||
"sheets(properties(sheetId,title,index,gridProperties(rowCount,columnCount)))"
|
||||
))
|
||||
payload=await asyncio.to_thread(execute,request,"spreadsheet metadata")
|
||||
return serialize_metadata(payload)
|
||||
except SheetsServiceError as e:
|
||||
raise HTTPException(status_code=e.status_code,detail=e.message)
|
||||
|
||||
async def list_tabs(self):
|
||||
"""Tab titles in sheet order."""
|
||||
metadata=await self.get_metadata()
|
||||
return [tab["title"] for tab in metadata["tabs"] if tab.get("title")]
|
||||
|
||||
async def read_range(self,tab,cell_range=None):
|
||||
"""Raw rows for a tab, or for a sub-range of it when cell_range is given."""
|
||||
try:
|
||||
values=await self._values()
|
||||
target=quote_tab(tab,cell_range)
|
||||
request=values.get(spreadsheetId=self.spreadsheet_id,range=target)
|
||||
payload=await asyncio.to_thread(execute,request,f"read {target}")
|
||||
rows=stringify_rows(payload.get("values"))
|
||||
return serialize_values(tab,cell_range,rows)
|
||||
except SheetsServiceError as e:
|
||||
raise HTTPException(status_code=e.status_code,detail=e.message)
|
||||
|
||||
async def read_records(self,tab):
|
||||
"""Rows keyed by the first row. Blank rows are skipped, short rows padded."""
|
||||
data=await self.read_range(tab)
|
||||
return serialize_records(tab,rows_to_records(data["rows"]))
|
||||
|
||||
async def read_all(self):
|
||||
"""Every tab as records, keyed by tab name."""
|
||||
tabs=await self.list_tabs()
|
||||
sheets={}
|
||||
for tab in tabs:
|
||||
data=await self.read_records(tab)
|
||||
sheets[tab]=data["records"]
|
||||
return {"sheets":sheets,"tabs":tabs,"total":len(tabs)}
|
||||
|
||||
# -- writes ------------------------------------------------------------
|
||||
|
||||
async def append_rows(self,tab,rows):
|
||||
"""Append rows below the tab's current content."""
|
||||
if not rows:
|
||||
raise HTTPException(status_code=422,detail="rows must not be empty")
|
||||
try:
|
||||
values=await self._values()
|
||||
target=quote_tab(tab)
|
||||
request=values.append(
|
||||
spreadsheetId=self.spreadsheet_id,
|
||||
range=target,
|
||||
valueInputOption="USER_ENTERED",
|
||||
insertDataOption="INSERT_ROWS",
|
||||
body={"values":rows},
|
||||
)
|
||||
payload=await asyncio.to_thread(execute,request,f"append to {target}")
|
||||
return serialize_append(tab,payload)
|
||||
except SheetsServiceError as e:
|
||||
raise HTTPException(status_code=e.status_code,detail=e.message)
|
||||
|
||||
async def update_range(self,tab,cell_range,rows):
|
||||
"""Overwrite an explicit A1 range with rows."""
|
||||
if not cell_range:
|
||||
raise HTTPException(status_code=422,detail="cell_range is required")
|
||||
if not rows:
|
||||
raise HTTPException(status_code=422,detail="rows must not be empty")
|
||||
try:
|
||||
values=await self._values()
|
||||
target=quote_tab(tab,cell_range)
|
||||
request=values.update(
|
||||
spreadsheetId=self.spreadsheet_id,
|
||||
range=target,
|
||||
valueInputOption="USER_ENTERED",
|
||||
body={"values":rows},
|
||||
)
|
||||
payload=await asyncio.to_thread(execute,request,f"update {target}")
|
||||
return serialize_update(tab,payload)
|
||||
except SheetsServiceError as e:
|
||||
raise HTTPException(status_code=e.status_code,detail=e.message)
|
||||
|
||||
async def clear_range(self,tab,cell_range):
|
||||
"""Clear the values in an explicit A1 range, leaving formatting intact."""
|
||||
if not cell_range:
|
||||
raise HTTPException(status_code=422,detail="cell_range is required")
|
||||
try:
|
||||
values=await self._values()
|
||||
target=quote_tab(tab,cell_range)
|
||||
request=values.clear(spreadsheetId=self.spreadsheet_id,range=target,body={})
|
||||
payload=await asyncio.to_thread(execute,request,f"clear {target}")
|
||||
return serialize_clear(tab,payload)
|
||||
except SheetsServiceError as e:
|
||||
raise HTTPException(status_code=e.status_code,detail=e.message)
|
||||
|
||||
# -- FormData import / query -------------------------------------------
|
||||
|
||||
async def import_sheet(self,tab):
|
||||
"""Read one tab from Google Sheets and replace its FormData rows."""
|
||||
session=self._require_session()
|
||||
if not tab or not str(tab).strip():
|
||||
raise HTTPException(status_code=422,detail="tab is required")
|
||||
tab=str(tab).strip()
|
||||
data=await self.read_records(tab)
|
||||
records=data["records"]
|
||||
headers=data["headers"]
|
||||
mapped=[]
|
||||
for index,record in enumerate(records):
|
||||
mapped.append(map_record_to_form_data(tab,record,headers,index+2))
|
||||
result=await FormData.replace_sheet(session,tab,mapped)
|
||||
stats=import_row_stats(mapped,headers)
|
||||
return serialize_import({
|
||||
"tab":tab,
|
||||
"rows_read":len(records),
|
||||
"inserted":result["inserted"],
|
||||
"deleted":result["deleted"],
|
||||
**stats,
|
||||
})
|
||||
|
||||
async def import_all(self):
|
||||
"""Import every tab sequentially; one tab failure does not abort the rest."""
|
||||
self._require_session()
|
||||
tabs=await self.list_tabs()
|
||||
reports=[]
|
||||
for tab in tabs:
|
||||
try:
|
||||
report=await self.import_sheet(tab)
|
||||
reports.append(report)
|
||||
except HTTPException as e:
|
||||
logger.warning("import_all tab %s failed: %s",tab,e.detail)
|
||||
reports.append(serialize_import({
|
||||
"tab":tab,"rows_read":0,"inserted":0,"deleted":0,
|
||||
"error":str(e.detail),
|
||||
}))
|
||||
except Exception as e:
|
||||
logger.exception("import_all tab %s failed",tab)
|
||||
reports.append(serialize_import({
|
||||
"tab":tab,"rows_read":0,"inserted":0,"deleted":0,
|
||||
"error":str(e),
|
||||
}))
|
||||
return serialize_import_all(reports)
|
||||
|
||||
async def get_form_data(self,sheet=None,search=None,top=None,skip=None):
|
||||
session=self._require_session()
|
||||
rows=await FormData.fetch_form_data(
|
||||
session,sheet=sheet,search=search,top=top,skip=skip,
|
||||
)
|
||||
total=await FormData.count_form_data(session,sheet=sheet,search=search)
|
||||
return [serialize_form_data(row) for row in rows],total
|
||||
|
||||
async def get_form_data_by_id(self,record_id):
|
||||
session=self._require_session()
|
||||
row=await FormData.get_form_data_by_id(session,record_id)
|
||||
if not row:
|
||||
raise HTTPException(status_code=404,detail="Form data not found")
|
||||
return serialize_form_data(row)
|
||||
|
||||
async def get_imported_sheets(self):
|
||||
session=self._require_session()
|
||||
sheets=await FormData.get_sheet_names(session)
|
||||
return serialize_sheet_summary(sheets)
|
||||
|
||||
async def delete_sheet_data(self,tab):
|
||||
session=self._require_session()
|
||||
if not tab or not str(tab).strip():
|
||||
raise HTTPException(status_code=422,detail="tab is required")
|
||||
deleted=await FormData.delete_by_sheet(session,str(tab).strip())
|
||||
return {"tab":str(tab).strip(),"deleted":deleted}
|
||||
|
||||
async def start_import(self,current_user=None,tab=None):
|
||||
"""Enqueue a sheet import on the shared Taskiq worker; return the run row.
|
||||
|
||||
If a queued/running import already exists, return it instead of stacking another.
|
||||
"""
|
||||
session=self._require_session()
|
||||
active=await SheetImportRun.get_active(session)
|
||||
if active:
|
||||
return serialize_import_run(active)
|
||||
|
||||
created_by=None
|
||||
if isinstance(current_user,dict) and current_user.get("id"):
|
||||
created_by=SheetImportRun._as_uuid(current_user.get("id"))
|
||||
|
||||
tab_value=str(tab).strip() if tab else None
|
||||
row=await SheetImportRun.insert_run(session,{
|
||||
"status":"queued",
|
||||
"created_by":created_by,
|
||||
"tab":tab_value,
|
||||
})
|
||||
|
||||
from g_sheet.tasks import import_sheets
|
||||
task=await import_sheets.kicker().with_labels(
|
||||
created_at=datetime.now(timezone.utc).isoformat(),
|
||||
correlation_id=str(row.id),
|
||||
queue="inbox",
|
||||
).kiq(str(row.id))
|
||||
row=await SheetImportRun.update_run(session,row.id,{"task_id":task.task_id})
|
||||
return serialize_import_run(row)
|
||||
|
||||
async def get_import_run(self,run_id=None):
|
||||
session=self._require_session()
|
||||
if run_id:
|
||||
row=await SheetImportRun.get_by_id(session,run_id)
|
||||
if not row:
|
||||
raise HTTPException(status_code=404,detail="Import run not found")
|
||||
return serialize_import_run(row)
|
||||
row=await SheetImportRun.get_active(session)
|
||||
if row:
|
||||
return serialize_import_run(row)
|
||||
from sqlmodel import select
|
||||
result=await session.execute(
|
||||
select(SheetImportRun).order_by(SheetImportRun.created_at.desc()).limit(1)
|
||||
)
|
||||
row=result.scalars().first()
|
||||
if not row:
|
||||
raise HTTPException(status_code=404,detail="No import runs yet")
|
||||
return serialize_import_run(row)
|
||||
|
||||
# -- health ------------------------------------------------------------
|
||||
|
||||
async def health_check(self):
|
||||
"""Credentials + sheet reachability as a status dict. Never raises."""
|
||||
if not self.spreadsheet_id:
|
||||
return serialize_health(False,"SPREADSHEET_ID is not configured")
|
||||
try:
|
||||
tabs=await self.list_tabs()
|
||||
return serialize_health(True,"spreadsheet reachable",tabs)
|
||||
except HTTPException as e:
|
||||
logger.warning("sheets health check failed: %s",e.detail)
|
||||
return serialize_health(False,str(e.detail))
|
||||
except Exception as e:
|
||||
logger.warning("sheets health check failed: %s",e)
|
||||
return serialize_health(False,str(e))
|
||||
|
|
@ -0,0 +1,97 @@
|
|||
"""Mailbox sync Taskiq tasks — Outlook pull + triage + ingest on own stream."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime,timezone
|
||||
|
||||
import redis.asyncio as redis
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from db_setup import session_scope
|
||||
from inbox.models import MailboxSyncRun
|
||||
from inbox.views import Email
|
||||
from taskiq_management.broker_setup import MAX_RETRIES,RETRY_DELAY
|
||||
from taskiq_management.mailbox_sync_broker_setup import mailbox_sync_broker
|
||||
from taskiq_management.middleware import PermanentTaskError
|
||||
|
||||
load_dotenv()
|
||||
|
||||
logger=logging.getLogger("inbox.mailbox_sync")
|
||||
REDIS_URL=os.getenv("REDIS_URL","redis://localhost:6379/0")
|
||||
_LOCK_KEY="inbox:mailbox_sync:lock"
|
||||
_LOCK_TTL=900
|
||||
|
||||
|
||||
async def _fail(run_id:str,error:str) -> dict:
|
||||
async with session_scope() as session:
|
||||
await MailboxSyncRun.update_run(session,run_id,{
|
||||
"status":"failed",
|
||||
"error":error,
|
||||
"finished_at":datetime.now(timezone.utc),
|
||||
})
|
||||
return {"status":"failed","error":error}
|
||||
|
||||
|
||||
@mailbox_sync_broker.task(
|
||||
task_name="inbox.sync_mailbox",
|
||||
retry_on_error=True,
|
||||
max_retries=MAX_RETRIES,
|
||||
delay=RETRY_DELAY,
|
||||
)
|
||||
async def sync_mailbox(run_id:str) -> dict:
|
||||
if not run_id or not str(run_id).strip():
|
||||
raise PermanentTaskError("run_id is required")
|
||||
run_id=str(run_id).strip()
|
||||
|
||||
client=redis.from_url(REDIS_URL,decode_responses=True)
|
||||
try:
|
||||
acquired=await client.set(_LOCK_KEY,run_id,nx=True,ex=_LOCK_TTL)
|
||||
if not acquired:
|
||||
return await _fail(run_id,"another mailbox sync is already running")
|
||||
|
||||
try:
|
||||
async with session_scope() as session:
|
||||
row=await MailboxSyncRun.get_by_id(session,run_id)
|
||||
if not row:
|
||||
raise PermanentTaskError(f"sync run {run_id} not found")
|
||||
await MailboxSyncRun.update_run(session,run_id,{
|
||||
"status":"running",
|
||||
"started_at":datetime.now(timezone.utc),
|
||||
"error":None,
|
||||
})
|
||||
top=row.top or 100
|
||||
skip=row.skip or 0
|
||||
test_on=True if row.test_on is None else bool(row.test_on)
|
||||
|
||||
async with session_scope() as session:
|
||||
service=Email(session=session)
|
||||
if not service.token:
|
||||
return await _fail(run_id,"EMAIL_API_TOKEN is not configured")
|
||||
try:
|
||||
summary=await service.run_mailbox_sync_page(
|
||||
top=top,skip=skip,test_on=test_on,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.exception("mailbox sync failed for run %s",run_id)
|
||||
return await _fail(run_id,str(e))
|
||||
|
||||
await MailboxSyncRun.update_run(session,run_id,{
|
||||
"status":"completed",
|
||||
"entries":summary["entries"],
|
||||
"triage":summary["triage"],
|
||||
"error":None,
|
||||
"finished_at":datetime.now(timezone.utc),
|
||||
})
|
||||
return {
|
||||
"status":"completed",
|
||||
"triage":summary["triage"],
|
||||
"entries":len(summary["entries"]),
|
||||
}
|
||||
finally:
|
||||
current=await client.get(_LOCK_KEY)
|
||||
if current==run_id:
|
||||
await client.delete(_LOCK_KEY)
|
||||
finally:
|
||||
await client.aclose()
|
||||
|
|
@ -106,9 +106,12 @@ class Inbox(SQLModel, table=True):
|
|||
.outerjoin(AtsResults,cls.ats_id==AtsResults.id)
|
||||
.where(Inbox_Messages.assigned_job_post_id.is_not(None))
|
||||
.where(Roles.role_name==EnumRoles.CANDIDATE.value)
|
||||
# Newest-first is the list contract; score is only a tiebreak
|
||||
# within the same instant. id keeps paging stable.
|
||||
.order_by(
|
||||
AtsResults.overall_score.desc().nulls_last(),
|
||||
cls.created_at.desc(),
|
||||
AtsResults.overall_score.desc().nulls_last(),
|
||||
cls.id.desc(),
|
||||
)
|
||||
)
|
||||
if job_post_id:
|
||||
|
|
@ -203,6 +206,9 @@ class Inbox(SQLModel, table=True):
|
|||
qry = qry.where(cls.user_id == user_id)
|
||||
if search:
|
||||
qry = qry.where(cls._candidate_search_filter(search))
|
||||
# Most-recent-first is the list contract; id breaks ties so a page
|
||||
# boundary can't drop or repeat a row when created_at collides.
|
||||
qry = qry.order_by(cls.created_at.desc(), cls.id.desc())
|
||||
qry = qry.limit(limit).offset(offset)
|
||||
result = await session.execute(qry)
|
||||
rows = result.scalars().all()
|
||||
|
|
|
|||
|
|
@ -393,15 +393,17 @@ async def score_inbox_candidates(
|
|||
@router.get("/candidate/scored/fetch")
|
||||
async def fetch_scored_candidates(
|
||||
job_id: str = Query(None),
|
||||
limit: int = Query(10, ge=1, le=100),
|
||||
offset: int = Query(0, ge=0),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.CANDIDATES_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
):
|
||||
"""Persisted leaderboard: completed by score desc, failures last. Without job_id
|
||||
returns the whole pool across jobs."""
|
||||
"""Persisted scored candidates, newest first. Without job_id returns the whole
|
||||
pool across jobs. `total` is the full result-set size, not the page length."""
|
||||
try:
|
||||
service=CandidateScoring(session=session)
|
||||
data=await service.fetch_candidates(job_id)
|
||||
return JSONResponse(content={"data":data,"total":len(data),"status_code":200})
|
||||
data,total=await service.fetch_candidates(job_id,limit=limit,offset=offset)
|
||||
return JSONResponse(content={"data":data,"total":total,"status_code":200})
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
|
|
@ -411,7 +413,7 @@ async def fetch_scored_candidates(
|
|||
@router.get("/job/fetch")
|
||||
async def fetch_job_posts(
|
||||
search: str | None = Query(None),
|
||||
top: int | None = Query(None),
|
||||
top: int | None = Query(10, ge=1, le=100),
|
||||
skip: int = Query(0, ge=0),
|
||||
ids: str | None = Query(None),
|
||||
active_only: bool = Query(True),
|
||||
|
|
@ -451,7 +453,9 @@ async def fetch_jobs(
|
|||
department: str | None = Query(None),
|
||||
requisition_status: str | None = Query(None),
|
||||
employment_type: str | None = Query(None),
|
||||
top: int | None = Query(None),
|
||||
# le=500 (not 100): the Jobs board loads a full client-side page for facets;
|
||||
# a 200 ceiling used to 422 the SPA and render an empty requisition list.
|
||||
top: int | None = Query(10, ge=1, le=500),
|
||||
skip: int = Query(0, ge=0),
|
||||
# Defaults False, unlike /job/fetch: a requisition list must show CLOSED
|
||||
# requisitions, and those carry is_active = false. Soft-deleted rows are still
|
||||
|
|
@ -521,8 +525,8 @@ async def fetch_candidate_by_id(
|
|||
@router.get("/candidate/fetch")
|
||||
async def fetch_candidate(
|
||||
user_id:str=Query(None),
|
||||
limit:int=Query(10),
|
||||
offset:int=Query(0),
|
||||
limit:int=Query(10,ge=1,le=100),
|
||||
offset:int=Query(0,ge=0),
|
||||
search:str=Query(None),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.CANDIDATES_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
|
|
@ -807,7 +811,7 @@ async def change_candidate_stage(
|
|||
@router.get("/pipeline/candidates/fetch")
|
||||
async def fetch_pipeline_candidates(
|
||||
job_post_id:Optional[uuid.UUID]=Query(None),
|
||||
limit:int=Query(200,ge=1,le=1000),
|
||||
limit:int=Query(10,ge=1,le=1000),
|
||||
offset:int=Query(0,ge=0),
|
||||
current_user: dict = Depends(require_permission(PermissionTag.PIPELINE_VIEW)),
|
||||
session: AsyncSession = Depends(get_session),
|
||||
|
|
|
|||
|
|
@ -96,9 +96,12 @@ class Manual_UPLOAD_CANDIDATE(SQLModel, table=True):
|
|||
&(AtsResults.job_post_id==cls.job_post_id)
|
||||
&(AtsResults.is_current==True), # noqa: E712
|
||||
)
|
||||
# Newest-first is the list contract; score is only a tiebreak
|
||||
# within the same instant. id keeps paging stable.
|
||||
.order_by(
|
||||
AtsResults.overall_score.desc().nulls_last(),
|
||||
cls.created_at.desc(),
|
||||
AtsResults.overall_score.desc().nulls_last(),
|
||||
cls.id.desc(),
|
||||
)
|
||||
)
|
||||
if job_post_id:
|
||||
|
|
@ -285,25 +288,40 @@ class Candidates(SQLModel, table=True):
|
|||
return result.scalars().first()
|
||||
|
||||
@classmethod
|
||||
async def get_candidates_by_job(cls, session: AsyncSession, job_id: str | None = None):
|
||||
"""Leaderboard order: completed by score desc, failures last, ties stable.
|
||||
async def get_candidates_by_job(
|
||||
cls,
|
||||
session: AsyncSession,
|
||||
job_id: str | None = None,
|
||||
limit: int | None = None,
|
||||
offset: int = 0,
|
||||
):
|
||||
"""Most-recent-first list, paged. Score is only a tiebreak within an instant.
|
||||
|
||||
job_id=None returns the whole pool across jobs (same ordering) for the
|
||||
frontend's unscoped Candidates/Talent Pool views.
|
||||
frontend's unscoped Candidates/Talent Pool views. Returns (rows, total) so
|
||||
the caller can page without a second count query of its own.
|
||||
"""
|
||||
statement = select(cls)
|
||||
if job_id is not None:
|
||||
uid = cls._as_uuid(job_id)
|
||||
if uid is None:
|
||||
return []
|
||||
return [], 0
|
||||
statement = statement.where(cls.job_id == uid)
|
||||
total = (
|
||||
await session.execute(select(func.count()).select_from(statement.subquery()))
|
||||
).scalar_one()
|
||||
statement = statement.order_by(
|
||||
cls.created_at.desc(),
|
||||
cls.status.asc(), # "completed" < "failed"
|
||||
cls.match_score.desc().nulls_last(),
|
||||
cls.created_at.asc(),
|
||||
cls.id.desc(),
|
||||
)
|
||||
if offset:
|
||||
statement = statement.offset(offset)
|
||||
if limit is not None:
|
||||
statement = statement.limit(limit)
|
||||
result = await session.execute(statement)
|
||||
return result.scalars().all()
|
||||
return list(result.scalars().all()), total
|
||||
|
||||
@classmethod
|
||||
async def get_completed_by_email_job(cls, session: AsyncSession, email, job_id):
|
||||
|
|
|
|||
|
|
@ -350,14 +350,16 @@ class CandidateScoring:
|
|||
raise HTTPException(status_code=400,detail="No attachments found for the given message(s)")
|
||||
return await self._score_and_persist(job_id,sources,"inbox",current_user)
|
||||
|
||||
async def fetch_candidates(self,job_id=None):
|
||||
async def fetch_candidates(self,job_id=None,limit=10,offset=0):
|
||||
# job_id omitted -> the whole pool across jobs (frontend Candidates/TalentPool).
|
||||
if job_id is not None:
|
||||
job=await JobPosts.get_job_post_by_id(self.session,job_id)
|
||||
if job is None or job.is_deleted:
|
||||
raise HTTPException(status_code=404,detail="Job post not found")
|
||||
rows=await Candidates.get_candidates_by_job(self.session,job_id)
|
||||
return [serialize_candidate(row) for row in rows]
|
||||
rows,total=await Candidates.get_candidates_by_job(
|
||||
self.session,job_id,limit=limit,offset=offset,
|
||||
)
|
||||
return [serialize_candidate(row) for row in rows],total
|
||||
|
||||
async def fetch_candidate_by_id(self,candidate_id):
|
||||
row=await Candidates.get_candidate_by_id(self.session,candidate_id)
|
||||
|
|
|
|||
|
|
@ -13,9 +13,11 @@ class Pipeline:
|
|||
def __init__(self,session:AsyncSession):
|
||||
self.session=session
|
||||
|
||||
async def get_all(self,job_post_id=None,limit=None,offset=0):
|
||||
# limit/offset are per-source, not a merged page: two tables, no common
|
||||
# order key. limit=200 returns up to 200 inbox AND up to 200 manual rows.
|
||||
async def get_all(self,job_post_id=None,limit=10,offset=0):
|
||||
# limit/offset are per-source, not a merged page: two tables that cannot be
|
||||
# paged as one. limit=10 returns up to 10 inbox AND up to 10 manual rows,
|
||||
# each newest-first by created_at. `counts`/`total` stay full-set sizes so
|
||||
# the caller can drive paging off them.
|
||||
try:
|
||||
inbox_data=await Inbox.get_all(self.session,job_post_id=job_post_id,limit=limit,offset=offset)
|
||||
manual_upload_data=await Manual_UPLOAD_CANDIDATE.get_all(self.session,job_post_id=job_post_id,limit=limit,offset=offset)
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ from contextlib import asynccontextmanager
|
|||
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi import FastAPI
|
||||
from fastapi.responses import JSONResponse
|
||||
from db_setup import lifespan as db_lifespan
|
||||
from inbox.app import router as inbox_router
|
||||
from users.app import router as users_router
|
||||
|
|
@ -91,6 +92,13 @@ app.add_middleware(
|
|||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
async def health():
|
||||
# Liveness only — no DB. Compose healthchecks and load balancers hit this.
|
||||
return JSONResponse(content={"status":"ok","status_code":200})
|
||||
|
||||
|
||||
app.include_router(inbox_router)
|
||||
app.include_router(users_router)
|
||||
app.include_router(role_router)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,59 @@
|
|||
"""Taskiq mailbox-sync broker — isolated Redis stream so Outlook pull/triage
|
||||
never blocks inbox match/ATS or the cv_upload queue.
|
||||
|
||||
Worker: taskiq worker taskiq_management.mailbox_sync_broker_setup:mailbox_sync_broker inbox.mailbox_sync_tasks
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from taskiq import TaskiqScheduler
|
||||
from taskiq.middlewares import SmartRetryMiddleware
|
||||
from taskiq.schedule_sources import LabelScheduleSource
|
||||
from taskiq_redis import (
|
||||
ListRedisScheduleSource,
|
||||
RedisAsyncResultBackend,
|
||||
RedisStreamBroker,
|
||||
)
|
||||
|
||||
from taskiq_management.broker_setup import MAX_RETRIES,RETRY_DELAY
|
||||
from taskiq_management.middleware import DeadLetterMiddleware
|
||||
|
||||
load_dotenv()
|
||||
|
||||
REDIS_URL=os.getenv("REDIS_URL","redis://localhost:6379/0")
|
||||
MAILBOX_SYNC_QUEUE_NAME=os.getenv("TASKIQ_MAILBOX_SYNC_QUEUE_NAME","mailbox_sync")
|
||||
|
||||
result_backend=RedisAsyncResultBackend(redis_url=REDIS_URL)
|
||||
mailbox_sync_schedule_source=ListRedisScheduleSource(
|
||||
url=REDIS_URL,prefix="taskiq:schedule:mailbox_sync",
|
||||
)
|
||||
|
||||
mailbox_sync_broker=(
|
||||
RedisStreamBroker(
|
||||
url=REDIS_URL,
|
||||
queue_name=MAILBOX_SYNC_QUEUE_NAME,
|
||||
consumer_group_name=os.getenv("TASKIQ_CONSUMER_GROUP","taskiq"),
|
||||
idle_timeout=int(os.getenv("TASKIQ_IDLE_TIMEOUT_MS","600000")),
|
||||
)
|
||||
.with_result_backend(result_backend)
|
||||
.with_middlewares(
|
||||
DeadLetterMiddleware(redis_url=REDIS_URL),
|
||||
SmartRetryMiddleware(
|
||||
default_retry_count=MAX_RETRIES,
|
||||
default_retry_label=True,
|
||||
default_delay=RETRY_DELAY,
|
||||
use_jitter=True,
|
||||
use_delay_exponent=True,
|
||||
max_delay_exponent=float(os.getenv("TASKIQ_MAX_DELAY","120")),
|
||||
schedule_source=mailbox_sync_schedule_source,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
mailbox_sync_scheduler=TaskiqScheduler(
|
||||
broker=mailbox_sync_broker,
|
||||
sources=[mailbox_sync_schedule_source,LabelScheduleSource(mailbox_sync_broker)],
|
||||
)
|
||||
|
|
@ -1,28 +1,30 @@
|
|||
# Live-code overlay. Mounts the source folders into the running containers, so what
|
||||
# executes is what is on disk on the host — edit, save, uvicorn reloads.
|
||||
# Optional local overlay — NOT required for day-to-day use.
|
||||
#
|
||||
# docker compose -f docker-compose.yml -f docker-compose.dev.yml up
|
||||
# Default workflow (host Postgres + loopback API ports) is in docker-compose.yml:
|
||||
# docker compose up -d --build
|
||||
#
|
||||
# The CV attachments mount from the base file still applies: compose merges volumes
|
||||
# by target path, and /app/inbox/decoded_attachments is nested under the /app mount,
|
||||
# so the daemon mounts the parent first and the attachments folder on top.
|
||||
# Use this file only when you want --reload and bind-mounted source:
|
||||
# docker compose -f docker-compose.yml -f docker-compose.dev.yml up -d --build
|
||||
#
|
||||
# Two mounts per Python service, not one: /app is the backend tree and /app/app is the
|
||||
# bulk-ats engine that backend/job/candidate imports. Mounting only ./backend over
|
||||
# /app would hide the engine baked into the image and every worker would fail on
|
||||
# import.
|
||||
#
|
||||
# The frontend stays as the built nginx image — a Vite dev server wants node_modules
|
||||
# on the mount, which is slow and fragile across a Windows bind mount. Run
|
||||
# `npm run dev` on the host for the frontend loop.
|
||||
# Do not set DB_HOST here. Keep PROD_ENV=false and DB_HOST=localhost in
|
||||
# backend/.env; db_setup rewrites localhost → host.docker.internal in Docker.
|
||||
|
||||
services:
|
||||
backend-api:
|
||||
command:
|
||||
["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000", "--reload"]
|
||||
[
|
||||
"uvicorn",
|
||||
"main:app",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8000",
|
||||
"--reload",
|
||||
]
|
||||
volumes:
|
||||
- ./backend:/app
|
||||
- ./app:/app/app
|
||||
- ${ATTACHMENTS_DIR:-./backend/inbox/decoded_attachments}:/app/inbox/decoded_attachments
|
||||
|
||||
ats-engine:
|
||||
command:
|
||||
|
|
@ -39,10 +41,15 @@ services:
|
|||
volumes:
|
||||
- ./app:/srv/app
|
||||
|
||||
frontend:
|
||||
ports:
|
||||
- "${FRONTEND_PORT:-5173}:80"
|
||||
|
||||
taskiq-worker:
|
||||
volumes:
|
||||
- ./backend:/app
|
||||
- ./app:/app/app
|
||||
- ${ATTACHMENTS_DIR:-./backend/inbox/decoded_attachments}:/app/inbox/decoded_attachments
|
||||
|
||||
taskiq-scheduler:
|
||||
volumes:
|
||||
|
|
@ -53,6 +60,7 @@ services:
|
|||
volumes:
|
||||
- ./backend:/app
|
||||
- ./app:/app/app
|
||||
- ${ATTACHMENTS_DIR:-./backend/inbox/decoded_attachments}:/app/inbox/decoded_attachments
|
||||
|
||||
taskiq-cv-scheduler:
|
||||
volumes:
|
||||
|
|
@ -63,3 +71,4 @@ services:
|
|||
volumes:
|
||||
- ./backend:/app
|
||||
- ./app:/app/app
|
||||
- ${ATTACHMENTS_DIR:-./backend/inbox/decoded_attachments}:/app/inbox/decoded_attachments
|
||||
|
|
|
|||
|
|
@ -0,0 +1,24 @@
|
|||
# Optional overlay: publish API / ATS / Redis on loopback for host tools
|
||||
# (curl, redis-cli, Postman). Not required for the SPA - nginx proxies
|
||||
# to backend-api on the Compose network.
|
||||
#
|
||||
# docker compose --env-file ./backend/.env -f docker-compose.yml -f docker-compose.host-ports.yml up -d
|
||||
#
|
||||
# If bind fails because an IDE (Cursor/VS Code) still holds the port after a
|
||||
# previous run, clear Port Forwarding in the IDE or override in backend/.env:
|
||||
# BACKEND_PORT=8001
|
||||
# ATS_PORT=8101
|
||||
# REDIS_PORT=6380
|
||||
|
||||
services:
|
||||
redis:
|
||||
ports:
|
||||
- "127.0.0.1:${REDIS_PORT:-6379}:6379"
|
||||
|
||||
backend-api:
|
||||
ports:
|
||||
- "127.0.0.1:${BACKEND_PORT:-8000}:8000"
|
||||
|
||||
ats-engine:
|
||||
ports:
|
||||
- "127.0.0.1:${ATS_PORT:-8100}:8100"
|
||||
|
|
@ -1,32 +1,34 @@
|
|||
# HR-ATS-Portal — every service and every image, in one file.
|
||||
# HR-ATS-Portal — single Compose file for local and production.
|
||||
#
|
||||
# docker compose build # hrms-backend / hrms-ats-engine / hrms-frontend
|
||||
# docker compose up -d
|
||||
# docker compose ps
|
||||
# docker compose logs -f backend-api
|
||||
# Sole env file: backend/.env (no repo-root .env). Pass it for Compose
|
||||
# variable substitution (${FRONTEND_PORT}, ${DB_*}, …):
|
||||
#
|
||||
# ── Postgres ──────────────────────────────────────────────────────────────────────
|
||||
# The `postgres` service at the bottom is behind a compose PROFILE, so it is defined
|
||||
# and buildable here but never starts with a plain `docker compose up`. The stack
|
||||
# talks to the PostgreSQL server already running on the host, via
|
||||
# DB_HOST=host.docker.internal (backend/.env says localhost — correct for a host
|
||||
# process, wrong inside a container).
|
||||
# docker compose --env-file ./backend/.env up -d --build
|
||||
# docker compose --env-file ./backend/.env ps
|
||||
# docker compose --env-file ./backend/.env logs -f backend-api
|
||||
#
|
||||
# Host Postgres must accept connections from the Docker bridge: listen_addresses = '*'
|
||||
# in postgresql.conf and a pg_hba.conf line for 172.16.0.0/12 (or the specific subnet).
|
||||
# Local (PROD_ENV=false, DB_HOST=localhost in backend/.env):
|
||||
# Containers reach host Postgres via host.docker.internal (db_setup rewrite
|
||||
# when IN_DOCKER=1). API / ATS / Redis stay on the Compose network by default
|
||||
# (avoids IDE/Cursor stale port-forwards fighting Docker on Windows). Opt in:
|
||||
# docker compose --env-file ./backend/.env -f docker-compose.yml -f docker-compose.host-ports.yml up -d
|
||||
#
|
||||
# To build/run the containerised database instead, see the comments on that service.
|
||||
# Prod (PROD_ENV=true, DB_* = RDS in backend/.env — edit manually):
|
||||
# Same command. No host rewrite; SSL require when DB_SSLMODE is blank.
|
||||
#
|
||||
# ── Ports ─────────────────────────────────────────────────────────────────────────
|
||||
# Stop a host `uvicorn` (8000) and `npm run dev` (5173) before starting these, or
|
||||
# override with BACKEND_PORT / FRONTEND_PORT / ATS_PORT. Windows lets a host process
|
||||
# bind 127.0.0.1:8000 while Docker binds 0.0.0.0:8000, and `localhost` resolves to ::1
|
||||
# first — both listen, and requests reach whichever won.
|
||||
# Optional Compose Postgres (empty volume, not host/RDS data):
|
||||
# docker compose --env-file ./backend/.env --profile postgres up -d postgres
|
||||
#
|
||||
# ── Overlay ───────────────────────────────────────────────────────────────────────
|
||||
# docker-compose.dev.yml adds live source mounts and --reload on top of this file. It
|
||||
# defines no services or images of its own; it only overrides the ones here:
|
||||
# docker compose -f docker-compose.yml -f docker-compose.dev.yml up
|
||||
# Optional live-reload / bind mounts (not required day-to-day):
|
||||
# docker compose --env-file ./backend/.env -f docker-compose.yml -f docker-compose.dev.yml up -d --build
|
||||
#
|
||||
# See DOCKER.md for env checklist and verification.
|
||||
|
||||
x-logging: &default-logging
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: "10m"
|
||||
max-file: "3"
|
||||
|
||||
x-backend-build: &backend-build
|
||||
# Root context, not ./backend: backend/job/candidate imports the bulk-ats engine
|
||||
|
|
@ -36,50 +38,47 @@ x-backend-build: &backend-build
|
|||
|
||||
x-backend-env: &backend-env
|
||||
PYTHONPATH: /app
|
||||
# backend/.env is written for host processes; these are the values a container needs.
|
||||
# Lets db_setup rewrite DB_HOST=localhost → host.docker.internal (local only).
|
||||
IN_DOCKER: "1"
|
||||
# Credentials and endpoints come from backend/.env via `env_file` below.
|
||||
# DB_HOST is deliberately ABSENT here: an empty `environment:` override would
|
||||
# blank env_file / RDS. Do not name credential keys under environment.
|
||||
#
|
||||
# DB_HOST defaults to the LOCAL Postgres server on the host. Set DB_HOST=postgres in
|
||||
# the shell (or a root .env) to point the whole stack at the container below instead
|
||||
# — that is the only value that has to change, since services reach it over the
|
||||
# compose network on 5432, not the published host port.
|
||||
DB_HOST: ${DB_HOST:-host.docker.internal}
|
||||
# REDIS_URL and BACKEND_URL are compose-network DNS names, correct everywhere.
|
||||
REDIS_URL: redis://redis:6379/0
|
||||
# Prefer root/.env EMAIL_URL (e.g. http://3.140.173.13:5000). Fall back to the
|
||||
# host-gateway alias when the mail service runs on this machine's :5000.
|
||||
EMAIL_URL: ${EMAIL_URL:-http://host.docker.internal:5000}
|
||||
BACKEND_URL: http://backend-api:8000
|
||||
# Apify talent sourcing: interpolated from the shell or a root .env, so a token
|
||||
# kept at repo root (APIFY_TOKEN) reaches containers without duplicating it
|
||||
# into backend/.env.
|
||||
# Apify token, if kept at repo root rather than in backend/.env. Harmless when
|
||||
# unset: the app treats empty as absent.
|
||||
APIFY_API_TOKEN: ${APIFY_API_TOKEN:-${APIFY_TOKEN:-}}
|
||||
|
||||
# The one shared folder. Every process that decodes, scores or serves a CV reads and
|
||||
# writes the same host directory, so a file written by the API is the same file the
|
||||
# worker opens. Absolute paths stored in the DB match across services because the
|
||||
# mount target is identical everywhere; inbox.plugins.resolve_attachment_path also
|
||||
# falls back to basename-under-this-directory for rows written by a host process.
|
||||
# Shared CV storage. Named volume so API + workers see the same files.
|
||||
# Optional docker-compose.dev.yml remounts ./backend/inbox/decoded_attachments.
|
||||
x-attachments: &attachments
|
||||
- ${ATTACHMENTS_DIR:-./backend/inbox/decoded_attachments}:/app/inbox/decoded_attachments
|
||||
- attachments-data:/app/inbox/decoded_attachments
|
||||
|
||||
x-backend-service: &backend-service
|
||||
build: *backend-build
|
||||
image: hrms-backend:local
|
||||
working_dir: /app
|
||||
env_file:
|
||||
# backend/.env is the source of truth (plain DB_* + PROD_ENV).
|
||||
- ./backend/.env
|
||||
# Optional local overrides (required:false). Do not set DB_HOST=postgres
|
||||
# here unless you intentionally start the postgres profile.
|
||||
- path: ./docker.local.env
|
||||
required: false
|
||||
environment: *backend-env
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_healthy
|
||||
# required:false — the default stack uses the host's Postgres and never starts
|
||||
# this one, and that must not be an error. When the profile IS active, startup
|
||||
# waits for it to pass pg_isready.
|
||||
# required:false — default stack never starts postgres (host Postgres or RDS).
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
required: false
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
|
||||
services:
|
||||
# --- Redis (broker + result backend for taskiq) ----------------------------------
|
||||
|
|
@ -87,8 +86,8 @@ services:
|
|||
image: redis:7-alpine
|
||||
container_name: hrms-redis
|
||||
command: ["redis-server", "--appendonly", "yes"]
|
||||
ports:
|
||||
- "${REDIS_PORT:-6379}:6379"
|
||||
# No host publish by default (Compose DNS redis:6379). Optional loopback:
|
||||
# docker-compose.host-ports.yml
|
||||
volumes:
|
||||
- redis-data:/data
|
||||
healthcheck:
|
||||
|
|
@ -97,24 +96,71 @@ services:
|
|||
timeout: 5s
|
||||
retries: 5
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
|
||||
# --- Postgres: DEFINED HERE, NOT STARTED BY DEFAULT -------------------------------
|
||||
# Default local path is host Postgres (DB_HOST=localhost → host.docker.internal).
|
||||
# Prod uses AWS RDS. Opt in only when you want a disposable Compose DB:
|
||||
#
|
||||
# docker compose --profile postgres up -d postgres
|
||||
# # then set DB_HOST=postgres in backend/.env and recreate backend services
|
||||
#
|
||||
postgres:
|
||||
profiles: ["postgres"]
|
||||
build:
|
||||
context: ./docker/postgres
|
||||
image: hrms-postgres:local
|
||||
container_name: hrms-postgres
|
||||
environment:
|
||||
# Compose substitution needs: docker compose --env-file ./backend/.env …
|
||||
# (there is no repo-root .env). Defaults apply if the flag is omitted.
|
||||
POSTGRES_USER: ${DB_USERNAME:-postgres}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-postgres}
|
||||
POSTGRES_DB: ${DB_NAME:-hrms}
|
||||
POSTGRES_INITDB_ARGS: "--encoding=UTF8"
|
||||
ports:
|
||||
# 5433: the host's own Postgres owns 5432.
|
||||
- "127.0.0.1:${POSTGRES_PORT:-5433}:5432"
|
||||
volumes:
|
||||
- postgres-data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test:
|
||||
["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-postgres} -d ${DB_NAME:-hrms}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 20s
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
|
||||
# --- portal API ------------------------------------------------------------------
|
||||
backend-api:
|
||||
<<: *backend-service
|
||||
container_name: hrms-backend-api
|
||||
command: ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"]
|
||||
command:
|
||||
[
|
||||
"uvicorn",
|
||||
"main:app",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8000",
|
||||
"--workers",
|
||||
"${UVICORN_WORKERS:-2}",
|
||||
"--proxy-headers",
|
||||
"--forwarded-allow-ips=*",
|
||||
]
|
||||
# Published for host tools / Vite (`VITE_API_BASE=http://127.0.0.1:8000`).
|
||||
ports:
|
||||
- "${BACKEND_PORT:-8000}:8000"
|
||||
volumes: *attachments
|
||||
healthcheck:
|
||||
# python:slim has no curl and the API exposes no /health route, so this is a
|
||||
# plain TCP check against the uvicorn socket.
|
||||
test:
|
||||
[
|
||||
"CMD",
|
||||
"python",
|
||||
"-c",
|
||||
"import socket;socket.create_connection(('127.0.0.1',8000),3).close()",
|
||||
"import urllib.request;urllib.request.urlopen('http://127.0.0.1:8000/health',timeout=3)",
|
||||
]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
|
|
@ -129,13 +175,13 @@ services:
|
|||
image: hrms-ats-engine:local
|
||||
container_name: hrms-ats-engine
|
||||
env_file:
|
||||
# OPENAI_API_KEY currently lives in backend/.env; a root .env (see .env.example)
|
||||
# overrides it when present, and the stack still comes up when it is not.
|
||||
- ./backend/.env
|
||||
- path: ./.env
|
||||
- path: ./docker.local.env
|
||||
required: false
|
||||
ports:
|
||||
- "${ATS_PORT:-8100}:8100"
|
||||
environment:
|
||||
IN_DOCKER: "1"
|
||||
# No host publish by default (backend uses ats-engine:8100). Optional:
|
||||
# docker-compose.host-ports.yml
|
||||
healthcheck:
|
||||
test:
|
||||
[
|
||||
|
|
@ -149,16 +195,20 @@ services:
|
|||
retries: 5
|
||||
start_period: 20s
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
|
||||
# --- React portal -----------------------------------------------------------------
|
||||
# Portal on FRONTEND_PORT (default 5173). API also on BACKEND_PORT (8000).
|
||||
frontend:
|
||||
build:
|
||||
context: ./frontend
|
||||
args:
|
||||
# Baked into the bundle at build time — change it and rebuild, not restart.
|
||||
VITE_API_BASE: ${VITE_API_BASE:-http://localhost:8000}
|
||||
# Empty = same-origin; nginx proxies API paths to backend-api (see nginx.conf).
|
||||
VITE_API_BASE: ${VITE_API_BASE:-}
|
||||
image: hrms-frontend:local
|
||||
container_name: hrms-frontend
|
||||
depends_on:
|
||||
backend-api:
|
||||
condition: service_healthy
|
||||
ports:
|
||||
- "${FRONTEND_PORT:-5173}:80"
|
||||
healthcheck:
|
||||
|
|
@ -167,6 +217,7 @@ services:
|
|||
timeout: 5s
|
||||
retries: 5
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
|
||||
# --- background processing (same image as backend-api, different command) ---------
|
||||
taskiq-worker:
|
||||
|
|
@ -254,45 +305,7 @@ services:
|
|||
TASKIQ_WORKER_NAME: mailbox-sync-worker-01
|
||||
volumes: *attachments
|
||||
|
||||
# --- Postgres: DEFINED HERE, NOT STARTED BY DEFAULT -------------------------------
|
||||
# The profile is what keeps it out of `docker compose build` and `docker compose up`.
|
||||
# Nothing about the default stack changes by its presence in this file.
|
||||
#
|
||||
# docker compose --profile postgres build postgres # build the image
|
||||
# docker compose --profile postgres up -d postgres # run it, host port 5433
|
||||
#
|
||||
# Pointing the app at it is a separate, deliberate step — set DB_HOST=postgres (see
|
||||
# x-backend-env) and recreate the services. The volume starts empty, so Alembic
|
||||
# rebuilds the schema on first boot; it does not share the host server's data.
|
||||
postgres:
|
||||
profiles: ["postgres"]
|
||||
build:
|
||||
context: ./docker/postgres
|
||||
image: hrms-postgres:local
|
||||
container_name: hrms-postgres
|
||||
environment:
|
||||
# Interpolated from the shell or a root .env, NOT from backend/.env — compose
|
||||
# variable substitution and container environment are different things.
|
||||
# Defaults match backend/.env.example.
|
||||
POSTGRES_USER: ${DB_USERNAME:-postgres}
|
||||
POSTGRES_PASSWORD: ${DB_PASSWORD:-postgres}
|
||||
POSTGRES_DB: ${DB_NAME:-hrms}
|
||||
POSTGRES_INITDB_ARGS: "--encoding=UTF8"
|
||||
ports:
|
||||
# 5433: the host's own Postgres server owns 5432. Only for host-side tools —
|
||||
# containers reach this one on 5432 over the compose network.
|
||||
- "${POSTGRES_PORT:-5433}:5432"
|
||||
volumes:
|
||||
- postgres-data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test:
|
||||
["CMD-SHELL", "pg_isready -U ${DB_USERNAME:-postgres} -d ${DB_NAME:-hrms}"]
|
||||
interval: 10s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
start_period: 20s
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
redis-data:
|
||||
postgres-data:
|
||||
attachments-data:
|
||||
|
|
|
|||
|
|
@ -1,15 +1,8 @@
|
|||
# syntax=docker/dockerfile:1
|
||||
#
|
||||
# Postgres image — BUILT, BUT NOT USED BY THE DEFAULT STACK.
|
||||
#
|
||||
# Every service in docker-compose.yml points at the Postgres already running on the
|
||||
# host (DB_HOST=host.docker.internal). This image exists so the database can be
|
||||
# containerised on demand — a clean machine, a throwaway test run, a second dev — and
|
||||
# it is kept in its own file (docker-compose.postgres.yml) so bringing it up is always
|
||||
# a deliberate act:
|
||||
#
|
||||
# docker compose -f docker-compose.yml -f docker-compose.postgres.yml build postgres
|
||||
# docker compose -f docker-compose.yml -f docker-compose.postgres.yml up -d postgres
|
||||
# Postgres image used by the production Compose stack (DB_HOST=postgres).
|
||||
# The volume starts empty on first boot; Alembic rebuilds the schema. It does
|
||||
# not share the host server's data — migrate with pg_dump/pg_restore if needed.
|
||||
#
|
||||
# Context is ./docker/postgres.
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,10 @@
|
|||
# Use 127.0.0.1, not localhost. On this machine localhost prefers ::1 and hits a
|
||||
# different listener (WSL/Docker on :8000) instead of the Windows uvicorn on 127.0.0.1.
|
||||
# Use 127.0.0.1, not localhost. On this machine localhost prefers ::1 and can
|
||||
# hit a different listener than the Windows uvicorn / Docker publish.
|
||||
#
|
||||
# Docker Compose (prod-like): leave empty and run `npm run dev` with the
|
||||
# host-ports overlay so Vite can proxy — or open the built SPA on :8080
|
||||
# (nginx same-origin proxy; no VITE_API_BASE needed).
|
||||
#
|
||||
# Hybrid (Vite on host + API in Docker): publish API first, then:
|
||||
# docker compose --env-file ./backend/.env -f docker-compose.yml -f docker-compose.host-ports.yml up -d
|
||||
VITE_API_BASE=http://127.0.0.1:8000
|
||||
|
|
|
|||
|
|
@ -14,11 +14,13 @@ RUN npm ci
|
|||
|
||||
COPY . .
|
||||
|
||||
# Vite inlines VITE_* at BUILD time, so the API origin is fixed when the image is
|
||||
# built, not when the container starts — rebuild the image to point it elsewhere.
|
||||
# `.env.production.local` outranks every other env file, so this wins over the empty
|
||||
# VITE_API_BASE in .env.production (which means "same origin, behind a proxy").
|
||||
ARG VITE_API_BASE=http://172.16.204.191:8000
|
||||
# Empty VITE_API_BASE = same-origin requests. nginx.conf proxies API paths to
|
||||
# backend-api:8000, so a LAN IP baked into the bundle can no longer send the
|
||||
# browser to a different listener than the one serving the SPA (the localhost
|
||||
# vs 127.0.0.1 vs Docker split that emptied /jobs).
|
||||
# Override with --build-arg VITE_API_BASE=https://api.example.com only when the
|
||||
# API is intentionally on another origin.
|
||||
ARG VITE_API_BASE=
|
||||
RUN printf 'VITE_API_BASE=%s\n' "$VITE_API_BASE" > .env.production.local \
|
||||
&& npm run build
|
||||
|
||||
|
|
|
|||
|
|
@ -5,9 +5,53 @@ server {
|
|||
root /usr/share/nginx/html;
|
||||
index index.html;
|
||||
|
||||
# One SPA at `/`. `vite dev` and `vite preview` serve index.html for every
|
||||
# unmatched path; vite.config.js notes that a static deploy needs the equivalent
|
||||
# rewrite rule. This is it — without it /auth/confirm-email (a router path, not a
|
||||
# CV / multipart uploads (MAX_PDF_SIZE_MB is 10; leave headroom for form fields).
|
||||
client_max_body_size 25m;
|
||||
|
||||
# Security headers on every response.
|
||||
add_header X-Content-Type-Options nosniff always;
|
||||
add_header X-Frame-Options DENY always;
|
||||
add_header Referrer-Policy strict-origin-when-cross-origin always;
|
||||
|
||||
# Same-origin API proxy. The SPA is built with an empty VITE_API_BASE so
|
||||
# fetch('/jobs/fetch') stays on this host. OpenAPI (/docs, /redoc,
|
||||
# /openapi.json) is intentionally NOT proxied.
|
||||
#
|
||||
# Paths that are BOTH React routes (/jobs, /inbox, …) and API prefixes must
|
||||
# require a sub-path: otherwise a cold open / refresh of /jobs is stolen by
|
||||
# the proxy and returns a FastAPI 404 instead of index.html.
|
||||
|
||||
# SPA page roots that also prefix API calls — sub-path required.
|
||||
location ~ ^/(jobs|inbox|pipeline|tasks|assessments|offers|managers|analytics|notifications)/ {
|
||||
proxy_pass http://backend-api:8000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
proxy_set_header Authorization $http_authorization;
|
||||
proxy_connect_timeout 10s;
|
||||
proxy_send_timeout 120s;
|
||||
proxy_read_timeout 120s;
|
||||
proxy_request_buffering off;
|
||||
}
|
||||
|
||||
# API-only prefixes (no SPA page at the bare path).
|
||||
location ~ ^/(health|users|roles|permissions|permission-tags|email|job|candidate|notes|interview|feedback|activity|org-settings|saved-searches|search|documents|sheet)(/|$) {
|
||||
proxy_pass http://backend-api:8000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
proxy_set_header Authorization $http_authorization;
|
||||
proxy_connect_timeout 10s;
|
||||
proxy_send_timeout 120s;
|
||||
proxy_read_timeout 120s;
|
||||
proxy_request_buffering off;
|
||||
}
|
||||
|
||||
# One SPA at `/`. Without this, /auth/confirm-email (a router path, not a
|
||||
# file) 404s when a confirmation email link is opened cold.
|
||||
location / {
|
||||
try_files $uri $uri/ /index.html;
|
||||
|
|
@ -17,11 +61,17 @@ server {
|
|||
location /assets/ {
|
||||
expires 1y;
|
||||
add_header Cache-Control "public, immutable";
|
||||
add_header X-Content-Type-Options nosniff always;
|
||||
add_header X-Frame-Options DENY always;
|
||||
add_header Referrer-Policy strict-origin-when-cross-origin always;
|
||||
}
|
||||
|
||||
# index.html must never be cached, or a redeploy keeps serving the old asset hashes.
|
||||
location = /index.html {
|
||||
add_header Cache-Control "no-store";
|
||||
add_header X-Content-Type-Options nosniff always;
|
||||
add_header X-Frame-Options DENY always;
|
||||
add_header Referrer-Policy strict-origin-when-cross-origin always;
|
||||
}
|
||||
|
||||
gzip on;
|
||||
|
|
|
|||
|
|
@ -14,18 +14,25 @@
|
|||
|
||||
import { downloadFile, request } from '../lib/apiClient'
|
||||
|
||||
/** Active job posts for pickers. Needs job_board.view OR candidates.view. */
|
||||
export function listJobs() {
|
||||
return request('/job/fetch')
|
||||
/** Active job posts for pickers. Needs job_board.view OR candidates.view.
|
||||
*
|
||||
* `top` is explicit because /job/fetch now defaults to 10 — a picker dropdown
|
||||
* that silently showed only the 10 newest jobs would hide the rest.
|
||||
*/
|
||||
export function listJobs({ top = 100 } = {}) {
|
||||
return request('/job/fetch', { params: { top } })
|
||||
}
|
||||
|
||||
/**
|
||||
* Persisted scoring leaderboard. Needs candidates.view.
|
||||
* Omit jobId for the whole pool across jobs; rows are ordered completed-by-
|
||||
* score-desc, then failed rows.
|
||||
* Persisted scored candidates. Needs candidates.view.
|
||||
* Omit jobId for the whole pool across jobs. Rows come back newest-first by
|
||||
* created_at and PAGED (limit defaults to 10 server-side); `total` in the
|
||||
* envelope is the full result-set size, not the page length.
|
||||
*/
|
||||
export function listCandidates({ jobId } = {}) {
|
||||
return request('/candidate/scored/fetch', { params: { job_id: jobId } })
|
||||
export function listCandidates({ jobId, limit, offset } = {}) {
|
||||
return request('/candidate/scored/fetch', {
|
||||
params: { job_id: jobId, limit, offset },
|
||||
})
|
||||
}
|
||||
|
||||
/** One scored candidate row by id. Needs candidates.view. 404s on unknown ids. */
|
||||
|
|
|
|||
|
|
@ -28,7 +28,8 @@ import * as tasksApi from '../api/tasks'
|
|||
import { JOB_STATUSES } from '../api/jobs'
|
||||
import { empTypes, fmtShort } from '../data/seed'
|
||||
|
||||
const JOB_LIMIT = 200
|
||||
// Backend allows up to 500; stay at 100 so we match Managers.jsx (shared qk.jobs.list).
|
||||
const JOB_LIMIT = 100
|
||||
|
||||
async function fetchJobs() {
|
||||
const res = await jobsApi.list({ top: JOB_LIMIT })
|
||||
|
|
|
|||
|
|
@ -19,7 +19,9 @@ async function fetchManagers() {
|
|||
}
|
||||
|
||||
async function fetchJobs() {
|
||||
const res = await jobsApi.list({ top: 200 })
|
||||
// Keep within GET /jobs/fetch `top` ceiling (and match Jobs.jsx) so a shared
|
||||
// qk.jobs.list() cache entry is never poisoned by a 422 from top=200.
|
||||
const res = await jobsApi.list({ top: 100 })
|
||||
const rows = Array.isArray(res?.data) ? res.data : []
|
||||
return rows.map(jobsApi.toJobView)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -16,6 +16,16 @@ export default defineConfig({
|
|||
base: '/',
|
||||
build: { outDir: 'dist', emptyOutDir: true, sourcemap: true },
|
||||
resolve: { alias: { '@': path.resolve(__dirname, 'src') } },
|
||||
server: { port: 5173 },
|
||||
server: {
|
||||
port: 5173,
|
||||
// Same-origin style for local Vite when VITE_API_BASE is empty.
|
||||
// Requires API published on the host (docker-compose.host-ports.yml).
|
||||
proxy: {
|
||||
'^/(health|users|roles|permissions|permission-tags|email|job|jobs|candidate|notes|interview|feedback|activity|pipeline|notifications|analytics|offers|tasks|assessments|org-settings|saved-searches|search|documents|sheet|managers|inbox)(/|$)': {
|
||||
target: 'http://127.0.0.1:8000',
|
||||
changeOrigin: true,
|
||||
},
|
||||
},
|
||||
},
|
||||
preview: { port: 4173 },
|
||||
})
|
||||
|
|
|
|||
Loading…
Reference in New Issue