card-grader/store.py
Barely Removable 3270a9fe69 Harden: cap request body before reading, WAL for concurrent writes, socket timeout, surface unmeasured edges
- _body() read the full declared Content-Length into memory before any size
  check, so MAX_UPLOAD_BYTES could only reject an upload already held in
  RAM. nginx caps this on the proxied path, but the container also listens
  on the LAN, so the app now enforces its own 20MB ceiling and closes the
  connection rather than reading.
- SQLite ran with the default rollback journal and 5s lock timeout, chosen
  when a row was a few KB; rows now carry the original photos, so two
  people grading at once could block each other's History. WAL + 30s.
- No socket timeout meant a stalled keep-alive connection held a worker
  thread indefinitely.
- An edge excluded as a different material was dropped from the UI with no
  explanation, presenting three sides as though they were all four.
2026-08-22 12:56:58 -07:00

348 lines
12 KiB
Python

"""SQLite: settings and a log of past grade estimates.
One database, two tables. There is no inventory or pricing here — this app
does exactly one thing (estimate a PSA grade from photos) and remembers what
it told you, so you can look back at a card without re-running the estimate.
"""
import base64
import json
import os
import sqlite3
import time
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
# Overridable so a container can point this at a mounted volume (e.g.
# /data/grades.db) instead of the app's own directory, which is what makes
# the data survive a container recreate/image update.
DB_PATH = os.environ.get("CARD_GRADER_DB_PATH") or os.path.join(BASE_DIR, "grades.db")
SCHEMA = """
CREATE TABLE IF NOT EXISTS settings (
key TEXT PRIMARY KEY,
value TEXT
);
CREATE TABLE IF NOT EXISTS grades (
id INTEGER PRIMARY KEY AUTOINCREMENT,
created_at TEXT NOT NULL,
label TEXT, -- your own name for the card, optional
card_type TEXT, -- pokemon | sports | other_tcg | other
card_note TEXT, -- what the model read off the card
image_count INTEGER,
thumbnail TEXT, -- small JPEG, base64 — first photo only
model TEXT,
estimated_grade INTEGER,
grade_low INTEGER,
grade_high INTEGER,
confidence TEXT,
categories_json TEXT,
edge_measurements_json TEXT,
centering_measurement_json TEXT,
aspect_measurement_json TEXT,
limitations_json TEXT,
note TEXT,
estimated_cost REAL,
usage_json TEXT,
source_images_json TEXT -- the original photo(s), for Regrade; NOT sent
-- to the browser in list/detail responses —
-- see get_grade_images vs get_grade/list_grades
);
CREATE INDEX IF NOT EXISTS idx_grades_created ON grades(created_at DESC);
"""
DEFAULT_SETTINGS = {
"anthropic_api_key": "",
"vision_model": "claude-sonnet-5",
"vision_effort": "low",
}
def connect():
# timeout: how long to wait for a writer's lock before giving up. The
# server is threaded, so two people grading at once genuinely collide —
# and since a row now carries the original photos, a write can be
# several megabytes and hold the lock long enough to matter. The 5s
# default was chosen for small writes; this isn't that any more.
conn = sqlite3.connect(DB_PATH, timeout=30.0)
conn.row_factory = sqlite3.Row
conn.execute("PRAGMA foreign_keys = ON")
# WAL lets readers carry on during a write instead of blocking on it,
# which is the difference between "someone else is grading" being
# invisible and it freezing everyone's History. Persists on the database
# file itself, so setting it per-connection is just belt-and-braces.
conn.execute("PRAGMA journal_mode = WAL")
conn.execute("PRAGMA synchronous = NORMAL")
return conn
def init():
conn = connect()
try:
conn.executescript(SCHEMA)
# CREATE TABLE IF NOT EXISTS never touches an already-existing table,
# so a column added after cards were already graded needs its own
# migration — guarded because re-running this against a database
# that already has the column would otherwise error every startup.
for statement in (
"ALTER TABLE grades ADD COLUMN aspect_measurement_json TEXT",
"ALTER TABLE grades ADD COLUMN source_images_json TEXT",
):
try:
conn.execute(statement)
except sqlite3.OperationalError:
pass
for key, value in DEFAULT_SETTINGS.items():
conn.execute(
"INSERT OR IGNORE INTO settings (key, value) VALUES (?, ?)",
(key, json.dumps(value)),
)
conn.commit()
finally:
conn.close()
def now():
return time.strftime("%Y-%m-%dT%H:%M:%S")
# --------------------------------------------------------------- settings
def get_settings():
conn = connect()
try:
rows = conn.execute("SELECT key, value FROM settings").fetchall()
finally:
conn.close()
out = dict(DEFAULT_SETTINGS)
for row in rows:
try:
out[row["key"]] = json.loads(row["value"])
except (ValueError, TypeError):
out[row["key"]] = row["value"]
return out
def save_settings(updates):
conn = connect()
try:
for key, value in updates.items():
if key not in DEFAULT_SETTINGS:
continue
conn.execute(
"INSERT INTO settings (key, value) VALUES (?, ?) "
"ON CONFLICT(key) DO UPDATE SET value=excluded.value",
(key, json.dumps(value)),
)
conn.commit()
finally:
conn.close()
return get_settings()
# ------------------------------------------------------------------ grades
def _encode_images(images):
"""(bytes, filename) pairs -> the JSON text stored in source_images_json."""
if not images:
return None
return json.dumps([
{"filename": name, "image_base64": base64.standard_b64encode(b).decode("ascii")}
for b, name in images
])
def _decode_images(raw):
if not raw:
return None
try:
items = json.loads(raw)
except (ValueError, TypeError):
return None
return [(base64.standard_b64decode(item["image_base64"]), item.get("filename") or "upload")
for item in items]
def save_grade(grade, thumbnail=None, label=None, source_images=None):
"""Persist one grading result. Returns the new row's id.
`source_images` are the original (bytes, filename) pairs that produced
this grade, kept so Regrade can re-run without asking for the photo(s)
again. Optional — a caller that skips this still gets everything else;
Regrade just falls back to prompting for photos on that row.
"""
conn = connect()
try:
cur = conn.execute(
"INSERT INTO grades "
"(created_at, label, card_type, card_note, image_count, thumbnail, "
" model, estimated_grade, "
" grade_low, grade_high, confidence, categories_json, "
" edge_measurements_json, centering_measurement_json, "
" aspect_measurement_json, "
" limitations_json, note, estimated_cost, usage_json, "
" source_images_json) "
"VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
(
now(), label, grade.get("card_type"), grade.get("card_note"),
grade.get("image_count"), thumbnail,
(grade.get("usage") or {}).get("model"),
grade.get("estimated_grade"), grade.get("grade_low"),
grade.get("grade_high"), grade.get("confidence"),
json.dumps(grade.get("categories")),
json.dumps(grade.get("edge_measurements")),
json.dumps(grade.get("centering_measurement")),
json.dumps(grade.get("aspect_measurement")),
json.dumps(grade.get("limitations")),
grade.get("note"), grade.get("estimated_cost"),
json.dumps(grade.get("usage")),
_encode_images(source_images),
),
)
conn.commit()
return cur.lastrowid
finally:
conn.close()
def update_grade_result(grade_id, grade, thumbnail=None, source_images=None):
"""Overwrite a grade's result in place (Regrade) — label is untouched.
created_at is bumped to now, so the card floats back to the top of
History as the most recently-active one, same as if it were freshly
graded. `source_images` only overwrites the stored photo(s) when the
caller actually supplies new ones (the re-pick fallback path); passing
None leaves whatever was already stored for this row alone.
"""
conn = connect()
try:
params = [
now(), grade.get("card_type"), grade.get("card_note"),
grade.get("image_count"),
(grade.get("usage") or {}).get("model"),
grade.get("estimated_grade"), grade.get("grade_low"),
grade.get("grade_high"), grade.get("confidence"),
json.dumps(grade.get("categories")),
json.dumps(grade.get("edge_measurements")),
json.dumps(grade.get("centering_measurement")),
json.dumps(grade.get("aspect_measurement")),
json.dumps(grade.get("limitations")),
grade.get("note"), grade.get("estimated_cost"),
json.dumps(grade.get("usage")),
]
sql = (
"UPDATE grades SET created_at=?, card_type=?, card_note=?, "
"image_count=?, model=?, estimated_grade=?, grade_low=?, "
"grade_high=?, confidence=?, categories_json=?, "
"edge_measurements_json=?, centering_measurement_json=?, "
"aspect_measurement_json=?, limitations_json=?, note=?, "
"estimated_cost=?, usage_json=?"
)
if thumbnail is not None:
sql += ", thumbnail=?"
params.append(thumbnail)
if source_images is not None:
sql += ", source_images_json=?"
params.append(_encode_images(source_images))
sql += " WHERE id=?"
params.append(grade_id)
conn.execute(sql, params)
conn.commit()
finally:
conn.close()
return get_grade(grade_id)
def get_grade_images(grade_id):
"""The original (bytes, filename) pairs for Regrade, or None if this
grade never had them stored (a card graded before this feature existed,
or one saved without the images path). Server-side use only — never
sent to the browser, unlike everything get_grade/list_grades return."""
conn = connect()
try:
row = conn.execute(
"SELECT source_images_json FROM grades WHERE id = ?", (grade_id,)
).fetchone()
finally:
conn.close()
return _decode_images(row["source_images_json"]) if row else None
def _row_to_grade(row):
d = dict(row)
for key in ("categories_json", "edge_measurements_json",
"centering_measurement_json", "aspect_measurement_json",
"limitations_json", "usage_json"):
out_key = key[:-len("_json")]
raw = d.pop(key, None)
try:
d[out_key] = json.loads(raw) if raw else None
except (ValueError, TypeError):
d[out_key] = None
# The raw photo(s) are only ever for the regrade endpoint to read
# server-side (see get_grade_images) — swapping this for a boolean here
# keeps every browser-facing response light, the same reason thumbnails
# are pre-shrunk rather than sending the original photo for display.
d["has_source_images"] = bool(d.pop("source_images_json", None))
return d
# Every column except source_images_json — that one is only ever read
# through get_grade_images, so a plain SELECT * here would pull the full
# original photo(s) off disk for every row just to discard them a moment
# later in _row_to_grade, silently defeating the whole point of keeping
# list/detail responses light.
_LIST_COLUMNS = (
"id, created_at, label, card_type, card_note, image_count, thumbnail, "
"model, estimated_grade, grade_low, grade_high, confidence, "
"categories_json, edge_measurements_json, centering_measurement_json, "
"aspect_measurement_json, limitations_json, note, estimated_cost, "
"usage_json, (source_images_json IS NOT NULL) AS source_images_json"
)
def list_grades(limit=200):
conn = connect()
try:
rows = conn.execute(
"SELECT {} FROM grades ORDER BY created_at DESC, id DESC LIMIT ?"
.format(_LIST_COLUMNS),
(limit,),
).fetchall()
finally:
conn.close()
return [_row_to_grade(r) for r in rows]
def get_grade(grade_id):
conn = connect()
try:
row = conn.execute(
"SELECT {} FROM grades WHERE id = ?".format(_LIST_COLUMNS),
(grade_id,),
).fetchone()
finally:
conn.close()
return _row_to_grade(row) if row else None
def update_grade_label(grade_id, label):
conn = connect()
try:
conn.execute("UPDATE grades SET label = ? WHERE id = ?", (label, grade_id))
conn.commit()
finally:
conn.close()
return get_grade(grade_id)
def delete_grade(grade_id):
conn = connect()
try:
conn.execute("DELETE FROM grades WHERE id = ?", (grade_id,))
conn.commit()
finally:
conn.close()