perf: load master data on install instead of every migrate
Move E-Taxes Item Group (~13k rows) and Classification code (~116 rows) out of the `fixtures` hook. Re-syncing these on every `bench migrate` was the dominant cost of the fixture sync step. Add invoice_az.master_data_loader.load_master_data() that bulk-inserts from the existing JSON files; it runs once via after_install and is idempotent (skips rows that already exist by primary key) so it can be re-invoked from the bench console to refresh master data. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
d9ee5f08e4
commit
6287f30409
|
|
@ -72,8 +72,10 @@ scheduler_events = {
|
||||||
def after_install():
|
def after_install():
|
||||||
"""Run installation tasks"""
|
"""Run installation tasks"""
|
||||||
from invoice_az.auth import setup_token_renewal
|
from invoice_az.auth import setup_token_renewal
|
||||||
|
from invoice_az.master_data_loader import load_master_data
|
||||||
|
|
||||||
setup_token_renewal()
|
setup_token_renewal()
|
||||||
|
load_master_data()
|
||||||
|
|
||||||
|
|
||||||
def after_migrate():
|
def after_migrate():
|
||||||
|
|
@ -84,16 +86,11 @@ def after_migrate():
|
||||||
|
|
||||||
# Fixtures for master data
|
# Fixtures for master data
|
||||||
# ------------------------
|
# ------------------------
|
||||||
fixtures = [
|
# Master data (E-Taxes Item Group ~13k rows, Classification code ~116 rows) is
|
||||||
{
|
# loaded once on `after_install` via invoice_az.master_data_loader, not on every
|
||||||
"doctype": "E-Taxes Item Group",
|
# `bench migrate`. To refresh from JSON on an existing site, call
|
||||||
"filters": []
|
# `invoice_az.master_data_loader.load_master_data()` from `bench --site <site> console`.
|
||||||
},
|
fixtures = []
|
||||||
{
|
|
||||||
"doctype": "Classification code",
|
|
||||||
"filters": []
|
|
||||||
}
|
|
||||||
]
|
|
||||||
|
|
||||||
# Apps
|
# Apps
|
||||||
# ------------------
|
# ------------------
|
||||||
|
|
|
||||||
|
|
@ -0,0 +1,82 @@
|
||||||
|
"""One-time master-data loader for E-Taxes reference doctypes.
|
||||||
|
|
||||||
|
Replaces fixture sync for ~13k E-Taxes Item Group + ~116 Classification code
|
||||||
|
rows. Loading via fixtures runs on every `bench migrate`; bulk SQL insert runs
|
||||||
|
once on `after_install`. Idempotent: skips rows that already exist by primary
|
||||||
|
key, so safe to re-run.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
|
||||||
|
import frappe
|
||||||
|
|
||||||
|
_FIXTURES_DIR = os.path.join(os.path.dirname(__file__), "fixtures")
|
||||||
|
|
||||||
|
_DATASETS = (
|
||||||
|
("E-Taxes Item Group", "e_taxes_item_group.json"),
|
||||||
|
("Classification code", "classification_code.json"),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def load_master_data():
|
||||||
|
"""Bulk-load reference data on fresh installs."""
|
||||||
|
for doctype, filename in _DATASETS:
|
||||||
|
path = os.path.join(_FIXTURES_DIR, filename)
|
||||||
|
if not os.path.exists(path):
|
||||||
|
print(f"[invoice_az] master data: {filename} not found, skipping")
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
inserted = _bulk_insert_from_json(doctype, path)
|
||||||
|
print(f"[invoice_az] master data: {doctype}: inserted {inserted} rows")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[invoice_az] master data: failed to load {doctype}: {e}")
|
||||||
|
frappe.db.rollback()
|
||||||
|
|
||||||
|
|
||||||
|
def _bulk_insert_from_json(doctype: str, path: str) -> int:
|
||||||
|
with open(path, encoding="utf-8") as f:
|
||||||
|
rows = json.load(f)
|
||||||
|
if not rows:
|
||||||
|
return 0
|
||||||
|
|
||||||
|
existing = {
|
||||||
|
r[0] for r in frappe.db.sql(f"SELECT name FROM `tab{doctype}`")
|
||||||
|
}
|
||||||
|
|
||||||
|
meta = frappe.get_meta(doctype)
|
||||||
|
field_names = {df.fieldname for df in meta.fields}
|
||||||
|
|
||||||
|
inserted = 0
|
||||||
|
chunk = []
|
||||||
|
CHUNK_SIZE = 500
|
||||||
|
|
||||||
|
for row in rows:
|
||||||
|
name = row.get("name")
|
||||||
|
if not name or name in existing:
|
||||||
|
continue
|
||||||
|
doc = frappe.new_doc(doctype)
|
||||||
|
doc.name = name
|
||||||
|
doc.flags.name_set = True
|
||||||
|
for k, v in row.items():
|
||||||
|
if k in ("doctype", "name", "modified", "docstatus"):
|
||||||
|
continue
|
||||||
|
if k in field_names:
|
||||||
|
doc.set(k, v)
|
||||||
|
chunk.append(doc)
|
||||||
|
existing.add(name)
|
||||||
|
|
||||||
|
if len(chunk) >= CHUNK_SIZE:
|
||||||
|
for d in chunk:
|
||||||
|
d.db_insert()
|
||||||
|
frappe.db.commit()
|
||||||
|
inserted += len(chunk)
|
||||||
|
chunk = []
|
||||||
|
|
||||||
|
if chunk:
|
||||||
|
for d in chunk:
|
||||||
|
d.db_insert()
|
||||||
|
frappe.db.commit()
|
||||||
|
inserted += len(chunk)
|
||||||
|
|
||||||
|
return inserted
|
||||||
Loading…
Reference in New Issue