Export records (#12887)

* Add 'bulkdumpdata' command

* Test --bulk option in CI
This commit is contained in:
Oliver
2026-09-19 19:43:09 +10:00
committed by GitHub
parent 2845a59207
commit ea553ea232
5 changed files with 193 additions and 34 deletions
+5 -4
View File
@@ -117,14 +117,15 @@ jobs:
invoke import-records -c -f ${{ env.DATA_FILE }} --strict
invoke import-records -c -f ${{ env.DATA_FILE }} --strict
cd src/backend/InvenTree && python manage.py check_dummy_data
invoke export-records -o -f ${{ env.DATA_FILE }}
invoke export-records -o -f ${{ env.DATA_FILE }} --bulk
python ../../../.github/scripts/check_exported_data.py ${{ env.DATA_FILE }}
- name: Bulk Import Sqlite Dataset
run: |
# Ensure that the 'bulk' import process works as expected
invoke import-records -c -f ${{ env.DATA_FILE }} --strict --bulk
cd src/backend/InvenTree && python manage.py check_dummy_data
invoke export-records -o -f ${{ env.DATA_FILE }} --prettify
# Ensure that the 'bulk' export process works as expected
invoke export-records -o -f ${{ env.DATA_FILE }} --prettify --bulk
python ../../../.github/scripts/check_exported_data.py ${{ env.DATA_FILE }}
content-excludes:
@@ -132,7 +133,7 @@ jobs:
# category of data (email logs, API tokens, SSO app/token data, user
# sessions, and group/user permissions) according to its --include-x /
# --exclude-x flags. Separate from the 'test' job above since it exercises
# a different axis of behaviour (export content, not the import/export
# a different axis of behavior (export content, not the import/export
# round-trip) and doesn't need the Sqlite half at all.
runs-on: ubuntu-latest
needs: paths-filter
@@ -276,7 +277,7 @@ jobs:
"
bulk-conflicts:
# Check for expected conflict behaviour when re-importing a dataset into a database that already contains that dataset.
# Check for expected conflict behavior when re-importing a dataset into a database that already contains that dataset.
runs-on: ubuntu-latest
needs: paths-filter
if: needs.paths-filter.outputs.server == 'true' || contains(github.event.pull_request.labels.*.name, 'full-run')
@@ -0,0 +1,66 @@
"""Custom management command to export fixtures faster, by caching natural-key lookups."""
from contextlib import contextmanager
from django.core.management.commands.dumpdata import Command as DumpDataCommand
from django.core.serializers import python as serializers_python
class Command(DumpDataCommand):
"""Dump fixtures using a natural-key cache for improved performance.
Behaves identically to the built-in 'dumpdata' command, with one difference:
- When --natural-foreign is used, Django's serializer resolves each natural-key
FK by fetching the full related row (getattr(obj, field.name)) with no caching
of its own - see cached_handle_fk_field(). A large export can have many rows
referencing the same handful of natural-keyed objects (e.g. thousands of
stock.StockItem rows all pointing at a few part.Part records), which otherwise
costs one query per *row* instead of one query per distinct related object.
This mirrors the caching bulkloaddata.py already does on the import side.
"""
def handle(self, *app_labels, **options):
"""Wrap the base dumpdata command with a natural-key FK resolution cache."""
with self._cached_natural_keys():
super().handle(*app_labels, **options)
@contextmanager
def _cached_natural_keys(self):
"""Cache natural-key foreign key resolutions for the duration of this block.
Keyed by (related model, raw FK id) rather than the natural key tuple
itself, so a cache hit never needs to touch the FK descriptor (and
therefore never issues a query) - only the raw id column already present
on the serialized object.
"""
cache = {}
original = serializers_python.Serializer.handle_fk_field
def cached_handle_fk_field(serializer_self, obj, field):
if not (
serializer_self.use_natural_foreign_keys
and hasattr(field.remote_field.model, 'natural_key')
):
return original(serializer_self, obj, field)
fk_id = getattr(obj, field.attname)
if fk_id is None:
serializer_self._current[field.name] = None
return
cache_key = (field.remote_field.model, fk_id)
if cache_key in cache:
serializer_self._current[field.name] = cache[cache_key]
return
original(serializer_self, obj, field)
cache[cache_key] = serializer_self._current[field.name]
serializers_python.Serializer.handle_fk_field = cached_handle_fk_field
try:
yield
finally:
serializers_python.Serializer.handle_fk_field = original
+4 -1
View File
@@ -49,7 +49,10 @@ def isWaitingForDatabase():
def isImportingData():
"""Returns True if the database is currently importing (or exporting) data, e.g. 'loaddata' command is performed."""
return any(x in sys.argv for x in ['flush', 'loaddata', 'bulkloaddata', 'dumpdata'])
return any(
x in sys.argv
for x in ['flush', 'loaddata', 'bulkloaddata', 'dumpdata', 'bulkdumpdata']
)
def isRunningMigrations():
@@ -161,6 +161,74 @@ class CommandTestCase(TestCase):
ContentType.objects.filter(pk__in=pks).delete()
tmp_file.unlink(missing_ok=True)
def test_bulkdumpdata_natural_key_caching(self):
"""Test that bulkdumpdata caches natural-key FK resolution during serialization."""
from django.contrib.admin.models import ADDITION, LogEntry
from django.contrib.contenttypes.models import ContentType
from django.core import serializers
from django.db import connection
from django.test.utils import CaptureQueriesContext
from InvenTree.management.commands.bulkdumpdata import (
Command as BulkDumpDataCommand,
)
user = User.objects.create_user(username='bulkdumpdata_test_user')
content_type = ContentType.objects.create(
app_label='bulkdumpdata_test', model='dummymodel'
)
entries = [
LogEntry.objects.create(
user=user,
content_type=content_type,
object_id=str(i),
object_repr=f'Object {i}',
action_flag=ADDITION,
change_message='Created',
)
for i in range(20)
]
pks = [e.pk for e in entries]
def make_queryset():
# A fresh queryset each time, so FK descriptor caching on the
# instances themselves can't mask whether *our* cache is doing
# the work
return LogEntry.objects.filter(pk__in=pks).order_by('pk')
try:
with CaptureQueriesContext(connection) as uncached:
uncached_data = serializers.serialize(
'json', make_queryset(), use_natural_foreign_keys=True
)
with CaptureQueriesContext(connection) as cached:
with BulkDumpDataCommand()._cached_natural_keys():
cached_data = serializers.serialize(
'json', make_queryset(), use_natural_foreign_keys=True
)
# Same output either way - caching must not change what gets exported
self.assertEqual(uncached_data, cached_data)
# Without caching: one extra query per row for each repeated
# natural-keyed FK (content_type and user are both natural-keyed
# here, so up to 2 extra queries per row -> 40, plus the main select)
self.assertGreaterEqual(len(uncached.captured_queries), 40)
# With caching: only the first reference to each distinct related
# object (one content_type, one user) issues a query - every other
# row is served from cache
self.assertLessEqual(len(cached.captured_queries), 4)
self.assertLess(
len(cached.captured_queries), len(uncached.captured_queries)
)
finally:
LogEntry.objects.filter(pk__in=pks).delete()
content_type.delete()
user.delete()
def test_backup_metadata(self):
"""Test the backup metadata functions."""
from InvenTree.backup import (
+28 -7
View File
@@ -1206,6 +1206,7 @@ def update(
'include_sso': 'Include SSO token data in the output file (default = False)',
'include_session': 'Include user session data in the output file (default = False)',
'prettify': 'Pretty-print the output file with indentation (default = False)',
'bulk': 'Use bulkdumpdata for improved performance on large datasets (default = False)',
'verbose': 'Print verbose output from management commands',
}
)
@@ -1221,6 +1222,7 @@ def export_records(
include_sso: bool = False,
include_session: bool = False,
prettify: bool = False,
bulk: bool = False,
verbose: bool = False,
):
"""Export all database records to a file."""
@@ -1246,7 +1248,7 @@ def export_records(
with tempfile.NamedTemporaryFile(
suffix='.json', encoding='utf-8', mode='w+t', delete=True
) as tmpfile:
cmd = f"dumpdata --natural-foreign --output '{tmpfile.name}' {excludes}"
cmd = f"{'bulkdumpdata' if bulk else 'dumpdata'} --natural-foreign --output '{tmpfile.name}' {excludes}"
if prettify:
cmd += ' --indent 2'
@@ -1259,8 +1261,7 @@ def export_records(
tmpfile.seek(0)
data = json.loads(tmpfile.read())
data_out = [
{
metadata_entry = {
'metadata': True,
'comment': 'This file contains a dump of the InvenTree database',
'exported_at': datetime.datetime.now().isoformat(),
@@ -1272,7 +1273,14 @@ def export_records(
'source_commit': get_commit_hash(),
'installed_apps': installed_apps(c),
}
]
def entries_out():
"""Filter and adjust entries as they are written, without ever materializing a second copy of the entire (potentially huge) dataset in memory.
Yields:
dict: The next entry to write to the output file.
"""
yield metadata_entry
for entry in data:
model_name = entry.get('model', None)
@@ -1289,11 +1297,24 @@ def export_records(
if model_name == 'auth.user':
entry['fields']['user_permissions'] = []
data_out.append(entry)
yield entry
# Write the processed data to file
indent = 2 if prettify else None
# Write the processed data to file, one entry at a time - avoids ever
# holding a second full copy of the (potentially huge) dataset in memory,
# and avoids a single json.dumps() call across the entire dataset at once
with open(target, 'w', encoding='utf-8') as f_out:
f_out.write(json.dumps(data_out, indent=2 if prettify else None))
f_out.write('[')
for i, entry in enumerate(entries_out()):
if i:
f_out.write(',')
if prettify:
f_out.write('\n')
f_out.write(json.dumps(entry, indent=indent))
if prettify:
f_out.write('\n')
f_out.write(']')
success('Data export completed')