diff --git a/.github/workflows/import_export.yaml b/.github/workflows/import_export.yaml index b984533d89..9d6bd645b4 100644 --- a/.github/workflows/import_export.yaml +++ b/.github/workflows/import_export.yaml @@ -117,14 +117,15 @@ jobs: invoke import-records -c -f ${{ env.DATA_FILE }} --strict invoke import-records -c -f ${{ env.DATA_FILE }} --strict cd src/backend/InvenTree && python manage.py check_dummy_data - invoke export-records -o -f ${{ env.DATA_FILE }} + invoke export-records -o -f ${{ env.DATA_FILE }} --bulk python ../../../.github/scripts/check_exported_data.py ${{ env.DATA_FILE }} - name: Bulk Import Sqlite Dataset run: | # Ensure that the 'bulk' import process works as expected invoke import-records -c -f ${{ env.DATA_FILE }} --strict --bulk cd src/backend/InvenTree && python manage.py check_dummy_data - invoke export-records -o -f ${{ env.DATA_FILE }} --prettify + # Ensure that the 'bulk' export process works as expected + invoke export-records -o -f ${{ env.DATA_FILE }} --prettify --bulk python ../../../.github/scripts/check_exported_data.py ${{ env.DATA_FILE }} content-excludes: @@ -132,7 +133,7 @@ jobs: # category of data (email logs, API tokens, SSO app/token data, user # sessions, and group/user permissions) according to its --include-x / # --exclude-x flags. Separate from the 'test' job above since it exercises - # a different axis of behaviour (export content, not the import/export + # a different axis of behavior (export content, not the import/export # round-trip) and doesn't need the Sqlite half at all. runs-on: ubuntu-latest needs: paths-filter @@ -276,7 +277,7 @@ jobs: " bulk-conflicts: - # Check for expected conflict behaviour when re-importing a dataset into a database that already contains that dataset. + # Check for expected conflict behavior when re-importing a dataset into a database that already contains that dataset. runs-on: ubuntu-latest needs: paths-filter if: needs.paths-filter.outputs.server == 'true' || contains(github.event.pull_request.labels.*.name, 'full-run') diff --git a/src/backend/InvenTree/InvenTree/management/commands/bulkdumpdata.py b/src/backend/InvenTree/InvenTree/management/commands/bulkdumpdata.py new file mode 100644 index 0000000000..464dfa45a6 --- /dev/null +++ b/src/backend/InvenTree/InvenTree/management/commands/bulkdumpdata.py @@ -0,0 +1,66 @@ +"""Custom management command to export fixtures faster, by caching natural-key lookups.""" + +from contextlib import contextmanager + +from django.core.management.commands.dumpdata import Command as DumpDataCommand +from django.core.serializers import python as serializers_python + + +class Command(DumpDataCommand): + """Dump fixtures using a natural-key cache for improved performance. + + Behaves identically to the built-in 'dumpdata' command, with one difference: + + - When --natural-foreign is used, Django's serializer resolves each natural-key + FK by fetching the full related row (getattr(obj, field.name)) with no caching + of its own - see cached_handle_fk_field(). A large export can have many rows + referencing the same handful of natural-keyed objects (e.g. thousands of + stock.StockItem rows all pointing at a few part.Part records), which otherwise + costs one query per *row* instead of one query per distinct related object. + This mirrors the caching bulkloaddata.py already does on the import side. + """ + + def handle(self, *app_labels, **options): + """Wrap the base dumpdata command with a natural-key FK resolution cache.""" + with self._cached_natural_keys(): + super().handle(*app_labels, **options) + + @contextmanager + def _cached_natural_keys(self): + """Cache natural-key foreign key resolutions for the duration of this block. + + Keyed by (related model, raw FK id) rather than the natural key tuple + itself, so a cache hit never needs to touch the FK descriptor (and + therefore never issues a query) - only the raw id column already present + on the serialized object. + """ + cache = {} + original = serializers_python.Serializer.handle_fk_field + + def cached_handle_fk_field(serializer_self, obj, field): + if not ( + serializer_self.use_natural_foreign_keys + and hasattr(field.remote_field.model, 'natural_key') + ): + return original(serializer_self, obj, field) + + fk_id = getattr(obj, field.attname) + + if fk_id is None: + serializer_self._current[field.name] = None + return + + cache_key = (field.remote_field.model, fk_id) + + if cache_key in cache: + serializer_self._current[field.name] = cache[cache_key] + return + + original(serializer_self, obj, field) + cache[cache_key] = serializer_self._current[field.name] + + serializers_python.Serializer.handle_fk_field = cached_handle_fk_field + try: + yield + finally: + serializers_python.Serializer.handle_fk_field = original diff --git a/src/backend/InvenTree/InvenTree/ready.py b/src/backend/InvenTree/InvenTree/ready.py index 3fe70bfd71..0e1fda2607 100644 --- a/src/backend/InvenTree/InvenTree/ready.py +++ b/src/backend/InvenTree/InvenTree/ready.py @@ -49,7 +49,10 @@ def isWaitingForDatabase(): def isImportingData(): """Returns True if the database is currently importing (or exporting) data, e.g. 'loaddata' command is performed.""" - return any(x in sys.argv for x in ['flush', 'loaddata', 'bulkloaddata', 'dumpdata']) + return any( + x in sys.argv + for x in ['flush', 'loaddata', 'bulkloaddata', 'dumpdata', 'bulkdumpdata'] + ) def isRunningMigrations(): diff --git a/src/backend/InvenTree/InvenTree/test_commands.py b/src/backend/InvenTree/InvenTree/test_commands.py index a63537b33a..b36cf9ef32 100644 --- a/src/backend/InvenTree/InvenTree/test_commands.py +++ b/src/backend/InvenTree/InvenTree/test_commands.py @@ -161,6 +161,74 @@ class CommandTestCase(TestCase): ContentType.objects.filter(pk__in=pks).delete() tmp_file.unlink(missing_ok=True) + def test_bulkdumpdata_natural_key_caching(self): + """Test that bulkdumpdata caches natural-key FK resolution during serialization.""" + from django.contrib.admin.models import ADDITION, LogEntry + from django.contrib.contenttypes.models import ContentType + from django.core import serializers + from django.db import connection + from django.test.utils import CaptureQueriesContext + + from InvenTree.management.commands.bulkdumpdata import ( + Command as BulkDumpDataCommand, + ) + + user = User.objects.create_user(username='bulkdumpdata_test_user') + content_type = ContentType.objects.create( + app_label='bulkdumpdata_test', model='dummymodel' + ) + + entries = [ + LogEntry.objects.create( + user=user, + content_type=content_type, + object_id=str(i), + object_repr=f'Object {i}', + action_flag=ADDITION, + change_message='Created', + ) + for i in range(20) + ] + pks = [e.pk for e in entries] + + def make_queryset(): + # A fresh queryset each time, so FK descriptor caching on the + # instances themselves can't mask whether *our* cache is doing + # the work + return LogEntry.objects.filter(pk__in=pks).order_by('pk') + + try: + with CaptureQueriesContext(connection) as uncached: + uncached_data = serializers.serialize( + 'json', make_queryset(), use_natural_foreign_keys=True + ) + + with CaptureQueriesContext(connection) as cached: + with BulkDumpDataCommand()._cached_natural_keys(): + cached_data = serializers.serialize( + 'json', make_queryset(), use_natural_foreign_keys=True + ) + + # Same output either way - caching must not change what gets exported + self.assertEqual(uncached_data, cached_data) + + # Without caching: one extra query per row for each repeated + # natural-keyed FK (content_type and user are both natural-keyed + # here, so up to 2 extra queries per row -> 40, plus the main select) + self.assertGreaterEqual(len(uncached.captured_queries), 40) + + # With caching: only the first reference to each distinct related + # object (one content_type, one user) issues a query - every other + # row is served from cache + self.assertLessEqual(len(cached.captured_queries), 4) + self.assertLess( + len(cached.captured_queries), len(uncached.captured_queries) + ) + finally: + LogEntry.objects.filter(pk__in=pks).delete() + content_type.delete() + user.delete() + def test_backup_metadata(self): """Test the backup metadata functions.""" from InvenTree.backup import ( diff --git a/tasks.py b/tasks.py index 8c583d7ff8..5507112747 100644 --- a/tasks.py +++ b/tasks.py @@ -1206,6 +1206,7 @@ def update( 'include_sso': 'Include SSO token data in the output file (default = False)', 'include_session': 'Include user session data in the output file (default = False)', 'prettify': 'Pretty-print the output file with indentation (default = False)', + 'bulk': 'Use bulkdumpdata for improved performance on large datasets (default = False)', 'verbose': 'Print verbose output from management commands', } ) @@ -1221,6 +1222,7 @@ def export_records( include_sso: bool = False, include_session: bool = False, prettify: bool = False, + bulk: bool = False, verbose: bool = False, ): """Export all database records to a file.""" @@ -1246,7 +1248,7 @@ def export_records( with tempfile.NamedTemporaryFile( suffix='.json', encoding='utf-8', mode='w+t', delete=True ) as tmpfile: - cmd = f"dumpdata --natural-foreign --output '{tmpfile.name}' {excludes}" + cmd = f"{'bulkdumpdata' if bulk else 'dumpdata'} --natural-foreign --output '{tmpfile.name}' {excludes}" if prettify: cmd += ' --indent 2' @@ -1259,41 +1261,60 @@ def export_records( tmpfile.seek(0) data = json.loads(tmpfile.read()) - data_out = [ - { - 'metadata': True, - 'comment': 'This file contains a dump of the InvenTree database', - 'exported_at': datetime.datetime.now().isoformat(), - 'exported_at_utc': datetime.datetime.now(datetime.UTC).isoformat(), - 'source_version': get_inventree_version(), - 'api_version': get_inventree_api_version(), - 'django_version': get_django_version(), - 'python_version': python_version(), - 'source_commit': get_commit_hash(), - 'installed_apps': installed_apps(c), - } - ] + metadata_entry = { + 'metadata': True, + 'comment': 'This file contains a dump of the InvenTree database', + 'exported_at': datetime.datetime.now().isoformat(), + 'exported_at_utc': datetime.datetime.now(datetime.UTC).isoformat(), + 'source_version': get_inventree_version(), + 'api_version': get_inventree_api_version(), + 'django_version': get_django_version(), + 'python_version': python_version(), + 'source_commit': get_commit_hash(), + 'installed_apps': installed_apps(c), + } - for entry in data: - model_name = entry.get('model', None) + def entries_out(): + """Filter and adjust entries as they are written, without ever materializing a second copy of the entire (potentially huge) dataset in memory. - # Ignore any temporary settings (start with underscore) - if model_name in ['common.inventreesetting', 'common.inventreeusersetting']: - if entry['fields'].get('key', '').startswith('_'): - continue + Yields: + dict: The next entry to write to the output file. + """ + yield metadata_entry - if include_permissions is False: - if model_name == 'auth.group': - entry['fields']['permissions'] = [] + for entry in data: + model_name = entry.get('model', None) - if model_name == 'auth.user': - entry['fields']['user_permissions'] = [] + # Ignore any temporary settings (start with underscore) + if model_name in ['common.inventreesetting', 'common.inventreeusersetting']: + if entry['fields'].get('key', '').startswith('_'): + continue - data_out.append(entry) + if include_permissions is False: + if model_name == 'auth.group': + entry['fields']['permissions'] = [] - # Write the processed data to file + if model_name == 'auth.user': + entry['fields']['user_permissions'] = [] + + yield entry + + indent = 2 if prettify else None + + # Write the processed data to file, one entry at a time - avoids ever + # holding a second full copy of the (potentially huge) dataset in memory, + # and avoids a single json.dumps() call across the entire dataset at once with open(target, 'w', encoding='utf-8') as f_out: - f_out.write(json.dumps(data_out, indent=2 if prettify else None)) + f_out.write('[') + for i, entry in enumerate(entries_out()): + if i: + f_out.write(',') + if prettify: + f_out.write('\n') + f_out.write(json.dumps(entry, indent=indent)) + if prettify: + f_out.write('\n') + f_out.write(']') success('Data export completed')