From dc9f2ab722f70abc362a3789a8275f3693fb82d4 Mon Sep 17 00:00:00 2001 From: Alan Peixinho Date: Wed, 9 Sep 2026 16:43:20 -0300 Subject: [PATCH 01/11] feat(backend): add commits and commit_parents schema (#2089) * Add `Commits` model: unique `git_commit_hash`, nullable author/committer/subject/message/`fetched_from_url` * Add `CommitParents` model: FKs to `commits.id`, `ord` with 0 = first parent, unique `(commit_id, ord)` * Add migration `0021_commits_and_commit_parents` (`commits`, `commit_parents` tables and indexes) * Join from `checkouts.git_commit_hash` without new checkout columns Signed-off-by: Alan Peixinho --- .../0021_commits_and_commit_parents.py | 69 +++++++++++++++++++ backend/kernelCI_app/models.py | 49 +++++++++++++ 2 files changed, 118 insertions(+) create mode 100644 backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py diff --git a/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py b/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py new file mode 100644 index 000000000..db1d7fb03 --- /dev/null +++ b/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py @@ -0,0 +1,69 @@ +# Generated by Django 5.2.16 on 2026-09-09 19:33 + +import django.db.models.deletion +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [ + ("kernelCI_app", "0020_add_labs_fk_indexes"), + ] + + operations = [ + migrations.CreateModel( + name="Commits", + fields=[ + ("id", models.AutoField(primary_key=True, serialize=False)), + ("git_commit_hash", models.TextField(unique=True)), + ("author_name", models.TextField(blank=True, null=True)), + ("author_email", models.TextField(blank=True, null=True)), + ("author_date", models.DateTimeField(blank=True, null=True)), + ("committer_name", models.TextField(blank=True, null=True)), + ("committer_email", models.TextField(blank=True, null=True)), + ("committer_date", models.DateTimeField(blank=True, null=True)), + ("subject", models.TextField(blank=True, null=True)), + ("message", models.TextField(blank=True, null=True)), + ("fetched_from_url", models.TextField(blank=True, null=True)), + ], + options={ + "db_table": "commits", + }, + ), + migrations.CreateModel( + name="CommitParents", + fields=[ + ("id", models.AutoField(primary_key=True, serialize=False)), + ("ord", models.SmallIntegerField()), + ( + "commit", + models.ForeignKey( + db_index=False, + on_delete=django.db.models.deletion.CASCADE, + related_name="parent_edges", + to="kernelCI_app.commits", + ), + ), + ( + "parent", + models.ForeignKey( + db_index=False, + on_delete=django.db.models.deletion.CASCADE, + related_name="child_edges", + to="kernelCI_app.commits", + ), + ), + ], + options={ + "db_table": "commit_parents", + "indexes": [ + models.Index(fields=["parent"], name="commit_parents_parent_id"), + ], + "constraints": [ + models.UniqueConstraint( + fields=("commit", "ord"), + name="commit_parents_commit_ord", + ), + ], + }, + ), + ] diff --git a/backend/kernelCI_app/models.py b/backend/kernelCI_app/models.py index f3e259dc9..7d48e6372 100644 --- a/backend/kernelCI_app/models.py +++ b/backend/kernelCI_app/models.py @@ -109,6 +109,55 @@ class Meta: ] +class Commits(models.Model): + id = models.AutoField(primary_key=True) + git_commit_hash = models.TextField(unique=True) + author_name = models.TextField(blank=True, null=True) + author_email = models.TextField(blank=True, null=True) + author_date = models.DateTimeField(blank=True, null=True) + committer_name = models.TextField(blank=True, null=True) + committer_email = models.TextField(blank=True, null=True) + committer_date = models.DateTimeField(blank=True, null=True) + subject = models.TextField(blank=True, null=True) + message = models.TextField(blank=True, null=True) + fetched_from_url = models.TextField(blank=True, null=True) + + class Meta: + db_table = "commits" + + def __str__(self) -> str: + return self.git_commit_hash + + +class CommitParents(models.Model): + id = models.AutoField(primary_key=True) + commit = models.ForeignKey( + Commits, + on_delete=models.CASCADE, + related_name="parent_edges", + db_index=False, + ) + parent = models.ForeignKey( + Commits, + on_delete=models.CASCADE, + related_name="child_edges", + db_index=False, + ) + ord = models.SmallIntegerField() + + class Meta: + db_table = "commit_parents" + constraints = [ + models.UniqueConstraint( + fields=["commit", "ord"], + name="commit_parents_commit_ord", + ), + ] + indexes = [ + models.Index(fields=["parent"], name="commit_parents_parent_id"), + ] + + class Builds(models.Model): field_timestamp = models.DateTimeField( db_column="_timestamp", blank=True, null=True From 4ef6f7fd403bc4d02679cf908f3774bbd78e9f3d Mon Sep 17 00:00:00 2001 From: Alan Peixinho Date: Wed, 9 Sep 2026 18:06:44 -0300 Subject: [PATCH 02/11] feat(backend): snapshot and restore commits in update_db Dump commits and commit_parents in full so git ancestry survives restore. Signed-off-by: Alan Peixinho --- backend/docs/update_db command.md | 12 +- .../management/commands/update_db.py | 128 +++++++++++++++++- 2 files changed, 134 insertions(+), 6 deletions(-) diff --git a/backend/docs/update_db command.md b/backend/docs/update_db command.md index 8acf81730..966b2038b 100644 --- a/backend/docs/update_db command.md +++ b/backend/docs/update_db command.md @@ -1,6 +1,6 @@ # update_db Command Documentation -The `update_db` command migrates data from the default database (kcidb) to the dashboard_db database within a specified time interval. All tables are updated by default, but you can select a specific one as well. +The `update_db` command snapshots dashboard tables to a `.tar.gz` and restores them. Most tables are limited to `--start-interval` / `--end-interval`. `commits` and `commit_parents` are copied in full for now (no time filter), so git ancestry is not cut when a parent has no checkout in the window. A later change may slice those tables. The migration preserves foreign key constraints. For example, if a test A references a build B in kcidb, but the build B doesn't exist in dashboard_db, then the test A will not be inserted in dashboard_db. @@ -8,13 +8,13 @@ The migration preserves foreign key constraints. For example, if a test A refere ### Required Parameters -- `--start-interval`: Start interval for filtering data (format: 'x days' or 'x hours'). The format follows the SQL filtering format. -- `--end-interval`: End interval for filtering data (format: 'x days' or 'x hours'). The format follows the SQL filtering format. +- `--start-interval`: Start interval for filtering data (format: 'x days' or 'x hours'). The format follows the SQL filtering format. Does not apply to `commits` or `commit_parents`. +- `--end-interval`: End interval for filtering data (format: 'x days' or 'x hours'). The format follows the SQL filtering format. Does not apply to `commits` or `commit_parents`. ### Optional Parameters - `--table`: Limit data copy to a specific table - - Valid options: `issues`, `checkouts`, `builds`, `tests`, `incidents` + - Valid options: `issues`, `checkouts`, `commits`, `commit_parents`, `builds`, `tests`, `incidents`, `latest_checkout`, `hardware_status`, `tree_listing`, `tree_tests_rollup` - If not provided, data from all tables will be copied - `--related-data-only`: Limits the selected data to data where the foreign key constraint is not broken. - Default: False. @@ -35,7 +35,7 @@ python manage.py update_db --start-interval "1 days" --end-interval "0 days" --t ## Migration Process -1. **Data Selection**: Selects records from the default database within the specified time range +1. **Data Selection**: Selects records from the default database within the specified time range, except `commits` and `commit_parents` (full table) 2. **Relationship Validation**: Ensures foreign key constraints are maintained 3. **Data Insertion**: Inserts valid data into the dashboard db 4. **Conflict Resolution**: Uses `ignore_conflicts=True` to handle duplicate records @@ -44,6 +44,8 @@ python manage.py update_db --start-interval "1 days" --end-interval "0 days" --t - Migration preserves JSON fields by parsing them appropriately - The skipped rows count are related to rows which didn't have relationships in dashboard_db. The processed rows count are related to the remaining rows that were selected but not skipped (even if they were inserted or had a conflict, which is how django returns the `bulk_create` result) +- `commits` and `commit_parents` are not filtered by time or origin. The full git graph is dumped so ancestors without a checkout in the window are kept. Time slicing may be added later. +- Surrogate ids are kept as in the source. Restore runs `setval` to `MAX(id)` so the next insert does not collide. ## Performance Considerations diff --git a/backend/kernelCI_app/management/commands/update_db.py b/backend/kernelCI_app/management/commands/update_db.py index 75b364a25..3be7d2dfd 100644 --- a/backend/kernelCI_app/management/commands/update_db.py +++ b/backend/kernelCI_app/management/commands/update_db.py @@ -18,6 +18,8 @@ from kernelCI_app.models import ( Builds, Checkouts, + CommitParents, + Commits, HardwareStatus, Incidents, Issues, @@ -143,7 +145,8 @@ def _invalid_table_error(self, table: str) -> str: return ( f"Unknown table '{table}'.\n" "\tValid options are: issues, checkouts, builds, tests, incidents, " - "latest_checkout, hardware_status, tree_listing, tree_tests_rollup." + "commits, commit_parents, latest_checkout, hardware_status, " + "tree_listing, tree_tests_rollup." ) def handle(self, *args, command, **options): @@ -223,6 +226,8 @@ def snapshot(self, table, snapshot_filepath: Path): case None: self.snapshot_issues() self.snapshot_checkouts() + self.snapshot_commits() + self.snapshot_commit_parents() self.snapshot_builds() self.snapshot_tests() self.snapshot_incidents() @@ -234,6 +239,10 @@ def snapshot(self, table, snapshot_filepath: Path): self.snapshot_issues() case "checkouts": self.snapshot_checkouts() + case "commits": + self.snapshot_commits() + case "commit_parents": + self.snapshot_commit_parents() case "builds": self.snapshot_builds() case "tests": @@ -262,6 +271,8 @@ def snapshot(self, table, snapshot_filepath: Path): def restore(self, snapshot_filepath: Path): self.snapshot_archive = tarfile.open(snapshot_filepath, "r:*") try: + self.restore_commits() + self.restore_commit_parents() self.restore_checkouts() self.restore_builds() self.restore_issues() @@ -304,6 +315,14 @@ def get_related_data( return related_ids, related_condition + def sync_id_sequence(self, table: str) -> None: + with connections["default"].cursor() as cursor: + cursor.execute( + f"SELECT setval(pg_get_serial_sequence(%s, 'id')," + f" COALESCE((SELECT MAX(id) FROM {table}), 1))", + [table], + ) + # ISSUES ######################################## def select_issues_data(self) -> list[tuple]: query = f""" @@ -461,6 +480,113 @@ def restore_checkouts(self) -> None: self.insert_checkouts_data(records) self.stdout.write("Checkouts migration completed") + # COMMITS ######################################## + def select_commits_data(self) -> Generator[list[tuple], None, None]: + query = """ + SELECT id, git_commit_hash, author_name, author_email, author_date, + committer_name, committer_email, committer_date, subject, + message, fetched_from_url + FROM commits + ORDER BY id + """ + with connections["default"].cursor() as cursor: + cursor.execute(query) + while batch := cursor.fetchmany(SELECT_BATCH_SIZE): + yield batch + + def insert_commits_data(self, records: list[tuple]) -> int: + original_commits = [ + Commits( + id=record[0], + git_commit_hash=record[1], + author_name=record[2] or None, + author_email=record[3] or None, + author_date=parse_datetime(record[4]) if record[4] else None, + committer_name=record[5] or None, + committer_email=record[6] or None, + committer_date=parse_datetime(record[7]) if record[7] else None, + subject=record[8] or None, + message=record[9] or None, + fetched_from_url=record[10] or None, + ) + for record in records + ] + migrated_commits = Commits.objects.bulk_create( + original_commits, + ignore_conflicts=True, + batch_size=DEFAULT_BATCH_SIZE, + ) + self.sync_id_sequence("commits") + total_inserted = len(migrated_commits) + self.stdout.write(f"Processed {total_inserted} Commits records") + return total_inserted + + def snapshot_commits(self) -> None: + with SpooledTemporaryFile(mode="w+b", max_size=MAX_MEMORY_BUFFER_BYTES) as file: + self.stdout.write("\nMigrating Commits...") + for record_batch in self.select_commits_data(): + self.insert_records(file, "commits", record_batch) + self.add_file_to_snapshot(file, "commits") + self.stdout.write("Commits migration completed") + + def restore_commits(self) -> None: + with TextIOWrapper(self.snapshot_archive.extractfile("commits.csv")) as file: + self.stdout.write("\nMigrating Commits...") + reader = csv.reader(file) + while records := self.read_records(reader, max_rows=SELECT_BATCH_SIZE): + self.insert_commits_data(records) + self.stdout.write("Commits migration completed") + + # COMMIT PARENTS ######################################## + def select_commit_parents_data(self) -> Generator[list[tuple], None, None]: + query = """ + SELECT id, commit_id, parent_id, ord + FROM commit_parents + ORDER BY id + """ + with connections["default"].cursor() as cursor: + cursor.execute(query) + while batch := cursor.fetchmany(SELECT_BATCH_SIZE): + yield batch + + def insert_commit_parents_data(self, records: list[tuple]) -> int: + original_parents = [ + CommitParents( + id=record[0], + commit_id=record[1], + parent_id=record[2], + ord=record[3], + ) + for record in records + ] + migrated_parents = CommitParents.objects.bulk_create( + original_parents, + ignore_conflicts=True, + batch_size=DEFAULT_BATCH_SIZE, + ) + self.sync_id_sequence("commit_parents") + total_inserted = len(migrated_parents) + self.stdout.write(f"Processed {total_inserted} CommitParents records") + return total_inserted + + def snapshot_commit_parents(self) -> None: + with SpooledTemporaryFile(mode="w+b", max_size=MAX_MEMORY_BUFFER_BYTES) as file: + self.stdout.write("\nMigrating CommitParents...") + for record_batch in self.select_commit_parents_data(): + self.insert_records(file, "commit_parents", record_batch) + self.add_file_to_snapshot(file, "commit_parents") + self.stdout.write("CommitParents migration completed") + + def restore_commit_parents(self) -> None: + with TextIOWrapper( + self.snapshot_archive.extractfile("commit_parents.csv") + ) as file: + self.stdout.write("\nMigrating CommitParents...") + reader = csv.reader(file) + while records := self.read_records(reader, max_rows=SELECT_BATCH_SIZE): + self.insert_commit_parents_data(records) + self.stdout.write("CommitParents migration completed") + # BUILDS ######################################## def select_builds_data(self) -> list[tuple]: related_checkout_ids, related_condition = self.get_related_data( From 3b018c138c7de7024b10716c3f80710eb194eb59 Mon Sep 17 00:00:00 2001 From: Alan Peixinho Date: Wed, 16 Sep 2026 18:41:27 -0300 Subject: [PATCH 03/11] feat(backend): split commit identity and message from commits (#2089) Keep commits rows narrow: unique git name/email in commit_identity, subject plus UTF-8 body in commit_message as bytea. update_db dumps and restores the new tables and still accepts a flat legacy commits CSV. --- backend/docs/update_db command.md | 2 +- .../helpers/commit_message_compression.py | 13 ++ .../management/commands/update_db.py | 191 ++++++++++++++++-- .../0021_commits_and_commit_parents.py | 59 +++++- backend/kernelCI_app/models.py | 49 ++++- 5 files changed, 282 insertions(+), 32 deletions(-) create mode 100644 backend/kernelCI_app/helpers/commit_message_compression.py diff --git a/backend/docs/update_db command.md b/backend/docs/update_db command.md index 966b2038b..9f68f0744 100644 --- a/backend/docs/update_db command.md +++ b/backend/docs/update_db command.md @@ -14,7 +14,7 @@ The migration preserves foreign key constraints. For example, if a test A refere ### Optional Parameters - `--table`: Limit data copy to a specific table - - Valid options: `issues`, `checkouts`, `commits`, `commit_parents`, `builds`, `tests`, `incidents`, `latest_checkout`, `hardware_status`, `tree_listing`, `tree_tests_rollup` + - Valid options: `issues`, `checkouts`, `commits`, `commit_identity`, `commit_message`, `commit_parents`, `builds`, `tests`, `incidents`, `latest_checkout`, `hardware_status`, `tree_listing`, `tree_tests_rollup` - If not provided, data from all tables will be copied - `--related-data-only`: Limits the selected data to data where the foreign key constraint is not broken. - Default: False. diff --git a/backend/kernelCI_app/helpers/commit_message_compression.py b/backend/kernelCI_app/helpers/commit_message_compression.py new file mode 100644 index 000000000..93524fea4 --- /dev/null +++ b/backend/kernelCI_app/helpers/commit_message_compression.py @@ -0,0 +1,13 @@ +def message_to_bytea(plain: str | None) -> bytes | None: + if plain is None: + return None + return plain.encode("utf-8") + + +def bytea_to_message(data: bytes | memoryview | None) -> str | None: + if data is None: + return None + raw = bytes(data) + if not raw: + return None + return raw.decode("utf-8") diff --git a/backend/kernelCI_app/management/commands/update_db.py b/backend/kernelCI_app/management/commands/update_db.py index 3be7d2dfd..fb1863c76 100644 --- a/backend/kernelCI_app/management/commands/update_db.py +++ b/backend/kernelCI_app/management/commands/update_db.py @@ -14,10 +14,16 @@ from django.db import connections, models from django.utils.dateparse import parse_datetime +from kernelCI_app.helpers.commit_message_compression import ( + bytea_to_message, + message_to_bytea, +) from kernelCI_app.management.commands.helpers.intervals import parse_interval from kernelCI_app.models import ( Builds, Checkouts, + CommitIdentity, + CommitMessage, CommitParents, Commits, HardwareStatus, @@ -145,7 +151,8 @@ def _invalid_table_error(self, table: str) -> str: return ( f"Unknown table '{table}'.\n" "\tValid options are: issues, checkouts, builds, tests, incidents, " - "commits, commit_parents, latest_checkout, hardware_status, " + "commits, commit_identity, commit_message, commit_parents, " + "latest_checkout, hardware_status, " "tree_listing, tree_tests_rollup." ) @@ -226,7 +233,9 @@ def snapshot(self, table, snapshot_filepath: Path): case None: self.snapshot_issues() self.snapshot_checkouts() + self.snapshot_commit_identity() self.snapshot_commits() + self.snapshot_commit_messages() self.snapshot_commit_parents() self.snapshot_builds() self.snapshot_tests() @@ -241,6 +250,10 @@ def snapshot(self, table, snapshot_filepath: Path): self.snapshot_checkouts() case "commits": self.snapshot_commits() + case "commit_identity": + self.snapshot_commit_identity() + case "commit_message": + self.snapshot_commit_messages() case "commit_parents": self.snapshot_commit_parents() case "builds": @@ -271,7 +284,9 @@ def snapshot(self, table, snapshot_filepath: Path): def restore(self, snapshot_filepath: Path): self.snapshot_archive = tarfile.open(snapshot_filepath, "r:*") try: + self.restore_commit_identity() self.restore_commits() + self.restore_commit_messages() self.restore_commit_parents() self.restore_checkouts() self.restore_builds() @@ -480,12 +495,62 @@ def restore_checkouts(self) -> None: self.insert_checkouts_data(records) self.stdout.write("Checkouts migration completed") + # COMMIT IDENTITY ######################################## + def select_commit_identity_data(self) -> Generator[list[tuple], None, None]: + query = """ + SELECT id, email, name + FROM commit_identity + ORDER BY id + """ + with connections["default"].cursor() as cursor: + cursor.execute(query) + while batch := cursor.fetchmany(SELECT_BATCH_SIZE): + yield batch + + def insert_commit_identity_data(self, records: list[tuple]) -> int: + identities = [ + CommitIdentity( + id=record[0], + email=record[1] or "", + name=record[2] or "", + ) + for record in records + ] + migrated = CommitIdentity.objects.bulk_create( + identities, + ignore_conflicts=True, + batch_size=DEFAULT_BATCH_SIZE, + ) + self.sync_id_sequence("commit_identity") + total_inserted = len(migrated) + self.stdout.write(f"Processed {total_inserted} CommitIdentity records") + return total_inserted + + def snapshot_commit_identity(self) -> None: + with SpooledTemporaryFile(mode="w+b", max_size=MAX_MEMORY_BUFFER_BYTES) as file: + self.stdout.write("\nMigrating CommitIdentity...") + for record_batch in self.select_commit_identity_data(): + self.insert_records(file, "commit_identity", record_batch) + self.add_file_to_snapshot(file, "commit_identity") + self.stdout.write("CommitIdentity migration completed") + + def restore_commit_identity(self) -> None: + if "commit_identity.csv" not in self.snapshot_archive.getnames(): + return + with TextIOWrapper( + self.snapshot_archive.extractfile("commit_identity.csv") + ) as file: + self.stdout.write("\nMigrating CommitIdentity...") + reader = csv.reader(file) + while records := self.read_records(reader, max_rows=SELECT_BATCH_SIZE): + self.insert_commit_identity_data(records) + self.stdout.write("CommitIdentity migration completed") + # COMMITS ######################################## def select_commits_data(self) -> Generator[list[tuple], None, None]: query = """ - SELECT id, git_commit_hash, author_name, author_email, author_date, - committer_name, committer_email, committer_date, subject, - message, fetched_from_url + SELECT id, git_commit_hash, author_identity_id, author_date, + committer_identity_id, committer_date, fetched_from_url FROM commits ORDER BY id """ @@ -494,28 +559,66 @@ def select_commits_data(self) -> Generator[list[tuple], None, None]: while batch := cursor.fetchmany(SELECT_BATCH_SIZE): yield batch + def _identity_id(self, email: str | None, name: str | None) -> int: + identity, _created = CommitIdentity.objects.get_or_create( + email=email or "", + name=name or "", + ) + return identity.id + def insert_commits_data(self, records: list[tuple]) -> int: - original_commits = [ - Commits( - id=record[0], - git_commit_hash=record[1], - author_name=record[2] or None, - author_email=record[3] or None, - author_date=parse_datetime(record[4]) if record[4] else None, - committer_name=record[5] or None, - committer_email=record[6] or None, - committer_date=parse_datetime(record[7]) if record[7] else None, - subject=record[8] or None, - message=record[9] or None, - fetched_from_url=record[10] or None, + original_commits: list[Commits] = [] + flat_messages: list[tuple[int, str | None, str | None]] = [] + + for record in records: + if len(record) == 11: + commit_id = int(record[0]) + author_id = self._identity_id(record[3], record[2]) + committer_id = self._identity_id(record[6], record[5]) + original_commits.append( + Commits( + id=commit_id, + git_commit_hash=record[1], + author_identity_id=author_id, + author_date=parse_datetime(record[4]) if record[4] else None, + committer_identity_id=committer_id, + committer_date=parse_datetime(record[7]) if record[7] else None, + fetched_from_url=record[10] or None, + ) + ) + flat_messages.append((commit_id, record[8] or None, record[9] or None)) + continue + + original_commits.append( + Commits( + id=record[0], + git_commit_hash=record[1], + author_identity_id=record[2], + author_date=parse_datetime(record[3]) if record[3] else None, + committer_identity_id=record[4], + committer_date=parse_datetime(record[5]) if record[5] else None, + fetched_from_url=record[6] or None, + ) ) - for record in records - ] + migrated_commits = Commits.objects.bulk_create( original_commits, ignore_conflicts=True, batch_size=DEFAULT_BATCH_SIZE, ) + if flat_messages: + CommitMessage.objects.bulk_create( + [ + CommitMessage( + commit_id=commit_id, + subject=subject, + message=message_to_bytea(message), + ) + for commit_id, subject, message in flat_messages + ], + ignore_conflicts=True, + batch_size=DEFAULT_BATCH_SIZE, + ) self.sync_id_sequence("commits") total_inserted = len(migrated_commits) self.stdout.write(f"Processed {total_inserted} Commits records") @@ -537,6 +640,56 @@ def restore_commits(self) -> None: self.insert_commits_data(records) self.stdout.write("Commits migration completed") + # COMMIT MESSAGE ######################################## + def select_commit_messages_data(self) -> Generator[list[tuple], None, None]: + query = """ + SELECT commit_id, subject, message + FROM commit_message + ORDER BY commit_id + """ + with connections["default"].cursor() as cursor: + cursor.execute(query) + while batch := cursor.fetchmany(SELECT_BATCH_SIZE): + yield [(row[0], row[1], bytea_to_message(row[2])) for row in batch] + + def insert_commit_messages_data(self, records: list[tuple]) -> int: + messages = [ + CommitMessage( + commit_id=record[0], + subject=record[1] or None, + message=message_to_bytea(record[2]) if record[2] else None, + ) + for record in records + ] + migrated = CommitMessage.objects.bulk_create( + messages, + ignore_conflicts=True, + batch_size=DEFAULT_BATCH_SIZE, + ) + total_inserted = len(migrated) + self.stdout.write(f"Processed {total_inserted} CommitMessage records") + return total_inserted + + def snapshot_commit_messages(self) -> None: + with SpooledTemporaryFile(mode="w+b", max_size=MAX_MEMORY_BUFFER_BYTES) as file: + self.stdout.write("\nMigrating CommitMessage...") + for record_batch in self.select_commit_messages_data(): + self.insert_records(file, "commit_message", record_batch) + self.add_file_to_snapshot(file, "commit_message") + self.stdout.write("CommitMessage migration completed") + + def restore_commit_messages(self) -> None: + if "commit_message.csv" not in self.snapshot_archive.getnames(): + return + with TextIOWrapper( + self.snapshot_archive.extractfile("commit_message.csv") + ) as file: + self.stdout.write("\nMigrating CommitMessage...") + reader = csv.reader(file) + while records := self.read_records(reader, max_rows=SELECT_BATCH_SIZE): + self.insert_commit_messages_data(records) + self.stdout.write("CommitMessage migration completed") + # COMMIT PARENTS ######################################## def select_commit_parents_data(self) -> Generator[list[tuple], None, None]: query = """ diff --git a/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py b/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py index db1d7fb03..a6ef280ee 100644 --- a/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py +++ b/backend/kernelCI_app/migrations/0021_commits_and_commit_parents.py @@ -10,25 +10,72 @@ class Migration(migrations.Migration): ] operations = [ + migrations.CreateModel( + name="CommitIdentity", + fields=[ + ("id", models.AutoField(primary_key=True, serialize=False)), + ("email", models.TextField(blank=True, default="")), + ("name", models.TextField(blank=True, default="")), + ], + options={ + "db_table": "commit_identity", + "constraints": [ + models.UniqueConstraint( + fields=("email", "name"), + name="commit_identity_email_name", + ), + ], + }, + ), migrations.CreateModel( name="Commits", fields=[ ("id", models.AutoField(primary_key=True, serialize=False)), ("git_commit_hash", models.TextField(unique=True)), - ("author_name", models.TextField(blank=True, null=True)), - ("author_email", models.TextField(blank=True, null=True)), ("author_date", models.DateTimeField(blank=True, null=True)), - ("committer_name", models.TextField(blank=True, null=True)), - ("committer_email", models.TextField(blank=True, null=True)), ("committer_date", models.DateTimeField(blank=True, null=True)), - ("subject", models.TextField(blank=True, null=True)), - ("message", models.TextField(blank=True, null=True)), ("fetched_from_url", models.TextField(blank=True, null=True)), + ( + "author_identity", + models.ForeignKey( + on_delete=django.db.models.deletion.PROTECT, + related_name="authored_commits", + to="kernelCI_app.commitidentity", + ), + ), + ( + "committer_identity", + models.ForeignKey( + on_delete=django.db.models.deletion.PROTECT, + related_name="committed_commits", + to="kernelCI_app.commitidentity", + ), + ), ], options={ "db_table": "commits", }, ), + migrations.CreateModel( + name="CommitMessage", + fields=[ + ("subject", models.TextField(blank=True, null=True)), + ("message", models.BinaryField(blank=True, null=True)), + ( + "commit", + models.ForeignKey( + db_column="commit_id", + on_delete=django.db.models.deletion.CASCADE, + primary_key=True, + related_name="message_row", + to="kernelCI_app.commits", + ), + ), + ], + options={ + "db_table": "commit_message", + }, + ), migrations.CreateModel( name="CommitParents", fields=[ diff --git a/backend/kernelCI_app/models.py b/backend/kernelCI_app/models.py index 7d48e6372..8379fdc5a 100644 --- a/backend/kernelCI_app/models.py +++ b/backend/kernelCI_app/models.py @@ -109,17 +109,39 @@ class Meta: ] +class CommitIdentity(models.Model): + id = models.AutoField(primary_key=True) + email = models.TextField(default="", blank=True) + name = models.TextField(default="", blank=True) + + class Meta: + db_table = "commit_identity" + constraints = [ + models.UniqueConstraint( + fields=["email", "name"], + name="commit_identity_email_name", + ), + ] + + def __str__(self) -> str: + return f"{self.name} <{self.email}>" + + class Commits(models.Model): id = models.AutoField(primary_key=True) git_commit_hash = models.TextField(unique=True) - author_name = models.TextField(blank=True, null=True) - author_email = models.TextField(blank=True, null=True) + author_identity = models.ForeignKey( + CommitIdentity, + on_delete=models.PROTECT, + related_name="authored_commits", + ) author_date = models.DateTimeField(blank=True, null=True) - committer_name = models.TextField(blank=True, null=True) - committer_email = models.TextField(blank=True, null=True) + committer_identity = models.ForeignKey( + CommitIdentity, + on_delete=models.PROTECT, + related_name="committed_commits", + ) committer_date = models.DateTimeField(blank=True, null=True) - subject = models.TextField(blank=True, null=True) - message = models.TextField(blank=True, null=True) fetched_from_url = models.TextField(blank=True, null=True) class Meta: @@ -129,6 +151,21 @@ def __str__(self) -> str: return self.git_commit_hash +class CommitMessage(models.Model): + commit = models.ForeignKey( + Commits, + on_delete=models.CASCADE, + primary_key=True, + related_name="message_row", + db_column="commit_id", + ) + subject = models.TextField(blank=True, null=True) + message = models.BinaryField(blank=True, null=True) + + class Meta: + db_table = "commit_message" + + class CommitParents(models.Model): id = models.AutoField(primary_key=True) commit = models.ForeignKey( From 8f8b96d3af3ced994969b76d5e671cf427a74e50 Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 11 Sep 2026 15:41:03 -0300 Subject: [PATCH 04/11] feat(backend): parse and fetch single-commit metadata (#2090) Add a SHA-only helper for commit author/parents so later sync can reuse parsing without duplicating git format handling. Signed-off-by: Felipe Bergamin --- .env.backend.example | 2 + .env.example | 2 + backend/Dockerfile | 1 + backend/kernelCI/settings.py | 4 + backend/kernelCI_app/helpers/gitCommit.py | 338 ++++++++++++++++++ .../helpers/fixtures/git_commit_data.py | 20 ++ .../tests/unitTests/helpers/gitCommit_test.py | 248 +++++++++++++ 7 files changed, 615 insertions(+) create mode 100644 backend/kernelCI_app/helpers/gitCommit.py create mode 100644 backend/kernelCI_app/tests/unitTests/helpers/fixtures/git_commit_data.py create mode 100644 backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py diff --git a/.env.backend.example b/.env.backend.example index bd61973aa..e8338d89f 100644 --- a/.env.backend.example +++ b/.env.backend.example @@ -25,6 +25,8 @@ PROMETHEUS_MULTIPROC_DIR=/tmp/metrics INGESTER_METRICS_PORT=8002 BACKEND_VOLUME_DIR=/volume_data +# Ephemeral git clones for SHA-only commit fetch (#2090). Prefer tmpfs. +GIT_SCRATCH_DIR=/dev/shm/kernelci-git-scratch ## Variables used for the notifications command. Check docs/notifications.md # EMAIL_HOST_USER="youruser@host" # (optional) diff --git a/.env.example b/.env.example index ccdf18e3a..82d66bb94 100644 --- a/.env.example +++ b/.env.example @@ -106,6 +106,8 @@ HEALTHCHECK_ID_NOTIFICATIONS_SUMMARY_MAESTRO= # Backend Volume # ----------------------------------------------------------------------------- BACKEND_VOLUME_DIR=/volume_data +# Ephemeral git clones for SHA-only commit fetch (#2090). Prefer tmpfs. +GIT_SCRATCH_DIR=/dev/shm/kernelci-git-scratch # ----------------------------------------------------------------------------- # Ingester (only needed with --profile=with_commands) diff --git a/backend/Dockerfile b/backend/Dockerfile index c7474f5ff..d6e454f10 100644 --- a/backend/Dockerfile +++ b/backend/Dockerfile @@ -19,6 +19,7 @@ RUN apk update \ libpq-dev \ postgresql \ curl \ + git \ && python3 -m venv $POETRY_HOME \ && $POETRY_HOME/bin/pip install poetry~=2.4.0 \ && ln -s $POETRY_HOME/bin/poetry /bin/poetry diff --git a/backend/kernelCI/settings.py b/backend/kernelCI/settings.py index d02d7ff76..7e69ef2f8 100644 --- a/backend/kernelCI/settings.py +++ b/backend/kernelCI/settings.py @@ -269,6 +269,10 @@ def get_json_env_var(name, default): # https://docs.djangoproject.com/en/5.0/ref/settings/#databases BACKEND_VOLUME_DIR = os.environ.get("BACKEND_VOLUME_DIR", "/volume_data") +# Throwaway git dirs for one-shot SHA fetches (#2090). Prefer tmpfs (e.g. /dev/shm). +GIT_SCRATCH_DIR = os.environ.get( + "GIT_SCRATCH_DIR", "/dev/shm/kernelci-git-scratch" +) DATABASE_ROUTERS = ["kernelCI_app.routers.databaseRouter.DatabaseRouter"] diff --git a/backend/kernelCI_app/helpers/gitCommit.py b/backend/kernelCI_app/helpers/gitCommit.py new file mode 100644 index 000000000..c9fb4e783 --- /dev/null +++ b/backend/kernelCI_app/helpers/gitCommit.py @@ -0,0 +1,338 @@ +"""Parse git commit metadata. Optional one-shot SHA fetch; no DB writes. + +Used as a fallback for hashes the mirrored-tree sync job (#2109) cannot +cover. Parse against an existing repo path so that job can reuse this +instead of duplicating commit-format parsing. +""" + +from __future__ import annotations + +import os +import re +import shutil +import subprocess +import tempfile +from dataclasses import dataclass +from datetime import datetime, timedelta, timezone +from pathlib import Path +from urllib.parse import urlparse + +from django.conf import settings + +from kernelCI_app.models import Checkouts + +FETCH_TIMEOUT_SECONDS = 60 +# One commit object is tiny; a full kernel history pack is hundreds of MB. +MAX_EPHEMERAL_PACK_BYTES = 2 * 1024 * 1024 +_IDENT_RE = re.compile(r"^([^<]*?) <([^>]*)> (\d+) ([+-]\d{4})$") +_GIT_ENV = { + "GIT_TERMINAL_PROMPT": "0", + "GCM_INTERACTIVE": "never", +} + + +class CommitMetadataError(Exception): + """Typed failure from parse or one-shot fetch. No metadata returned.""" + + +class InvalidGitUrlError(CommitMetadataError): + pass + + +class MissingGitUrlError(CommitMetadataError): + pass + + +class FetchFailedError(CommitMetadataError): + pass + + +class OversizedPackError(CommitMetadataError): + pass + + +class UnexpectedFetchObjectsError(CommitMetadataError): + pass + + +class CommitParseError(CommitMetadataError): + pass + + +@dataclass(frozen=True) +class CommitMetadata: + git_commit_hash: str + author_name: str | None + author_email: str | None + author_date: datetime | None + committer_name: str | None + committer_email: str | None + committer_date: datetime | None + subject: str | None + message: str | None + parent_hashes: tuple[str, ...] + + +def sanitize_git_url(git_url: str | None) -> str | None: + """Treeproof-style cleanup. Returns None for malformed URLs; never raises.""" + if not isinstance(git_url, str) or not git_url.strip(): + return None + + raw = git_url.strip().rstrip("/") + if "://" not in raw: + return None + + parsed = urlparse(raw) + if not parsed.scheme: + return None + if parsed.scheme != "file" and not parsed.netloc: + return None + if not [segment for segment in parsed.path.split("/") if segment]: + return None + return raw + + +def resolve_checkout_git_url(git_commit_hash: str) -> str | None: + """Pick a fetch URL from checkouts for this hash. Prefer maestro / git.kernel.org.""" + rows = ( + Checkouts.objects.filter( + git_commit_hash=git_commit_hash, + git_repository_url__isnull=False, + ) + .values_list("origin", "git_repository_url") + .distinct() + ) + + best_url: str | None = None + best_rank: tuple[int, str] | None = None + for origin, url in rows: + cleaned = sanitize_git_url(url) + if cleaned is None: + continue + rank = _url_preference(origin or "", cleaned) + if best_rank is None or rank < best_rank: + best_rank = rank + best_url = cleaned + return best_url + + +def parse_commit_object(*, raw: str, git_commit_hash: str) -> CommitMetadata: + """Parse a raw commit object (`git cat-file -p`). No git, no fetch.""" + if not git_commit_hash: + raise CommitParseError("missing git_commit_hash") + + headers, separator, message = raw.partition("\n\n") + if not separator: + raise CommitParseError("commit object has no header/message separator") + + parent_hashes: list[str] = [] + author = (None, None, None) + committer = (None, None, None) + for line in _header_lines(headers): + if line.startswith("parent "): + parent_hashes.append(line.removeprefix("parent ").strip()) + elif line.startswith("author "): + author = _parse_ident(line.removeprefix("author ")) + elif line.startswith("committer "): + committer = _parse_ident(line.removeprefix("committer ")) + + subject = message.split("\n", 1)[0] if message else None + if subject == "": + subject = None + + return CommitMetadata( + git_commit_hash=git_commit_hash, + author_name=author[0], + author_email=author[1], + author_date=author[2], + committer_name=committer[0], + committer_email=committer[1], + committer_date=committer[2], + subject=subject, + message=message if message else None, + parent_hashes=tuple(parent_hashes), + ) + + +def parse_commit(*, repo_path: str, git_commit_hash: str) -> CommitMetadata: + """Read one commit from an existing repo. Does not fetch.""" + try: + full_hash = ( + _git( + Path(repo_path), + "rev-parse", + "--verify", + f"{git_commit_hash}^{{commit}}", + ) + .decode() + .strip() + ) + raw = _git(Path(repo_path), "cat-file", "-p", full_hash).decode( + "utf-8", errors="replace" + ) + except FetchFailedError as exc: + raise CommitParseError(str(exc)) from exc + return parse_commit_object(raw=raw, git_commit_hash=full_hash) + + +def fetch_commit_metadata( + git_commit_hash: str, + url: str | None = None, +) -> CommitMetadata: + """Fetch a single SHA into a throwaway bare repo, parse it, wipe the repo. + + Not the fill path for `commits`: no ancestry, no parent-object fetch. + """ + remote_url = _resolve_fetch_url(git_commit_hash, url) + scratch = Path(settings.GIT_SCRATCH_DIR) + scratch.mkdir(parents=True, exist_ok=True) + repo_dir = Path(tempfile.mkdtemp(prefix="commit-fetch-", dir=scratch)) + try: + _git(repo_dir, "init", "--bare") + _git(repo_dir, "remote", "add", "origin", remote_url) + _git( + repo_dir, + "fetch", + "--no-tags", + "--depth=1", + "--filter=tree:0", + "origin", + git_commit_hash, + timeout=FETCH_TIMEOUT_SECONDS, + ) + assert_single_commit_fetch(repo_dir) + return parse_commit(repo_path=str(repo_dir), git_commit_hash=git_commit_hash) + except CommitMetadataError: + raise + except Exception as exc: + raise FetchFailedError( + f"fetch {git_commit_hash} from {remote_url} failed: {exc}" + ) from exc + finally: + shutil.rmtree(repo_dir, ignore_errors=True) + + +def assert_single_commit_fetch(repo_dir: Path) -> None: + """Fail if the remote ignored shallow/filter and sent a full or treeful pack.""" + pack_bytes = _pack_bytes(repo_dir) + if pack_bytes > MAX_EPHEMERAL_PACK_BYTES: + raise OversizedPackError( + f"ephemeral fetch pack is {pack_bytes} bytes (max {MAX_EPHEMERAL_PACK_BYTES})" + ) + + counts = _object_type_counts(repo_dir) + if counts.get("tree", 0) or counts.get("blob", 0): + raise UnexpectedFetchObjectsError( + f"ephemeral fetch included trees/blobs: {counts}" + ) + if counts.get("commit", 0) != 1: + raise UnexpectedFetchObjectsError( + f"ephemeral fetch must contain exactly one commit: {counts}" + ) + + +def _resolve_fetch_url(git_commit_hash: str, url: str | None) -> str: + if url is not None: + cleaned = sanitize_git_url(url) + if cleaned is None: + raise InvalidGitUrlError(f"malformed git url: {url!r}") + return cleaned + + resolved = resolve_checkout_git_url(git_commit_hash) + if resolved is None: + raise MissingGitUrlError(f"no usable git url for {git_commit_hash}") + return resolved + + +def _url_preference(origin: str, url: str) -> tuple[int, str]: + host = urlparse(url).netloc.lower() + kernel_org = "git.kernel.org" in host + maestro = origin == "maestro" + if maestro and kernel_org: + tier = 0 + elif kernel_org: + tier = 1 + elif maestro: + tier = 2 + else: + tier = 3 + return (tier, url) + + +def _header_lines(headers: str): + for line in headers.split("\n"): + if line.startswith(" "): + continue + yield line + + +def _parse_ident( + ident: str, +) -> tuple[str | None, str | None, datetime | None]: + match = _IDENT_RE.match(ident.strip()) + if match is None: + return (None, None, None) + name, email, unix, tz = match.groups() + name = name.strip() or None + email = email.strip() or None + return (name, email, _parse_git_date(int(unix), tz)) + + +def _parse_git_date(unix: int, tz: str) -> datetime: + sign = 1 if tz[0] == "+" else -1 + hours = int(tz[1:3]) + minutes = int(tz[3:5]) + offset = timedelta(hours=hours, minutes=minutes) * sign + return datetime.fromtimestamp(unix, tz=timezone(offset)) + + +def _pack_bytes(repo_dir: Path) -> int: + pack_dir = repo_dir / "objects" / "pack" + if not pack_dir.is_dir(): + return 0 + return sum( + path.stat().st_size for path in pack_dir.glob("*.pack") if path.is_file() + ) + + +def _object_type_counts(repo_dir: Path) -> dict[str, int]: + output = _git( + repo_dir, + "cat-file", + "--batch-check=%(objecttype)", + "--batch-all-objects", + ).decode() + counts: dict[str, int] = {} + for line in output.splitlines(): + object_type = line.strip() + if object_type: + counts[object_type] = counts.get(object_type, 0) + 1 + return counts + + +def _git_executable() -> str: + git = shutil.which("git") + if git is None: + raise FetchFailedError("git executable not found") + return git + + +def _git(repo_dir: Path, *args: str, timeout: int = 30) -> bytes: + env = os.environ.copy() + env.update(_GIT_ENV) + command = [_git_executable(), "-C", str(repo_dir), *args] + try: + result = subprocess.run( # noqa: S603 + command, + check=False, + capture_output=True, + timeout=timeout, + env=env, + ) + except subprocess.TimeoutExpired as exc: + raise FetchFailedError(f"git {' '.join(args)} timed out") from exc + + if result.returncode != 0: + stderr = result.stderr.decode("utf-8", errors="replace").strip() + raise FetchFailedError(f"git {' '.join(args)} failed: {stderr}") + return result.stdout diff --git a/backend/kernelCI_app/tests/unitTests/helpers/fixtures/git_commit_data.py b/backend/kernelCI_app/tests/unitTests/helpers/fixtures/git_commit_data.py new file mode 100644 index 000000000..1822d7598 --- /dev/null +++ b/backend/kernelCI_app/tests/unitTests/helpers/fixtures/git_commit_data.py @@ -0,0 +1,20 @@ +MERGE_COMMIT_HASH = "0123456789abcdef0123456789abcdef01234567" +FIRST_PARENT = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" +SECOND_PARENT = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + +# gpgsig uses space-prefixed continuation lines, including a "blank" signature line. +MERGE_COMMIT_OBJECT = ( + "tree 4b825dc642cb6eb9a060e54bf8d69288fbee4904\n" + "parent aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa\n" + "parent bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb\n" + "author Alice Author 1000000000 +0000\n" + "committer Bob Committer 1000000060 -0500\n" + "gpgsig -----BEGIN PGP SIGNATURE-----\n" + " \n" + " iQIzBAABCAAdFiEE\n" + " -----END PGP SIGNATURE-----\n" + "\n" + "Add feature foo\n" + "\n" + "Longer body that is not the subject.\n" +) diff --git a/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py b/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py new file mode 100644 index 000000000..36c2382e8 --- /dev/null +++ b/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py @@ -0,0 +1,248 @@ +import os +import shutil +import subprocess +from datetime import datetime, timedelta, timezone +from pathlib import Path +from unittest.mock import patch + +import pytest +from django.test import override_settings + +from kernelCI_app.helpers.gitCommit import ( + MAX_EPHEMERAL_PACK_BYTES, + CommitParseError, + FetchFailedError, + InvalidGitUrlError, + MissingGitUrlError, + OversizedPackError, + UnexpectedFetchObjectsError, + assert_single_commit_fetch, + fetch_commit_metadata, + parse_commit, + parse_commit_object, + resolve_checkout_git_url, + sanitize_git_url, +) +from kernelCI_app.tests.unitTests.helpers.fixtures.git_commit_data import ( + FIRST_PARENT, + MERGE_COMMIT_HASH, + MERGE_COMMIT_OBJECT, + SECOND_PARENT, +) + + +def _run_git(repo: Path, *args: str) -> str: + env = { + **os.environ, + "GIT_AUTHOR_NAME": "Alice Author", + "GIT_AUTHOR_EMAIL": "alice@example.com", + "GIT_COMMITTER_NAME": "Bob Committer", + "GIT_COMMITTER_EMAIL": "bob@example.com", + "GIT_AUTHOR_DATE": "2001-09-09T01:46:40+0000", + "GIT_COMMITTER_DATE": "2001-09-09T01:47:40-0500", + } + git = shutil.which("git") + assert git is not None + result = subprocess.run( # noqa: S603 + [git, "-C", str(repo), *args], + check=True, + capture_output=True, + text=True, + env=env, + ) + return result.stdout.strip() + + +def _build_repo_with_merge(tmp_path: Path) -> tuple[Path, str, str, str]: + repo = tmp_path / "src" + repo.mkdir() + _run_git(repo, "init", "-b", "main") + _run_git(repo, "config", "user.name", "Alice Author") + _run_git(repo, "config", "user.email", "alice@example.com") + _run_git(repo, "config", "uploadpack.allowFilter", "true") + _run_git(repo, "config", "uploadpack.allowAnySHA1InWant", "true") + (repo / "a.txt").write_text("a\n") + _run_git(repo, "add", "a.txt") + _run_git(repo, "commit", "-m", "root commit") + root = _run_git(repo, "rev-parse", "HEAD") + + _run_git(repo, "checkout", "-b", "other") + (repo / "b.txt").write_text("b\n") + _run_git(repo, "add", "b.txt") + _run_git(repo, "commit", "-m", "side commit") + side = _run_git(repo, "rev-parse", "HEAD") + + _run_git(repo, "checkout", "main") + _run_git(repo, "merge", "--no-ff", "-m", "merge side\n\nmerge body\n", "other") + merge = _run_git(repo, "rev-parse", "HEAD") + return repo, root, side, merge + + +class TestParseCommitObject: + def test_author_committer_subject_ordered_parents(self): + metadata = parse_commit_object( + raw=MERGE_COMMIT_OBJECT, git_commit_hash=MERGE_COMMIT_HASH + ) + + assert metadata.git_commit_hash == MERGE_COMMIT_HASH + assert metadata.author_name == "Alice Author" + assert metadata.author_email == "alice@example.com" + assert metadata.author_date == datetime( + 2001, 9, 9, 1, 46, 40, tzinfo=timezone.utc + ) + assert metadata.committer_name == "Bob Committer" + assert metadata.committer_email == "bob@example.com" + assert metadata.committer_date == datetime( + 2001, 9, 8, 20, 47, 40, tzinfo=timezone(timedelta(hours=-5)) + ) + assert metadata.subject == "Add feature foo" + assert metadata.message.startswith("Add feature foo\n") + assert "Longer body" in metadata.message + assert metadata.parent_hashes == (FIRST_PARENT, SECOND_PARENT) + + def test_missing_separator_raises(self): + with pytest.raises(CommitParseError): + parse_commit_object(raw="tree abc\n", git_commit_hash=MERGE_COMMIT_HASH) + + +class TestParseCommitFromRepo: + def test_parse_existing_repo_does_not_fetch(self, tmp_path, monkeypatch): + repo, root, side, merge = _build_repo_with_merge(tmp_path) + calls: list[list[str]] = [] + real_run = subprocess.run + + def wrapped(*args, **kwargs): + command = args[0] if args else kwargs.get("args") + if isinstance(command, list): + calls.append(command) + return real_run(*args, **kwargs) + + monkeypatch.setattr("kernelCI_app.helpers.gitCommit.subprocess.run", wrapped) + + metadata = parse_commit(repo_path=str(repo), git_commit_hash=merge) + + assert metadata.git_commit_hash == merge + assert metadata.parent_hashes == (root, side) + assert metadata.subject == "merge side" + assert metadata.message.startswith("merge side\n") + assert all("fetch" not in command for command in calls) + + +class TestSanitizeGitUrl: + def test_accepts_https_and_file(self): + https = "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git/" + assert ( + sanitize_git_url(https) + == "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git" + ) + assert sanitize_git_url("file:///tmp/linux.git") == "file:///tmp/linux.git" + + def test_malformed_does_not_raise(self): + for url in ( + "", + " ", + None, + "https://example.com/", + "https://example.com", + "linux.git", + "git@github.com:org/repo.git", + "not a url at all!!!", + ): + assert sanitize_git_url(url) is None + + +class TestResolveCheckoutGitUrl: + @patch("kernelCI_app.helpers.gitCommit.Checkouts.objects") + def test_prefers_maestro_kernel_org(self, mock_objects): + rows = mock_objects.filter.return_value.values_list.return_value + rows.distinct.return_value = [ + ("redhat", "https://github.com/foo/linux.git"), + ( + "maestro", + "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git", + ), + ("maestro", "https://github.com/torvalds/linux.git"), + ] + assert resolve_checkout_git_url("abc") == ( + "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git" + ) + + @patch("kernelCI_app.helpers.gitCommit.Checkouts.objects") + def test_skips_malformed_urls(self, mock_objects): + rows = mock_objects.filter.return_value.values_list.return_value + rows.distinct.return_value = [ + ("maestro", "git@github.com:org/repo.git"), + ("maestro", "https://example.com/"), + ( + "redhat", + "https://git.kernel.org/pub/scm/linux/kernel/git/redhat/linux.git", + ), + ] + assert resolve_checkout_git_url("abc") == ( + "https://git.kernel.org/pub/scm/linux/kernel/git/redhat/linux.git" + ) + + +class TestFetchCommitMetadata: + def test_fetch_from_local_repo_then_wipes(self, tmp_path): + repo, _root, _side, merge = _build_repo_with_merge(tmp_path) + scratch = tmp_path / "scratch" + with override_settings(GIT_SCRATCH_DIR=str(scratch)): + metadata = fetch_commit_metadata(merge, url=f"file://{repo}") + + assert metadata.git_commit_hash == merge + assert metadata.subject == "merge side" + assert metadata.parent_hashes[0] == _root + leftover = list(scratch.glob("commit-fetch-*")) + assert leftover == [] + + def test_fetch_failure(self, tmp_path): + scratch = tmp_path / "scratch" + with override_settings(GIT_SCRATCH_DIR=str(scratch)): + with pytest.raises(FetchFailedError): + fetch_commit_metadata("a" * 40, url="file:///no/such/repo.git") + leftover = list(scratch.glob("commit-fetch-*")) + assert leftover == [] + + def test_bad_url(self): + with pytest.raises(InvalidGitUrlError): + fetch_commit_metadata("a" * 40, url="git@github.com:org/repo.git") + + @patch("kernelCI_app.helpers.gitCommit.resolve_checkout_git_url", return_value=None) + def test_missing_url(self, _mock_resolve): + with pytest.raises(MissingGitUrlError): + fetch_commit_metadata("a" * 40) + + def test_oversized_pack(self, tmp_path): + repo = tmp_path / "bare.git" + (repo / "objects" / "pack").mkdir(parents=True) + pack = repo / "objects" / "pack" / "pack-deadbeef.pack" + pack.write_bytes(b"\0" * (MAX_EPHEMERAL_PACK_BYTES + 1)) + with pytest.raises(OversizedPackError): + assert_single_commit_fetch(repo) + + def test_full_pack_extra_commits(self, tmp_path, monkeypatch): + repo = tmp_path / "bare.git" + repo.mkdir() + monkeypatch.setattr( + "kernelCI_app.helpers.gitCommit._object_type_counts", + lambda _repo: {"commit": 12}, + ) + monkeypatch.setattr( + "kernelCI_app.helpers.gitCommit._pack_bytes", lambda _repo: 10 + ) + with pytest.raises(UnexpectedFetchObjectsError): + assert_single_commit_fetch(repo) + + def test_full_pack_includes_trees(self, tmp_path, monkeypatch): + repo = tmp_path / "bare.git" + repo.mkdir() + monkeypatch.setattr( + "kernelCI_app.helpers.gitCommit._object_type_counts", + lambda _repo: {"commit": 1, "tree": 4, "blob": 20}, + ) + monkeypatch.setattr( + "kernelCI_app.helpers.gitCommit._pack_bytes", lambda _repo: 10 + ) + with pytest.raises(UnexpectedFetchObjectsError): + assert_single_commit_fetch(repo) From a4fb0ae26b45d9e3cdc9d760f1cf733850a3bc5d Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 25 Sep 2026 16:38:04 -0300 Subject: [PATCH 05/11] fix(backend): address review on commit metadata fetch - Move fetch timeout and max pack size to Django settings (GIT_FETCH_TIMEOUT_SECONDS, GIT_FETCH_MAX_PACK_BYTES) - Document URL preference tiers and why the git CLI is used - Reuse author/committer constants in tests --- .env.backend.example | 2 ++ .env.example | 2 ++ backend/kernelCI/settings.py | 7 ++-- backend/kernelCI_app/helpers/gitCommit.py | 31 +++++++++++----- .../tests/unitTests/helpers/gitCommit_test.py | 35 +++++++++++-------- 5 files changed, 53 insertions(+), 24 deletions(-) diff --git a/.env.backend.example b/.env.backend.example index e8338d89f..7ff045c0a 100644 --- a/.env.backend.example +++ b/.env.backend.example @@ -27,6 +27,8 @@ INGESTER_METRICS_PORT=8002 BACKEND_VOLUME_DIR=/volume_data # Ephemeral git clones for SHA-only commit fetch (#2090). Prefer tmpfs. GIT_SCRATCH_DIR=/dev/shm/kernelci-git-scratch +# GIT_FETCH_TIMEOUT_SECONDS=60 +# GIT_FETCH_MAX_PACK_BYTES=2097152 ## Variables used for the notifications command. Check docs/notifications.md # EMAIL_HOST_USER="youruser@host" # (optional) diff --git a/.env.example b/.env.example index 82d66bb94..0f84f1cc3 100644 --- a/.env.example +++ b/.env.example @@ -108,6 +108,8 @@ HEALTHCHECK_ID_NOTIFICATIONS_SUMMARY_MAESTRO= BACKEND_VOLUME_DIR=/volume_data # Ephemeral git clones for SHA-only commit fetch (#2090). Prefer tmpfs. GIT_SCRATCH_DIR=/dev/shm/kernelci-git-scratch +# GIT_FETCH_TIMEOUT_SECONDS=60 +# GIT_FETCH_MAX_PACK_BYTES=2097152 # ----------------------------------------------------------------------------- # Ingester (only needed with --profile=with_commands) diff --git a/backend/kernelCI/settings.py b/backend/kernelCI/settings.py index 7e69ef2f8..f14af4eb9 100644 --- a/backend/kernelCI/settings.py +++ b/backend/kernelCI/settings.py @@ -270,8 +270,11 @@ def get_json_env_var(name, default): BACKEND_VOLUME_DIR = os.environ.get("BACKEND_VOLUME_DIR", "/volume_data") # Throwaway git dirs for one-shot SHA fetches (#2090). Prefer tmpfs (e.g. /dev/shm). -GIT_SCRATCH_DIR = os.environ.get( - "GIT_SCRATCH_DIR", "/dev/shm/kernelci-git-scratch" +GIT_SCRATCH_DIR = os.environ.get("GIT_SCRATCH_DIR", "/dev/shm/kernelci-git-scratch") +GIT_FETCH_TIMEOUT_SECONDS = int(os.environ.get("GIT_FETCH_TIMEOUT_SECONDS", "60")) +# One commit object is tiny; a full kernel history pack is hundreds of MB. +GIT_FETCH_MAX_PACK_BYTES = int( + os.environ.get("GIT_FETCH_MAX_PACK_BYTES", str(2 * 1024 * 1024)) ) DATABASE_ROUTERS = ["kernelCI_app.routers.databaseRouter.DatabaseRouter"] diff --git a/backend/kernelCI_app/helpers/gitCommit.py b/backend/kernelCI_app/helpers/gitCommit.py index c9fb4e783..83334ce9d 100644 --- a/backend/kernelCI_app/helpers/gitCommit.py +++ b/backend/kernelCI_app/helpers/gitCommit.py @@ -21,9 +21,6 @@ from kernelCI_app.models import Checkouts -FETCH_TIMEOUT_SECONDS = 60 -# One commit object is tiny; a full kernel history pack is hundreds of MB. -MAX_EPHEMERAL_PACK_BYTES = 2 * 1024 * 1024 _IDENT_RE = re.compile(r"^([^<]*?) <([^>]*)> (\d+) ([+-]\d{4})$") _GIT_ENV = { "GIT_TERMINAL_PROMPT": "0", @@ -93,7 +90,11 @@ def sanitize_git_url(git_url: str | None) -> str | None: def resolve_checkout_git_url(git_commit_hash: str) -> str | None: - """Pick a fetch URL from checkouts for this hash. Prefer maestro / git.kernel.org.""" + """Pick a fetch URL from checkouts for this hash. + + A SHA is not unique in checkouts: the same commit is often submitted from + several clones. Ranking is in `_url_preference`. + """ rows = ( Checkouts.objects.filter( git_commit_hash=git_commit_hash, @@ -155,7 +156,11 @@ def parse_commit_object(*, raw: str, git_commit_hash: str) -> CommitMetadata: def parse_commit(*, repo_path: str, git_commit_hash: str) -> CommitMetadata: - """Read one commit from an existing repo. Does not fetch.""" + """Read one commit from an existing repo. Does not fetch. + + Uses the git CLI (not pygit2/GitPython): fetch needs --filter=tree:0 and + --depth=1, git is already on the image, and GitPython still shells out. + """ try: full_hash = ( _git( @@ -198,7 +203,7 @@ def fetch_commit_metadata( "--filter=tree:0", "origin", git_commit_hash, - timeout=FETCH_TIMEOUT_SECONDS, + timeout=settings.GIT_FETCH_TIMEOUT_SECONDS, ) assert_single_commit_fetch(repo_dir) return parse_commit(repo_path=str(repo_dir), git_commit_hash=git_commit_hash) @@ -215,9 +220,10 @@ def fetch_commit_metadata( def assert_single_commit_fetch(repo_dir: Path) -> None: """Fail if the remote ignored shallow/filter and sent a full or treeful pack.""" pack_bytes = _pack_bytes(repo_dir) - if pack_bytes > MAX_EPHEMERAL_PACK_BYTES: + max_pack_bytes = settings.GIT_FETCH_MAX_PACK_BYTES + if pack_bytes > max_pack_bytes: raise OversizedPackError( - f"ephemeral fetch pack is {pack_bytes} bytes (max {MAX_EPHEMERAL_PACK_BYTES})" + f"ephemeral fetch pack is {pack_bytes} bytes (max {max_pack_bytes})" ) counts = _object_type_counts(repo_dir) @@ -245,6 +251,15 @@ def _resolve_fetch_url(git_commit_hash: str, url: str | None) -> str: def _url_preference(origin: str, url: str) -> tuple[int, str]: + """Rank a checkout row when one SHA has several git_repository_url values. + + Lower rank is chosen. The URL string breaks ties inside a tier. + + 0 origin is maestro and the host is git.kernel.org + 1 host is git.kernel.org, any origin + 2 origin is maestro, any other host + 3 everything else + """ host = urlparse(url).netloc.lower() kernel_org = "git.kernel.org" in host maestro = origin == "maestro" diff --git a/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py b/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py index 36c2382e8..d3ec67c9d 100644 --- a/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py +++ b/backend/kernelCI_app/tests/unitTests/helpers/gitCommit_test.py @@ -6,10 +6,10 @@ from unittest.mock import patch import pytest +from django.conf import settings from django.test import override_settings from kernelCI_app.helpers.gitCommit import ( - MAX_EPHEMERAL_PACK_BYTES, CommitParseError, FetchFailedError, InvalidGitUrlError, @@ -30,16 +30,23 @@ SECOND_PARENT, ) +AUTHOR_NAME = "Alice Author" +AUTHOR_EMAIL = "alice@example.com" +COMMITTER_NAME = "Bob Committer" +COMMITTER_EMAIL = "bob@example.com" +AUTHOR_DATE = "2001-09-09T01:46:40+0000" +COMMITTER_DATE = "2001-09-09T01:47:40-0500" + def _run_git(repo: Path, *args: str) -> str: env = { **os.environ, - "GIT_AUTHOR_NAME": "Alice Author", - "GIT_AUTHOR_EMAIL": "alice@example.com", - "GIT_COMMITTER_NAME": "Bob Committer", - "GIT_COMMITTER_EMAIL": "bob@example.com", - "GIT_AUTHOR_DATE": "2001-09-09T01:46:40+0000", - "GIT_COMMITTER_DATE": "2001-09-09T01:47:40-0500", + "GIT_AUTHOR_NAME": AUTHOR_NAME, + "GIT_AUTHOR_EMAIL": AUTHOR_EMAIL, + "GIT_COMMITTER_NAME": COMMITTER_NAME, + "GIT_COMMITTER_EMAIL": COMMITTER_EMAIL, + "GIT_AUTHOR_DATE": AUTHOR_DATE, + "GIT_COMMITTER_DATE": COMMITTER_DATE, } git = shutil.which("git") assert git is not None @@ -57,8 +64,8 @@ def _build_repo_with_merge(tmp_path: Path) -> tuple[Path, str, str, str]: repo = tmp_path / "src" repo.mkdir() _run_git(repo, "init", "-b", "main") - _run_git(repo, "config", "user.name", "Alice Author") - _run_git(repo, "config", "user.email", "alice@example.com") + _run_git(repo, "config", "user.name", AUTHOR_NAME) + _run_git(repo, "config", "user.email", AUTHOR_EMAIL) _run_git(repo, "config", "uploadpack.allowFilter", "true") _run_git(repo, "config", "uploadpack.allowAnySHA1InWant", "true") (repo / "a.txt").write_text("a\n") @@ -85,13 +92,13 @@ def test_author_committer_subject_ordered_parents(self): ) assert metadata.git_commit_hash == MERGE_COMMIT_HASH - assert metadata.author_name == "Alice Author" - assert metadata.author_email == "alice@example.com" + assert metadata.author_name == AUTHOR_NAME + assert metadata.author_email == AUTHOR_EMAIL assert metadata.author_date == datetime( 2001, 9, 9, 1, 46, 40, tzinfo=timezone.utc ) - assert metadata.committer_name == "Bob Committer" - assert metadata.committer_email == "bob@example.com" + assert metadata.committer_name == COMMITTER_NAME + assert metadata.committer_email == COMMITTER_EMAIL assert metadata.committer_date == datetime( 2001, 9, 8, 20, 47, 40, tzinfo=timezone(timedelta(hours=-5)) ) @@ -217,7 +224,7 @@ def test_oversized_pack(self, tmp_path): repo = tmp_path / "bare.git" (repo / "objects" / "pack").mkdir(parents=True) pack = repo / "objects" / "pack" / "pack-deadbeef.pack" - pack.write_bytes(b"\0" * (MAX_EPHEMERAL_PACK_BYTES + 1)) + pack.write_bytes(b"\0" * (settings.GIT_FETCH_MAX_PACK_BYTES + 1)) with pytest.raises(OversizedPackError): assert_single_commit_fetch(repo) From 9d87b79bd482a87b04989feb39f01d4815235568 Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 18 Sep 2026 15:56:04 -0300 Subject: [PATCH 06/11] feat(backend): sync git commit metadata from a persistent mirror (#2079) Fetch allowlisted trees into GIT_MIRROR_DIR and ingest the tip delta into commits / commit_parents (not checkout hashes). Probe filter support, keep kernel.org packs, split mirror vs ingest crons, and exclude stored tips with ^sha on rev-list stdin. Signed-off-by: Felipe Bergamin --- .env.backend.example | 2 + .env.example | 2 + backend/kernelCI/settings.py | 12 + backend/kernelCI_app/helpers/commitSync.py | 811 ++++++++++++++++++ backend/kernelCI_app/helpers/gitCommit.py | 45 +- .../management/commands/sync_commit_ingest.py | 46 + .../management/commands/sync_commit_mirror.py | 62 ++ .../unitTests/helpers/commitSync_test.py | 572 ++++++++++++ backend/scripts/measure_git_mirror.py | 297 +++++++ docker-compose-next.yml | 2 + docker-compose.dev.yml | 2 + docker-compose.yml | 2 + docs/sync-commits.md | 245 ++++++ 13 files changed, 2089 insertions(+), 11 deletions(-) create mode 100644 backend/kernelCI_app/helpers/commitSync.py create mode 100644 backend/kernelCI_app/management/commands/sync_commit_ingest.py create mode 100644 backend/kernelCI_app/management/commands/sync_commit_mirror.py create mode 100644 backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py create mode 100755 backend/scripts/measure_git_mirror.py create mode 100644 docs/sync-commits.md diff --git a/.env.backend.example b/.env.backend.example index 7ff045c0a..4b3046db5 100644 --- a/.env.backend.example +++ b/.env.backend.example @@ -29,6 +29,8 @@ BACKEND_VOLUME_DIR=/volume_data GIT_SCRATCH_DIR=/dev/shm/kernelci-git-scratch # GIT_FETCH_TIMEOUT_SECONDS=60 # GIT_FETCH_MAX_PACK_BYTES=2097152 +# Persistent treeless git mirror for sync_commit_mirror (#2109). Dedicated volume, not BACKEND_VOLUME_DIR. +GIT_MIRROR_DIR=/var/lib/kernelci/git-mirror ## Variables used for the notifications command. Check docs/notifications.md # EMAIL_HOST_USER="youruser@host" # (optional) diff --git a/.env.example b/.env.example index 0f84f1cc3..250f434b2 100644 --- a/.env.example +++ b/.env.example @@ -110,6 +110,8 @@ BACKEND_VOLUME_DIR=/volume_data GIT_SCRATCH_DIR=/dev/shm/kernelci-git-scratch # GIT_FETCH_TIMEOUT_SECONDS=60 # GIT_FETCH_MAX_PACK_BYTES=2097152 +# Persistent treeless mirror for sync_commit_mirror (#2109). Dedicated volume. +GIT_MIRROR_DIR=/var/lib/kernelci/git-mirror # ----------------------------------------------------------------------------- # Ingester (only needed with --profile=with_commands) diff --git a/backend/kernelCI/settings.py b/backend/kernelCI/settings.py index f14af4eb9..c17e0fd1f 100644 --- a/backend/kernelCI/settings.py +++ b/backend/kernelCI/settings.py @@ -253,6 +253,16 @@ def get_json_env_var(name, default): f"--index={HARDWARE_REGISTRY_INDEX_URL}", ], ), + ( + "0 4 * * *", + "django.core.management.call_command", + ["sync_commit_mirror"], + ), + ( + "0 10 * * *", + "django.core.management.call_command", + ["sync_commit_ingest"], + ), ] # Email settings for SMTP backend @@ -276,6 +286,8 @@ def get_json_env_var(name, default): GIT_FETCH_MAX_PACK_BYTES = int( os.environ.get("GIT_FETCH_MAX_PACK_BYTES", str(2 * 1024 * 1024)) ) +# Persistent treeless mirror for commit metadata sync (#2109). Not BACKEND_VOLUME_DIR. +GIT_MIRROR_DIR = os.environ.get("GIT_MIRROR_DIR", "/var/lib/kernelci/git-mirror") DATABASE_ROUTERS = ["kernelCI_app.routers.databaseRouter.DatabaseRouter"] diff --git a/backend/kernelCI_app/helpers/commitSync.py b/backend/kernelCI_app/helpers/commitSync.py new file mode 100644 index 000000000..2bdedca09 --- /dev/null +++ b/backend/kernelCI_app/helpers/commitSync.py @@ -0,0 +1,811 @@ +"""Fill `commits` / `commit_parents` from a persistent treeless mirror (#2109). + +Parse uses the existing-repo helper from #2090. One-shot SHA fetch is only +the optional gap-fill for hashes the mirror does not cover. +""" + +from __future__ import annotations + +import hashlib +import logging +import tempfile +import time +from collections.abc import Iterator, Sequence +from enum import Enum +from pathlib import Path + +from django.conf import settings + +from kernelCI_app.constants.tree_names import TREE_NAMES_FILENAME +from kernelCI_app.helpers.gitCommit import ( + CommitMetadata, + CommitMetadataError, + FetchFailedError, + fetch_commit_metadata, + parse_commit_object, + run_git, + sanitize_git_url, +) +from kernelCI_app.helpers.logger import out +from kernelCI_app.helpers.trees import get_tree_file_data +from kernelCI_app.management.commands.treeproof import Command as TreeproofCommand +from kernelCI_app.models import Checkouts, CommitParents, Commits + +logger = logging.getLogger(__name__) + +TIPS_FILENAME = "synced-tips" +UPSERT_BATCH_SIZE = 1000 +DEFAULT_FETCH_TIMEOUT_SECONDS = 1800 +REV_LIST_TIMEOUT_SECONDS = 3600 +# One `cat-file --batch` process per chunk instead of two per commit. Chunked so a +# full-history import streams to the DB instead of buffering every message in RAM. +PARSE_BATCH_SIZE = 2000 +CAT_FILE_TIMEOUT_SECONDS = 600 +PROGRESS_LOG_EVERY = 20000 +# Runaway backstop only. A treeless mainline pack is ~850MB; an unfiltered one +# (git.kernel.org cannot filter) is ~3.5GB. Both are legitimate first fetches. +MAX_REMOTE_PACK_BYTES = 6 * 1024 * 1024 * 1024 +LS_REMOTE_TIMEOUT_SECONDS = 300 +FETCH_ATTEMPTS = 2 +# The more trees the mirror holds, the more `have` lines negotiation carries, and +# git.kernel.org's frontend starts answering HTTP 400. Unchunked HTTP/1.1 gets +# through; the low-speed deadline keeps git from sitting on the dead connection +# afterwards. Auto-gc would repack a multi-GB mirror in the middle of a run. +MIRROR_GIT_CONFIG = { + "http.version": "HTTP/1.1", + "http.postBuffer": str(512 * 1024 * 1024), + "http.lowSpeedLimit": "1000", + "http.lowSpeedTime": "60", + "gc.auto": "0", +} + + +class FetchOutcome(Enum): + OK = "ok" + FAILED = "failed" + REJECTED = "rejected" + + +def allowlisted_tree_urls(*, refresh: bool = True) -> list[str]: + """Known-good tree URLs from treeproof, not per-hash checkout git_repository_url. + + Regenerates the mapping like the ingester does, so the job does not silently + no-op when nobody ran `treeproof` on this volume yet. + """ + file_data = ( + TreeproofCommand().generate_tree_names() if refresh else get_tree_file_data() + ) + trees = file_data.get("trees") if isinstance(file_data, dict) else None + if not isinstance(trees, dict): + return [] + + urls: list[str] = [] + seen: set[str] = set() + for tree_data in trees.values(): + raw = tree_data.get("url") if isinstance(tree_data, dict) else None + cleaned = sanitize_git_url(raw) + if cleaned is None or cleaned in seen: + continue + seen.add(cleaned) + urls.append(cleaned) + return urls + + +def remote_name_for_url(url: str) -> str: + return "r" + hashlib.sha256(url.encode()).hexdigest()[:16] + + +def ensure_mirror(repo_dir: Path) -> None: + repo_dir.mkdir(parents=True, exist_ok=True) + if not (repo_dir / "HEAD").exists(): + run_git(repo_dir, "init", "--bare") + for key, value in MIRROR_GIT_CONFIG.items(): + run_git(repo_dir, "config", key, value) + + +def list_tips(repo_dir: Path) -> tuple[str, ...]: + output = run_git( + repo_dir, "for-each-ref", "--format=%(objectname)", "refs/remotes" + ).decode() + return tuple(sorted({line.strip() for line in output.splitlines() if line.strip()})) + + +def read_stored_tips(repo_dir: Path) -> tuple[str, ...]: + path = repo_dir / TIPS_FILENAME + if not path.is_file(): + return () + return tuple(line.strip() for line in path.read_text().splitlines() if line.strip()) + + +def write_stored_tips(repo_dir: Path, tips: Sequence[str]) -> None: + text = "\n".join(tips) + if text: + text += "\n" + (repo_dir / TIPS_FILENAME).write_text(text) + + +def checkout_branches_by_url() -> dict[str, set[str]]: + """Branches we already see in checkouts, keyed by sanitized git url.""" + mapping: dict[str, set[str]] = {} + try: + rows = ( + Checkouts.objects.filter(git_repository_url__isnull=False) + .exclude(git_repository_branch="") + .values_list("git_repository_url", "git_repository_branch") + .distinct() + ) + for raw_url, branch in rows: + if not branch: + continue + cleaned = sanitize_git_url(raw_url) + if cleaned is None: + continue + mapping.setdefault(cleaned, set()).add(branch) + except Exception as exc: + logger.warning("could not read checkout branches: %s", exc) + return mapping + + +def fetch_remote( + repo_dir: Path, + *, + url: str, + timeout: int = DEFAULT_FETCH_TIMEOUT_SECONDS, + verbose_git: bool = False, + branches: set[str] | None = None, + skip_unfilterable: bool = False, +) -> bool: + """Fetch the checkout branches of one remote into the shared mirror. + + Servers that cannot filter (git.kernel.org advertises only `fetch=shallow`) + send trees and blobs. We still take them: objects are shared across remotes, + so only the first kernel tree is expensive. `skip_unfilterable` opts out. + """ + name = remote_name_for_url(url) + wanted = ( + branches if branches is not None else checkout_branches_by_url().get(url, set()) + ) + try: + _ensure_remote(repo_dir, name=name, url=url) + available, supports_filter = _probe_remote(repo_dir, name) + _configure_promisor(repo_dir, name=name, enabled=supports_filter) + except CommitMetadataError as exc: + logger.warning("skip remote %s (%s): %s", name, url, exc) + return False + + if not supports_filter: + if skip_unfilterable: + out(" server cannot filter; skipping (--skip-unfilterable)") + logger.warning("skip remote %s (%s): server cannot filter", name, url) + return False + out(" server cannot filter; fetching with trees/blobs") + + branch_specs = _branch_refspecs(name, wanted, available) + if branch_specs: + out(f" refs: {len(branch_specs)} checkout branch(es)") + else: + out(" no checkout branch on this remote; fetching HEAD only") + branch_specs = [f"+HEAD:refs/remotes/{name}/HEAD"] + + for attempt in range(1, FETCH_ATTEMPTS + 1): + outcome = _fetch_refspecs_or_rollback( + repo_dir, + name=name, + url=url, + refspecs=branch_specs, + timeout=timeout, + verbose_git=verbose_git, + use_filter=supports_filter, + ) + if outcome is FetchOutcome.OK: + return True + if outcome is FetchOutcome.REJECTED: + # Same server, same filter behaviour: a retry downloads it all again. + logger.warning("skip remote %s (%s) this run: pack rejected", name, url) + return False + # A dropped connection mid-negotiation is the common failure, and the objects + # that already landed are still there, so the retry resumes instead of redoing. + if attempt < FETCH_ATTEMPTS: + out(f" fetch failed; retrying ({attempt + 1}/{FETCH_ATTEMPTS})") + + logger.warning("skip remote %s (%s) this run: fetch failed", name, url) + return False + + +def mirror_size_bytes(repo_dir: Path) -> int: + if not repo_dir.is_dir(): + return 0 + return sum(path.stat().st_size for path in repo_dir.rglob("*") if path.is_file()) + + +def new_commit_hashes(repo_dir: Path, old_tips: Sequence[str]) -> list[str]: + args = ["rev-list", "--reverse", "--topo-order", "--remotes"] + stdin: bytes | None = None + if old_tips: + # `--not --stdin` does *not* mark stdin lines uninteresting (git treats + # `--not` as applying to the next CLI revision, and `--stdin` is a flag). + # Prefix each tip with `^` instead, which rev-list does honour on stdin. + args.append("--stdin") + stdin = "".join(f"^{tip}\n" for tip in old_tips).encode() + try: + output = run_git( + repo_dir, *args, timeout=REV_LIST_TIMEOUT_SECONDS, stdin=stdin + ).decode() + except FetchFailedError as exc: + logger.warning("rev-list failed: %s", exc) + return [] + return [line.strip() for line in output.splitlines() if line.strip()] + + +def iter_parsed_commits( + repo_dir: Path, hashes: Sequence[str] +) -> Iterator[list[CommitMetadata]]: + """Yield parsed commits in chunks, keeping rev-list's topological order.""" + for start in range(0, len(hashes), PARSE_BATCH_SIZE): + chunk = hashes[start : start + PARSE_BATCH_SIZE] + stdin = ("\n".join(chunk) + "\n").encode() + try: + raw = run_git( + repo_dir, + "cat-file", + "--batch", + stdin=stdin, + timeout=CAT_FILE_TIMEOUT_SECONDS, + ) + except FetchFailedError as exc: + logger.warning("skip batch starting at %s: %s", chunk[0], exc) + continue + yield _parse_batch_output(raw) + + +def parse_commits(repo_dir: Path, hashes: Sequence[str]) -> list[CommitMetadata]: + return [ + metadata + for batch in iter_parsed_commits(repo_dir, hashes) + for metadata in batch + ] + + +def upsert_commits(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: + """Insert commits then parent edges. No stubs. Callers pass topo order.""" + commit_count = 0 + edge_count = 0 + for start in range(0, len(metadatas), UPSERT_BATCH_SIZE): + batch = metadatas[start : start + UPSERT_BATCH_SIZE] + commits, edges = _upsert_batch(batch) + commit_count += commits + edge_count += edges + return commit_count, edge_count + + +def missing_checkout_hashes() -> list[str]: + existing = Commits.objects.values("git_commit_hash") + return list( + Checkouts.objects.filter(git_commit_hash__isnull=False) + .exclude(git_commit_hash__in=existing) + .values_list("git_commit_hash", flat=True) + .distinct() + ) + + +def fill_checkout_gaps(*, dry_run: bool = False) -> tuple[int, int]: + """One-shot SHA fetch (#2090) for checkout hashes the mirror missed.""" + out("looking for checkout hashes still missing from commits...") + gaps = missing_checkout_hashes() + out(f"{len(gaps)} checkout hashes to fetch one by one") + + fetched = 0 + written = 0 + for index, git_commit_hash in enumerate(gaps, start=1): + try: + metadata = fetch_commit_metadata(git_commit_hash) + except CommitMetadataError as exc: + logger.warning("skip gap %s: %s", git_commit_hash, exc) + continue + fetched += 1 + if not dry_run: + commits, _edges = upsert_commits([metadata]) + written += commits + if index % 100 == 0 or index == len(gaps): + out(f"gaps {index}/{len(gaps)}: {fetched} fetched, {written} written") + return fetched, written + + +def _fetch_allowlisted_trees( + repo_dir: Path, + *, + dry_run: bool, + fetch_timeout: int, + verbose_git: bool, + skip_unfilterable: bool, + size_before: int, +) -> tuple[int, int]: + remotes_ok = 0 + remotes_failed = 0 + out("resolving tree allowlist...") + urls = allowlisted_tree_urls(refresh=not dry_run) + if not urls: + logger.warning( + "no allowlisted tree urls in %s; run `manage.py treeproof`", + TREE_NAMES_FILENAME, + ) + out(f"fetching {len(urls)} trees (checkout branches only, treeless)") + branches_by_url = checkout_branches_by_url() + for index, url in enumerate(urls, start=1): + out(f"[{index}/{len(urls)}] fetching {url}") + started = time.monotonic() + succeeded = fetch_remote( + repo_dir, + url=url, + timeout=fetch_timeout, + verbose_git=verbose_git, + branches=branches_by_url.get(url, set()), + skip_unfilterable=skip_unfilterable, + ) + elapsed = time.monotonic() - started + if succeeded: + remotes_ok += 1 + out(f"[{index}/{len(urls)}] done in {elapsed:.0f}s") + else: + remotes_failed += 1 + out(f"[{index}/{len(urls)}] failed after {elapsed:.0f}s, moving on") + grew = mirror_size_bytes(repo_dir) - size_before + out( + f"fetch finished: {remotes_ok} ok, {remotes_failed} failed, " + f"mirror grew {_human_bytes(grew)}" + ) + return remotes_ok, remotes_failed + + +def _ingest_new_commits( + repo_dir: Path, old_tips: Sequence[str], *, dry_run: bool +) -> tuple[int, int, int]: + out("enumerating new commit objects...") + started = time.monotonic() + hashes = new_commit_hashes(repo_dir, old_tips) + out( + f"{len(hashes)} new commits to ingest " + f"(enumerated in {time.monotonic() - started:.0f}s)" + ) + + parsed = 0 + commits_written = 0 + edges_written = 0 + logged_at = 0 + for batch in iter_parsed_commits(repo_dir, hashes): + parsed += len(batch) + if not dry_run: + commits, edges = upsert_commits(batch) + commits_written += commits + edges_written += edges + if parsed - logged_at >= PROGRESS_LOG_EVERY or parsed == len(hashes): + logged_at = parsed + out( + f"parsed {parsed}/{len(hashes)} commits, " + f"wrote {commits_written} commits and {edges_written} parent edges" + ) + + if not dry_run: + write_stored_tips(repo_dir, list_tips(repo_dir)) + out("tips recorded; next run only ingests what is new") + return parsed, commits_written, edges_written + + +def sync_commit_metadata( + *, + mirror_dir: Path | None = None, + dry_run: bool = False, + skip_fetch: bool = False, + skip_ingest: bool = False, + fill_gaps: bool = False, + fetch_timeout: int = DEFAULT_FETCH_TIMEOUT_SECONDS, + verbose_git: bool = False, + skip_unfilterable: bool = False, +) -> dict[str, int]: + if skip_fetch and skip_ingest: + raise ValueError("skip_fetch and skip_ingest cannot both be set") + + repo_dir = Path(mirror_dir or settings.GIT_MIRROR_DIR) + ensure_mirror(repo_dir) + old_tips = read_stored_tips(repo_dir) + size_before = mirror_size_bytes(repo_dir) + out( + f"mirror={repo_dir} size={_human_bytes(size_before)} " + f"known_tips={len(old_tips)} dry_run={dry_run} " + f"skip_fetch={skip_fetch} skip_ingest={skip_ingest}" + ) + + remotes_ok = 0 + remotes_failed = 0 + if skip_fetch: + out("skipping fetch, using objects already in the mirror") + else: + remotes_ok, remotes_failed = _fetch_allowlisted_trees( + repo_dir, + dry_run=dry_run, + fetch_timeout=fetch_timeout, + verbose_git=verbose_git, + skip_unfilterable=skip_unfilterable, + size_before=size_before, + ) + + empty = { + "remotes_ok": remotes_ok, + "remotes_failed": remotes_failed, + "parsed": 0, + "commits_written": 0, + "edges_written": 0, + "gaps_fetched": 0, + "gaps_written": 0, + } + if skip_ingest: + out("skipping ingest; tips unchanged") + return empty + + parsed, commits_written, edges_written = _ingest_new_commits( + repo_dir, old_tips, dry_run=dry_run + ) + gaps_fetched = 0 + gaps_written = 0 + if fill_gaps: + gaps_fetched, gaps_written = fill_checkout_gaps(dry_run=dry_run) + + return { + "remotes_ok": remotes_ok, + "remotes_failed": remotes_failed, + "parsed": parsed, + "commits_written": commits_written, + "edges_written": edges_written, + "gaps_fetched": gaps_fetched, + "gaps_written": gaps_written, + } + + +def _human_bytes(num_bytes: int) -> str: + size = float(num_bytes) + for unit in ("B", "KiB", "MiB", "GiB"): + if abs(size) < 1024: + return f"{size:.1f}{unit}" + size /= 1024 + return f"{size:.1f}TiB" + + +def _parse_batch_output(raw: bytes) -> list[CommitMetadata]: + """Split `cat-file --batch` records: ` \\n\\n`.""" + parsed: list[CommitMetadata] = [] + offset = 0 + while offset < len(raw): + line_end = raw.find(b"\n", offset) + if line_end == -1: + break + header = raw[offset:line_end].decode("utf-8", errors="replace") + offset = line_end + 1 + + fields = header.split() + if len(fields) < 3: + # " missing" / " ambiguous": no payload follows. + logger.warning("skip cat-file entry: %s", header) + continue + + oid, object_type, size_text = fields[0], fields[1], fields[2] + try: + size = int(size_text) + except ValueError: + logger.warning("unparseable cat-file header, dropping batch: %s", header) + break + + payload = raw[offset : offset + size] + offset += size + 1 + if object_type != "commit": + continue + + try: + parsed.append( + parse_commit_object( + raw=payload.decode("utf-8", errors="replace"), git_commit_hash=oid + ) + ) + except CommitMetadataError as exc: + logger.warning("skip parse %s: %s", oid, exc) + return parsed + + +def _probe_remote(repo_dir: Path, name: str) -> tuple[set[str], bool]: + """Return (branch names, server supports partial-clone filter). + + One ls-remote answers both. Capabilities are only exposed in git's packet + trace, so we point GIT_TRACE_PACKET at a file and read the `fetch=` line. + """ + with tempfile.TemporaryDirectory() as scratch: + trace = Path(scratch) / "packet-trace" + try: + output = run_git( + repo_dir, + "-c", + "protocol.version=2", + "ls-remote", + "--heads", + name, + timeout=LS_REMOTE_TIMEOUT_SECONDS, + extra_env={"GIT_TRACE_PACKET": str(trace)}, + ).decode() + except CommitMetadataError as exc: + logger.warning("ls-remote %s failed: %s", name, exc) + return set(), False + supports_filter = _trace_advertises_filter(trace) + + heads: set[str] = set() + for line in output.splitlines(): + parts = line.split() + if len(parts) < 2 or not parts[1].startswith("refs/heads/"): + continue + heads.add(parts[1].removeprefix("refs/heads/")) + return heads, supports_filter + + +def _trace_advertises_filter(trace: Path) -> bool: + if not trace.is_file(): + return False + for line in trace.read_text(errors="replace").splitlines(): + _, separator, capability = line.partition("fetch=") + if separator and "filter" in capability.split(): + return True + return False + + +def _branch_refspecs(name: str, wanted: set[str], available: set[str]) -> list[str]: + if not wanted: + return [] + names = sorted(wanted & available) if available else sorted(wanted) + return [f"+refs/heads/{branch}:refs/remotes/{name}/{branch}" for branch in names] + + +def _fetch_refspecs_or_rollback( + repo_dir: Path, + *, + name: str, + url: str, + refspecs: list[str], + timeout: int, + verbose_git: bool, + use_filter: bool = True, +) -> FetchOutcome: + packs_before = _pack_paths(repo_dir) + refs_before = _remote_refs(repo_dir, name) + try: + run_git( + repo_dir, + "fetch", + "--prune", + "--no-tags", + *(["--filter=tree:0"] if use_filter else []), + *(["--progress"] if verbose_git else []), + name, + *refspecs, + timeout=timeout, + stream=verbose_git, + ) + except CommitMetadataError as exc: + logger.warning("fetch %s (%s) failed: %s", name, url, exc) + _rollback_fetch( + repo_dir, + name=name, + packs_before=packs_before, + refs_before=refs_before, + keep_packs=True, + ) + return FetchOutcome.FAILED + + reason = _new_packs_rejected(repo_dir, packs_before, expect_treeless=use_filter) + if reason is None: + return FetchOutcome.OK + + logger.warning("reject pack from %s (%s): %s", name, url, reason) + _rollback_fetch( + repo_dir, name=name, packs_before=packs_before, refs_before=refs_before + ) + return FetchOutcome.REJECTED + + +def _pack_paths(repo_dir: Path) -> set[Path]: + pack_dir = repo_dir / "objects" / "pack" + if not pack_dir.is_dir(): + return set() + return {path for path in pack_dir.iterdir() if path.is_file()} + + +def _remote_refs(repo_dir: Path, name: str) -> dict[str, str]: + try: + output = run_git( + repo_dir, + "for-each-ref", + "--format=%(objectname) %(refname)", + f"refs/remotes/{name}", + ).decode() + except FetchFailedError: + return {} + refs: dict[str, str] = {} + for line in output.splitlines(): + sha, _, ref = line.partition(" ") + if sha and ref: + refs[ref] = sha + return refs + + +def _rollback_fetch( + repo_dir: Path, + *, + name: str, + packs_before: set[Path], + refs_before: dict[str, str], + keep_packs: bool = False, +) -> None: + """Restore refs, and drop the new packs unless the caller wants to keep them. + + A pack that finished writing holds valid objects even when the fetch died + afterwards, and git shares them with every other remote. Deleting it means + paying for the same download again; only the unfinished `tmp_pack_*` is junk. + """ + current_refs = _remote_refs(repo_dir, name) + for ref in current_refs: + if ref not in refs_before: + try: + run_git(repo_dir, "update-ref", "-d", ref) + except FetchFailedError as exc: + logger.warning("could not drop ref %s: %s", ref, exc) + for ref, sha in refs_before.items(): + try: + run_git(repo_dir, "update-ref", ref, sha) + except FetchFailedError as exc: + logger.warning("could not restore ref %s: %s", ref, exc) + + for path in _pack_paths(repo_dir) - packs_before: + if keep_packs and not path.name.startswith("tmp_pack"): + continue + try: + path.unlink() + except OSError as exc: + logger.warning("could not remove %s: %s", path, exc) + + +def _new_packs_rejected( + repo_dir: Path, packs_before: set[Path], *, expect_treeless: bool +) -> str | None: + for path in sorted(_pack_paths(repo_dir) - packs_before): + reason = _pack_rejection_reason(repo_dir, path, expect_treeless=expect_treeless) + if reason is not None: + return reason + return None + + +def _pack_rejection_reason( + repo_dir: Path, path: Path, *, expect_treeless: bool +) -> str | None: + if path.name.startswith("tmp_pack"): + return f"incomplete pack {path.name}" + if path.suffix != ".pack": + return None + + size = path.stat().st_size + if size > MAX_REMOTE_PACK_BYTES: + return f"{path.name} is {size} bytes (max {MAX_REMOTE_PACK_BYTES})" + + if not expect_treeless: + return None + + idx = path.with_suffix(".idx") + if not idx.is_file(): + return f"{path.name} has no index" + + # The server said it could filter, so trees/blobs mean it did not honour it. + types = _pack_object_types(repo_dir, idx) + leaked = types & {"tree", "blob"} + if leaked: + return f"{path.name} contains {', '.join(sorted(leaked))}" + return None + + +def _pack_object_types(repo_dir: Path, idx: Path) -> set[str]: + try: + output = run_git( + repo_dir, + "verify-pack", + "-v", + str(idx.relative_to(repo_dir)), + timeout=120, + ).decode() + except (CommitMetadataError, ValueError) as exc: + logger.warning("verify-pack %s failed: %s", idx, exc) + return set() + types: set[str] = set() + for line in output.splitlines(): + fields = line.split() + if len(fields) >= 2 and fields[1] in {"commit", "tree", "blob", "tag"}: + types.add(fields[1]) + return types + + +def _ensure_remote(repo_dir: Path, *, name: str, url: str) -> None: + existing = _remote_urls(repo_dir) + if name not in existing: + run_git(repo_dir, "remote", "add", name, url) + elif existing[name] != url: + run_git(repo_dir, "remote", "set-url", name, url) + + +def _configure_promisor(repo_dir: Path, *, name: str, enabled: bool) -> None: + for key, value in (("promisor", "true"), ("partialclonefilter", "tree:0")): + try: + if enabled: + run_git(repo_dir, "config", f"remote.{name}.{key}", value) + else: + run_git(repo_dir, "config", "--unset", f"remote.{name}.{key}") + except FetchFailedError: + # `--unset` exits non-zero when the key was never set. + pass + + +def _remote_urls(repo_dir: Path) -> dict[str, str]: + try: + output = run_git(repo_dir, "remote", "-v").decode() + except FetchFailedError: + return {} + urls: dict[str, str] = {} + for line in output.splitlines(): + parts = line.split() + if len(parts) >= 2 and parts[0] not in urls: + urls[parts[0]] = parts[1] + return urls + + +def _upsert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: + rows = [ + Commits( + git_commit_hash=metadata.git_commit_hash, + author_name=metadata.author_name, + author_email=metadata.author_email, + author_date=metadata.author_date, + committer_name=metadata.committer_name, + committer_email=metadata.committer_email, + committer_date=metadata.committer_date, + subject=metadata.subject, + message=metadata.message, + ) + for metadata in metadatas + ] + Commits.objects.bulk_create( + rows, ignore_conflicts=True, batch_size=UPSERT_BATCH_SIZE + ) + + hashes = {metadata.git_commit_hash for metadata in metadatas} + for metadata in metadatas: + hashes.update(metadata.parent_hashes) + hash_to_id = dict( + Commits.objects.filter(git_commit_hash__in=hashes).values_list( + "git_commit_hash", "id" + ) + ) + + edges: list[CommitParents] = [] + for metadata in metadatas: + commit_id = hash_to_id.get(metadata.git_commit_hash) + if commit_id is None: + continue + for ord_, parent_hash in enumerate(metadata.parent_hashes): + parent_id = hash_to_id.get(parent_hash) + if parent_id is None: + logger.warning( + "skip parent edge %s -> %s (parent not in commits)", + metadata.git_commit_hash, + parent_hash, + ) + continue + edges.append( + CommitParents(commit_id=commit_id, parent_id=parent_id, ord=ord_) + ) + + if edges: + CommitParents.objects.bulk_create( + edges, ignore_conflicts=True, batch_size=UPSERT_BATCH_SIZE + ) + return len(metadatas), len(edges) diff --git a/backend/kernelCI_app/helpers/gitCommit.py b/backend/kernelCI_app/helpers/gitCommit.py index 83334ce9d..e89dd5d45 100644 --- a/backend/kernelCI_app/helpers/gitCommit.py +++ b/backend/kernelCI_app/helpers/gitCommit.py @@ -10,6 +10,7 @@ import os import re import shutil +import signal import subprocess import tempfile from dataclasses import dataclass @@ -332,22 +333,44 @@ def _git_executable() -> str: return git -def _git(repo_dir: Path, *args: str, timeout: int = 30) -> bytes: +def run_git( + repo_dir: Path, + *args: str, + timeout: int = 30, + stdin: bytes | None = None, + stream: bool = False, + extra_env: dict[str, str] | None = None, +) -> bytes: + """Run git in repo_dir. With stream=True git writes straight to our console, + so long fetches show their own progress; stdout is then not captured.""" env = os.environ.copy() env.update(_GIT_ENV) + if extra_env: + env.update(extra_env) command = [_git_executable(), "-C", str(repo_dir), *args] + pipe = None if stream else subprocess.PIPE + # Own session so the timeout kills git *and* the curl/remote-https children it + # spawned. subprocess.run only signals git itself, and a child still holding the + # pipes blocks the read forever - a stalled fetch then hangs the whole job. + process = subprocess.Popen( # noqa: S603 + command, + stdin=subprocess.PIPE if stdin is not None else None, + stdout=pipe, + stderr=pipe, + env=env, + start_new_session=True, + ) try: - result = subprocess.run( # noqa: S603 - command, - check=False, - capture_output=True, - timeout=timeout, - env=env, - ) + stdout, stderr_bytes = process.communicate(input=stdin, timeout=timeout) except subprocess.TimeoutExpired as exc: + os.killpg(process.pid, signal.SIGKILL) + process.communicate() raise FetchFailedError(f"git {' '.join(args)} timed out") from exc - if result.returncode != 0: - stderr = result.stderr.decode("utf-8", errors="replace").strip() + if process.returncode != 0: + stderr = (stderr_bytes or b"").decode("utf-8", errors="replace").strip() raise FetchFailedError(f"git {' '.join(args)} failed: {stderr}") - return result.stdout + return stdout or b"" + + +_git = run_git diff --git a/backend/kernelCI_app/management/commands/sync_commit_ingest.py b/backend/kernelCI_app/management/commands/sync_commit_ingest.py new file mode 100644 index 000000000..b1119bcbf --- /dev/null +++ b/backend/kernelCI_app/management/commands/sync_commit_ingest.py @@ -0,0 +1,46 @@ +from pathlib import Path + +from django.conf import settings +from django.core.management.base import BaseCommand + +from kernelCI_app.helpers.commitSync import sync_commit_metadata +from kernelCI_app.helpers.logger import out + + +class Command(BaseCommand): + help = ( + "Upsert commits / commit_parents from objects already in the persistent " + "mirror (tip delta). Does not write checkouts.git_commit_message." + ) + + def add_arguments(self, parser): + parser.add_argument( + "--dry-run", + action="store_true", + help="Parse as requested, write nothing to the database or tips file.", + ) + parser.add_argument( + "--fill-gaps", + action="store_true", + help="One-shot SHA fetch for checkout hashes still missing from commits.", + ) + parser.add_argument( + "--mirror-dir", + type=str, + default=None, + help="Persistent treeless repo path (default: GIT_MIRROR_DIR).", + ) + + def handle(self, *args, **options): + mirror_dir = options["mirror_dir"] or settings.GIT_MIRROR_DIR + stats = sync_commit_metadata( + mirror_dir=Path(mirror_dir), + dry_run=options["dry_run"], + skip_fetch=True, + fill_gaps=options["fill_gaps"], + ) + out( + "sync_commit_ingest parsed=%(parsed)s commits_written=%(commits_written)s " + "edges_written=%(edges_written)s gaps_fetched=%(gaps_fetched)s " + "gaps_written=%(gaps_written)s" % stats + ) diff --git a/backend/kernelCI_app/management/commands/sync_commit_mirror.py b/backend/kernelCI_app/management/commands/sync_commit_mirror.py new file mode 100644 index 000000000..1f4306c50 --- /dev/null +++ b/backend/kernelCI_app/management/commands/sync_commit_mirror.py @@ -0,0 +1,62 @@ +from pathlib import Path + +from django.conf import settings +from django.core.management.base import BaseCommand + +from kernelCI_app.helpers.commitSync import ( + DEFAULT_FETCH_TIMEOUT_SECONDS, + sync_commit_metadata, +) +from kernelCI_app.helpers.logger import out + + +class Command(BaseCommand): + help = ( + "Fetch allowlisted git trees into the persistent mirror. " + "Does not write commits, commit_parents, or synced-tips." + ) + + def add_arguments(self, parser): + parser.add_argument( + "--dry-run", + action="store_true", + help="Fetch as requested, do not regenerate tree-names.yaml.", + ) + parser.add_argument( + "--mirror-dir", + type=str, + default=None, + help="Persistent treeless repo path (default: GIT_MIRROR_DIR).", + ) + parser.add_argument( + "--fetch-timeout", + type=int, + default=DEFAULT_FETCH_TIMEOUT_SECONDS, + help="Per-remote git fetch timeout in seconds.", + ) + parser.add_argument( + "--skip-unfilterable", + action="store_true", + help="""Skip servers without partial-clone support instead of taking + their trees/blobs. Smaller mirror, fewer trees covered.""", + ) + parser.add_argument( + "--verbose-git", + action="store_true", + help="Let git print its own fetch progress. Noisy; meant for watching a run.", + ) + + def handle(self, *args, **options): + mirror_dir = options["mirror_dir"] or settings.GIT_MIRROR_DIR + stats = sync_commit_metadata( + mirror_dir=Path(mirror_dir), + dry_run=options["dry_run"], + skip_ingest=True, + fetch_timeout=options["fetch_timeout"], + verbose_git=options["verbose_git"], + skip_unfilterable=options["skip_unfilterable"], + ) + out( + "sync_commit_mirror remotes_ok=%(remotes_ok)s remotes_failed=%(remotes_failed)s" + % stats + ) diff --git a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py new file mode 100644 index 000000000..d74a0e76f --- /dev/null +++ b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py @@ -0,0 +1,572 @@ +import os +import shutil +import subprocess +from datetime import datetime, timezone +from pathlib import Path + +import pytest +from django.core.management import call_command +from django.test import override_settings + +from kernelCI_app.helpers import commitSync +from kernelCI_app.helpers.commitSync import ( + TIPS_FILENAME, + _parse_batch_output, + allowlisted_tree_urls, + ensure_mirror, + fetch_remote, + list_tips, + new_commit_hashes, + parse_commits, + sync_commit_metadata, + upsert_commits, +) +from kernelCI_app.helpers.gitCommit import CommitMetadata + + +def _run_git(repo: Path, *args: str) -> str: + env = { + **os.environ, + "GIT_AUTHOR_NAME": "Alice Author", + "GIT_AUTHOR_EMAIL": "alice@example.com", + "GIT_COMMITTER_NAME": "Bob Committer", + "GIT_COMMITTER_EMAIL": "bob@example.com", + "GIT_AUTHOR_DATE": "2001-09-09T01:46:40+0000", + "GIT_COMMITTER_DATE": "2001-09-09T01:47:40+0000", + } + git = shutil.which("git") + assert git is not None + result = subprocess.run( # noqa: S603 + [git, "-C", str(repo), *args], + check=True, + capture_output=True, + text=True, + env=env, + ) + return result.stdout.strip() + + +def _linear_repo(tmp_path: Path) -> tuple[Path, list[str]]: + repo = tmp_path / "src" + repo.mkdir() + _run_git(repo, "init", "-b", "main") + _run_git(repo, "config", "user.name", "Alice Author") + _run_git(repo, "config", "user.email", "alice@example.com") + _run_git(repo, "config", "uploadpack.allowFilter", "true") + hashes: list[str] = [] + for name in ("root", "skipped", "tip"): + (repo / "f.txt").write_text(name + "\n") + _run_git(repo, "add", "f.txt") + _run_git(repo, "commit", "-m", name) + hashes.append(_run_git(repo, "rev-parse", "HEAD")) + return repo, hashes + + +def _metadata( + git_commit_hash: str, + *parents: str, + subject: str = "subj", +) -> CommitMetadata: + now = datetime(2001, 9, 9, 1, 46, 40, tzinfo=timezone.utc) + return CommitMetadata( + git_commit_hash=git_commit_hash, + author_name="Alice Author", + author_email="alice@example.com", + author_date=now, + committer_name="Bob Committer", + committer_email="bob@example.com", + committer_date=now, + subject=subject, + message=subject + "\n", + parent_hashes=parents, + ) + + +@pytest.fixture +def commit_store(monkeypatch): + commits_by_hash: dict[str, object] = {} + parent_rows: list[object] = [] + next_id = {"n": 1} + + class _Filter: + def __init__(self, hashes: set[str]): + self.hashes = hashes + + def values_list(self, *_args): + return [ + (git_hash, obj.id) + for git_hash, obj in commits_by_hash.items() + if git_hash in self.hashes + ] + + class _Commits: + def bulk_create(self, rows, **_kwargs): + for row in rows: + if row.git_commit_hash in commits_by_hash: + continue + row.id = next_id["n"] + next_id["n"] += 1 + commits_by_hash[row.git_commit_hash] = row + return rows + + def filter(self, git_commit_hash__in=None): + return _Filter(set(git_commit_hash__in or [])) + + def values(self, *_args): + return [{"git_commit_hash": git_hash} for git_hash in commits_by_hash] + + class _Parents: + def bulk_create(self, rows, **_kwargs): + parent_rows.extend(rows) + return rows + + monkeypatch.setattr("kernelCI_app.helpers.commitSync.Commits.objects", _Commits()) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.CommitParents.objects", _Parents() + ) + return commits_by_hash, parent_rows + + +_TREES_FILE = { + "trees": { + "bad": {"url": "git@github.com:org/repo.git"}, + "mainline": { + "url": "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git/" + }, + "dup": { + "url": "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git" + }, + } +} +_MAINLINE_URL = "https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git" + + +class TestAllowlistedTreeUrls: + def test_skips_malformed_and_dedupes(self, monkeypatch): + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.TreeproofCommand.generate_tree_names", + lambda _self: _TREES_FILE, + ) + assert allowlisted_tree_urls() == [_MAINLINE_URL] + + def test_without_refresh_reads_file_only(self, monkeypatch): + def fail(_self): + raise AssertionError("should not regenerate the tree names file") + + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.TreeproofCommand.generate_tree_names", fail + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.get_tree_file_data", + lambda: _TREES_FILE, + ) + assert allowlisted_tree_urls(refresh=False) == [_MAINLINE_URL] + + +class TestParseBatchOutput: + def test_reads_records_and_skips_missing(self): + good = b"tree abc\nauthor A 1000000000 +0000\n\nsubject line\n" + raw = ( + b"%s commit %d\n%s\n" % (b"aa" * 20, len(good), good) + + b"%s missing\n" % (b"bb" * 20) + + b"%s commit %d\n%s\n" % (b"cc" * 20, len(good), good) + ) + + parsed = _parse_batch_output(raw) + + assert [item.git_commit_hash for item in parsed] == ["aa" * 20, "cc" * 20] + assert parsed[0].subject == "subject line" + + def test_payload_with_newlines_does_not_desync_records(self): + body = b"tree abc\nauthor A 1000000000 +0000\n\nfirst\n\nsecond\n" + raw = b"%s commit %d\n%s\n%s commit %d\n%s\n" % ( + b"aa" * 20, + len(body), + body, + b"cc" * 20, + len(body), + body, + ) + + parsed = _parse_batch_output(raw) + + assert len(parsed) == 2 + assert parsed[1].git_commit_hash == "cc" * 20 + assert parsed[1].message == "first\n\nsecond\n" + + +class TestUpsertCommits: + def test_topo_order_first_parent_edges(self, commit_store): + commits_by_hash, parent_rows = commit_store + root = _metadata("aa" * 20, subject="root") + skipped = _metadata("bb" * 20, "aa" * 20, subject="skipped") + tip = _metadata("cc" * 20, "bb" * 20, subject="tip") + + commit_count, edge_count = upsert_commits([root, skipped, tip]) + + assert commit_count == 3 + assert edge_count == 2 + assert set(commits_by_hash) == {"aa" * 20, "bb" * 20, "cc" * 20} + ords = {(row.commit_id, row.parent_id, row.ord) for row in parent_rows} + assert ords == { + (commits_by_hash["bb" * 20].id, commits_by_hash["aa" * 20].id, 0), + (commits_by_hash["cc" * 20].id, commits_by_hash["bb" * 20].id, 0), + } + + def test_skips_edge_when_parent_row_missing(self, commit_store): + _commits_by_hash, parent_rows = commit_store + child = _metadata("dd" * 20, "ee" * 20, subject="orphan-parent") + + commit_count, edge_count = upsert_commits([child]) + + assert commit_count == 1 + assert edge_count == 0 + assert parent_rows == [] + + +class TestSyncCommitMetadata: + def test_skipped_intermediate_commit_is_ingested( + self, tmp_path, monkeypatch, commit_store + ): + repo, hashes = _linear_repo(tmp_path) + root, skipped, tip = hashes + mirror = tmp_path / "mirror" + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.allowlisted_tree_urls", + lambda **_kwargs: [f"file://{repo}"], + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.checkout_branches_by_url", + lambda: {}, + ) + + stats = sync_commit_metadata(mirror_dir=mirror) + commits_by_hash, parent_rows = commit_store + + assert stats["remotes_ok"] == 1 + assert stats["remotes_failed"] == 0 + assert stats["commits_written"] == 3 + assert skipped in commits_by_hash + assert set(commits_by_hash) == {root, skipped, tip} + first_parents = { + (row.commit_id, row.parent_id, row.ord) + for row in parent_rows + if row.ord == 0 + } + assert ( + commits_by_hash[skipped].id, + commits_by_hash[root].id, + 0, + ) in first_parents + assert ( + commits_by_hash[tip].id, + commits_by_hash[skipped].id, + 0, + ) in first_parents + + def test_one_remote_fails_others_still_write( + self, tmp_path, monkeypatch, commit_store + ): + repo, hashes = _linear_repo(tmp_path) + mirror = tmp_path / "mirror" + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.allowlisted_tree_urls", + lambda **_kwargs: ["file:///no/such/repo.git", f"file://{repo}"], + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.checkout_branches_by_url", + lambda: {}, + ) + + stats = sync_commit_metadata(mirror_dir=mirror) + + assert stats["remotes_failed"] == 1 + assert stats["remotes_ok"] == 1 + assert stats["commits_written"] == 3 + assert set(commit_store[0]) == set(hashes) + + def test_dry_run_writes_nothing(self, tmp_path, monkeypatch, commit_store): + repo, _hashes = _linear_repo(tmp_path) + mirror = tmp_path / "mirror" + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.allowlisted_tree_urls", + lambda **_kwargs: [f"file://{repo}"], + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.checkout_branches_by_url", + lambda: {}, + ) + + stats = sync_commit_metadata(mirror_dir=mirror, dry_run=True) + + assert stats["parsed"] == 3 + assert stats["commits_written"] == 0 + assert commit_store[0] == {} + assert commit_store[1] == [] + assert not (mirror / TIPS_FILENAME).exists() + + def test_fetch_only_then_ingest_matches_combined( + self, tmp_path, monkeypatch, commit_store + ): + repo, hashes = _linear_repo(tmp_path) + mirror = tmp_path / "mirror" + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.allowlisted_tree_urls", + lambda **_kwargs: [f"file://{repo}"], + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.checkout_branches_by_url", + lambda: {}, + ) + + fetch_stats = sync_commit_metadata(mirror_dir=mirror, skip_ingest=True) + assert fetch_stats["remotes_ok"] == 1 + assert fetch_stats["commits_written"] == 0 + assert commit_store[0] == {} + assert not (mirror / TIPS_FILENAME).exists() + + ingest_stats = sync_commit_metadata(mirror_dir=mirror, skip_fetch=True) + assert ingest_stats["commits_written"] == 3 + assert set(commit_store[0]) == set(hashes) + assert (mirror / TIPS_FILENAME).is_file() + + again = sync_commit_metadata(mirror_dir=mirror, skip_fetch=True) + assert again["parsed"] == 0 + assert again["commits_written"] == 0 + + def test_skip_fetch_and_skip_ingest_is_invalid(self, tmp_path): + with pytest.raises(ValueError, match="cannot both be set"): + sync_commit_metadata( + mirror_dir=tmp_path / "mirror", + skip_fetch=True, + skip_ingest=True, + ) + + def test_enumerate_includes_skipped_without_db(self, tmp_path): + repo, hashes = _linear_repo(tmp_path) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + assert fetch_remote(mirror, url=f"file://{repo}", timeout=30) + enumerated = new_commit_hashes(mirror, ()) + parsed = parse_commits(mirror, enumerated) + assert [item.git_commit_hash for item in parsed] == hashes + assert new_commit_hashes(mirror, list_tips(mirror)) == [] + + def test_fill_gaps_uses_one_shot_fetch(self, tmp_path, monkeypatch, commit_store): + gap_hash = "ff" * 20 + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.allowlisted_tree_urls", + lambda **_kwargs: [], + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.missing_checkout_hashes", + lambda: [gap_hash], + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.fetch_commit_metadata", + lambda git_commit_hash: _metadata(git_commit_hash, subject="gap"), + ) + + stats = sync_commit_metadata( + mirror_dir=tmp_path / "mirror", + skip_fetch=True, + fill_gaps=True, + ) + + assert stats["gaps_fetched"] == 1 + assert stats["gaps_written"] == 1 + assert gap_hash in commit_store[0] + + +class TestFetchGuards: + def test_does_not_fetch_unlisted_topic_branch(self, tmp_path, monkeypatch): + repo, hashes = _linear_repo(tmp_path) + _run_git(repo, "checkout", "-b", "topic") + (repo / "f.txt").write_text("topic\n") + _run_git(repo, "add", "f.txt") + _run_git(repo, "commit", "-m", "topic only") + topic = _run_git(repo, "rev-parse", "HEAD") + _run_git(repo, "checkout", "main") + + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + assert fetch_remote( + mirror, + url=f"file://{repo}", + timeout=30, + branches={"main"}, + ) + enumerated = new_commit_hashes(mirror, ()) + assert topic not in enumerated + assert set(enumerated) == set(hashes) + + def test_oversized_pack_is_rolled_back(self, tmp_path, monkeypatch): + repo, _hashes = _linear_repo(tmp_path) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.MAX_REMOTE_PACK_BYTES", + 1, + ) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + assert ( + fetch_remote(mirror, url=f"file://{repo}", timeout=30, branches=set()) + is False + ) + pack_dir = mirror / "objects" / "pack" + packs = list(pack_dir.glob("*.pack")) if pack_dir.is_dir() else [] + assert packs == [] + assert new_commit_hashes(mirror, ()) == [] + + def test_rejected_pack_is_not_downloaded_twice(self, tmp_path, monkeypatch): + repo, _hashes = _linear_repo(tmp_path) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.MAX_REMOTE_PACK_BYTES", + 1, + ) + attempts: list[list[str]] = [] + real_fetch = commitSync._fetch_refspecs_or_rollback + + def counting_fetch(repo_dir, *, refspecs, **kwargs): + attempts.append(refspecs) + return real_fetch(repo_dir, refspecs=refspecs, **kwargs) + + monkeypatch.setattr(commitSync, "_fetch_refspecs_or_rollback", counting_fetch) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + + assert ( + fetch_remote(mirror, url=f"file://{repo}", timeout=30, branches={"main"}) + is False + ) + assert len(attempts) == 1 + + def test_failed_fetch_retries_and_keeps_objects(self, tmp_path, monkeypatch): + repo, hashes = _linear_repo(tmp_path) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + real_fetch = commitSync._fetch_refspecs_or_rollback + attempts: list[int] = [] + + def fail_once(repo_dir, **kwargs): + attempts.append(1) + if len(attempts) == 1: + # Let git write the pack, then report the transport dying after it. + real_fetch(repo_dir, **kwargs) + return commitSync.FetchOutcome.FAILED + return real_fetch(repo_dir, **kwargs) + + monkeypatch.setattr(commitSync, "_fetch_refspecs_or_rollback", fail_once) + + assert fetch_remote(mirror, url=f"file://{repo}", timeout=30, branches={"main"}) + assert len(attempts) == 2 + assert set(new_commit_hashes(mirror, ())) == set(hashes) + + def test_failed_fetch_keeps_downloaded_pack(self, tmp_path, monkeypatch): + repo, _hashes = _linear_repo(tmp_path) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + real_fetch = commitSync._fetch_refspecs_or_rollback + + def always_fail(repo_dir, **kwargs): + real_fetch(repo_dir, **kwargs) + return commitSync.FetchOutcome.FAILED + + monkeypatch.setattr(commitSync, "_fetch_refspecs_or_rollback", always_fail) + + assert ( + fetch_remote(mirror, url=f"file://{repo}", timeout=30, branches={"main"}) + is False + ) + # Objects stay so the next run resumes instead of paying for them again. + assert list((mirror / "objects" / "pack").glob("*.pack")) + + def test_unfilterable_server_is_still_fetched(self, tmp_path, monkeypatch): + repo, hashes = _linear_repo(tmp_path) + monkeypatch.setattr( + commitSync, + "_probe_remote", + lambda _repo_dir, _name: ({"main"}, False), + ) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + + assert fetch_remote(mirror, url=f"file://{repo}", timeout=30, branches={"main"}) + assert set(new_commit_hashes(mirror, ())) == set(hashes) + + def test_skip_unfilterable_opts_out(self, tmp_path, monkeypatch): + repo, _hashes = _linear_repo(tmp_path) + monkeypatch.setattr( + commitSync, + "_probe_remote", + lambda _repo_dir, _name: ({"main"}, False), + ) + + def fail(*_args, **_kwargs): + raise AssertionError("must not fetch when opted out") + + monkeypatch.setattr(commitSync, "_fetch_refspecs_or_rollback", fail) + mirror = tmp_path / "mirror" + ensure_mirror(mirror) + + assert ( + fetch_remote( + mirror, + url=f"file://{repo}", + timeout=30, + branches={"main"}, + skip_unfilterable=True, + ) + is False + ) + + +class TestSyncCommitCommands: + def test_mirror_command_skips_ingest(self, monkeypatch): + seen: dict[str, object] = {} + + def fake_sync(**kwargs): + seen.update(kwargs) + return { + "remotes_ok": 0, + "remotes_failed": 0, + "parsed": 0, + "commits_written": 0, + "edges_written": 0, + "gaps_fetched": 0, + "gaps_written": 0, + } + + monkeypatch.setattr( + "kernelCI_app.management.commands.sync_commit_mirror.sync_commit_metadata", + fake_sync, + ) + with override_settings(GIT_MIRROR_DIR="/tmp/mirror"): + call_command("sync_commit_mirror", "--dry-run") + assert seen["skip_ingest"] is True + assert seen["dry_run"] is True + assert "skip_fetch" not in seen or seen.get("skip_fetch") is False + + def test_ingest_command_skips_fetch(self, monkeypatch): + seen: dict[str, object] = {} + + def fake_sync(**kwargs): + seen.update(kwargs) + return { + "remotes_ok": 0, + "remotes_failed": 0, + "parsed": 0, + "commits_written": 0, + "edges_written": 0, + "gaps_fetched": 0, + "gaps_written": 0, + } + + monkeypatch.setattr( + "kernelCI_app.management.commands.sync_commit_ingest.sync_commit_metadata", + fake_sync, + ) + with override_settings(GIT_MIRROR_DIR="/tmp/mirror"): + call_command("sync_commit_ingest", "--fill-gaps") + assert seen["skip_fetch"] is True + assert seen["fill_gaps"] is True + assert "skip_ingest" not in seen or seen.get("skip_ingest") is False diff --git a/backend/scripts/measure_git_mirror.py b/backend/scripts/measure_git_mirror.py new file mode 100755 index 000000000..22e4be0a5 --- /dev/null +++ b/backend/scripts/measure_git_mirror.py @@ -0,0 +1,297 @@ +#!/usr/bin/env python3 +"""Grow a bare mirror the same way sync_commit_mirror fetches, and print disk after each URL. + +No Django, no DB writes. Reads one git URL per line (comments and blanks skipped). + +Without checkout branches we fetch HEAD only — the job's fallback, and the bulk of +object-store growth. Extra checkout branches mostly share those objects. + +Pack rejection is skipped on purpose: a 3.5GB kernel.org pack is what we want to see. + + python3 backend/scripts/measure_git_mirror.py --urls urls.txt --mirror-dir /tmp/git-mirror +""" + +from __future__ import annotations + +import argparse +import hashlib +import os +import signal +import subprocess +import sys +import tempfile +import time +from pathlib import Path + +DEFAULT_FETCH_TIMEOUT = 1800 +LS_REMOTE_TIMEOUT = 300 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--urls", + type=Path, + required=True, + help="Text file: one git URL per line.", + ) + parser.add_argument( + "--mirror-dir", + type=Path, + default=Path("/tmp/kernelci-git-mirror-measure"), + help="Persistent bare repo (created if missing).", + ) + parser.add_argument("--fetch-timeout", type=int, default=DEFAULT_FETCH_TIMEOUT) + parser.add_argument( + "--skip-unfilterable", + action="store_true", + help="Skip servers that do not advertise protocol-v2 filter (job flag).", + ) + parser.add_argument( + "--all-heads", + action="store_true", + help="Fetch every remote branch instead of HEAD only (upper bound).", + ) + args = parser.parse_args() + + urls = _read_urls(args.urls) + if not urls: + print("no urls in", args.urls, file=sys.stderr) + return 1 + + repo = args.mirror_dir + _ensure_bare(repo) + print(f"mirror={repo} urls={len(urls)} start={_human(_du(repo))}", flush=True) + + ok = failed = skipped = 0 + for index, url in enumerate(urls, start=1): + before = _du(repo) + t0 = time.monotonic() + print(f"\n[{index}/{len(urls)}] {url}", flush=True) + status = _fetch_one( + repo, + url, + timeout=args.fetch_timeout, + skip_unfilterable=args.skip_unfilterable, + all_heads=args.all_heads, + ) + elapsed = time.monotonic() - t0 + after = _du(repo) + print( + f" {status} +{_human(after - before)} total={_human(after)} {elapsed:.0f}s", + flush=True, + ) + if status == "ok": + ok += 1 + elif status == "skipped": + skipped += 1 + else: + failed += 1 + + print( + f"\ndone ok={ok} failed={failed} skipped={skipped} total={_human(_du(repo))}", + flush=True, + ) + return 0 if failed == 0 else 1 + + +def _read_urls(path: Path) -> list[str]: + urls: list[str] = [] + seen: set[str] = set() + for raw in path.read_text().splitlines(): + line = raw.strip() + if not line or line.startswith("#"): + continue + url = line.rstrip("/") + if url in seen: + continue + seen.add(url) + urls.append(url) + return urls + + +def _ensure_bare(repo: Path) -> None: + repo.mkdir(parents=True, exist_ok=True) + if not (repo / "HEAD").exists(): + _git(repo, "init", "--bare") + # As the shared mirror grows, negotiation POSTs carry more `have` lines and + # git.kernel.org's frontend answers HTTP 400. Unchunked HTTP/1.1 gets through. + _git(repo, "config", "http.version", "HTTP/1.1") + _git(repo, "config", "http.postBuffer", str(512 * 1024 * 1024)) + # After that 400, git sits on the dead connection instead of exiting. Give curl + # a stall deadline, and keep auto-gc from repacking a multi-GB mirror mid-run. + _git(repo, "config", "http.lowSpeedLimit", "1000") + _git(repo, "config", "http.lowSpeedTime", "60") + _git(repo, "config", "gc.auto", "0") + + +def _fetch_one( + repo: Path, + url: str, + *, + timeout: int, + skip_unfilterable: bool, + all_heads: bool, +) -> str: + name = "r" + hashlib.sha256(url.encode()).hexdigest()[:16] + try: + _ensure_remote(repo, name, url) + heads, supports_filter = _probe(repo, name) + _configure_promisor(repo, name, enabled=supports_filter) + except (subprocess.CalledProcessError, subprocess.TimeoutExpired) as exc: + print(f" probe failed: {exc}", flush=True) + return "failed" + + if not supports_filter: + print(" server cannot filter; fetching with trees/blobs", flush=True) + if skip_unfilterable: + return "skipped" + else: + print(" filter=tree:0", flush=True) + + if all_heads and heads: + refspecs = [ + f"+refs/heads/{branch}:refs/remotes/{name}/{branch}" + for branch in sorted(heads) + ] + print(f" refs: {len(refspecs)} heads", flush=True) + else: + refspecs = [f"+HEAD:refs/remotes/{name}/HEAD"] + print(" refs: HEAD only", flush=True) + + cmd = [ + "fetch", + "--prune", + "--no-tags", + *(["--filter=tree:0"] if supports_filter else []), + "--progress", + name, + *refspecs, + ] + try: + _git(repo, *cmd, timeout=timeout, stream=True) + except (subprocess.CalledProcessError, subprocess.TimeoutExpired) as exc: + print(f" fetch failed: {exc}", flush=True) + return "failed" + return "ok" + + +def _ensure_remote(repo: Path, name: str, url: str) -> None: + existing = _remote_urls(repo) + if name not in existing: + _git(repo, "remote", "add", name, url) + elif existing[name] != url: + _git(repo, "remote", "set-url", name, url) + + +def _configure_promisor(repo: Path, name: str, *, enabled: bool) -> None: + for key, value in (("promisor", "true"), ("partialclonefilter", "tree:0")): + if enabled: + _git(repo, "config", f"remote.{name}.{key}", value) + else: + try: + _git(repo, "config", "--unset", f"remote.{name}.{key}") + except subprocess.CalledProcessError: + pass + + +def _remote_urls(repo: Path) -> dict[str, str]: + try: + output = _git(repo, "remote", "-v").decode() + except subprocess.CalledProcessError: + return {} + urls: dict[str, str] = {} + for line in output.splitlines(): + fields = line.split() + if len(fields) >= 2: + urls[fields[0]] = fields[1] + return urls + + +def _probe(repo: Path, name: str) -> tuple[set[str], bool]: + with tempfile.TemporaryDirectory() as scratch: + trace = Path(scratch) / "packet-trace" + output = _git( + repo, + "-c", + "protocol.version=2", + "ls-remote", + "--heads", + name, + timeout=LS_REMOTE_TIMEOUT, + extra_env={"GIT_TRACE_PACKET": str(trace)}, + ).decode() + supports_filter = _trace_advertises_filter(trace) + + heads: set[str] = set() + for line in output.splitlines(): + parts = line.split() + if len(parts) >= 2 and parts[1].startswith("refs/heads/"): + heads.add(parts[1].removeprefix("refs/heads/")) + return heads, supports_filter + + +def _trace_advertises_filter(trace: Path) -> bool: + if not trace.is_file(): + return False + for line in trace.read_text(errors="replace").splitlines(): + _, sep, capability = line.partition("fetch=") + if sep and "filter" in capability.split(): + return True + return False + + +def _git( + repo: Path, + *args: str, + timeout: int = 60, + stream: bool = False, + extra_env: dict[str, str] | None = None, +) -> bytes: + env = { + **os.environ, + "GIT_TERMINAL_PROMPT": "0", + "GIT_OPTIONAL_LOCKS": "0", + "LC_ALL": "C", + } + if extra_env: + env.update(extra_env) + cmd = ["git", "-C", str(repo), *args] + # Own session so a timeout kills git *and* its curl/remote-https children. + # subprocess.run only signals the direct child, which is how one stuck fetch + # held the whole run. + proc = subprocess.Popen( # noqa: S603 + cmd, + env=env, + start_new_session=True, + stdout=None if stream else subprocess.PIPE, + stderr=None if stream else subprocess.PIPE, + ) + try: + stdout, _ = proc.communicate(timeout=timeout) + except subprocess.TimeoutExpired: + os.killpg(proc.pid, signal.SIGKILL) + proc.communicate() + raise + if proc.returncode != 0: + raise subprocess.CalledProcessError(proc.returncode, cmd) + return stdout or b"" + + +def _du(repo: Path) -> int: + if not repo.is_dir(): + return 0 + return sum(p.stat().st_size for p in repo.rglob("*") if p.is_file()) + + +def _human(n: int) -> str: + size = float(max(n, 0)) + for unit in ("B", "KiB", "MiB", "GiB", "TiB"): + if size < 1024 or unit == "TiB": + return f"{size:.1f}{unit}" + size /= 1024 + return f"{n}B" + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docker-compose-next.yml b/docker-compose-next.yml index 29c762000..38b359f03 100644 --- a/docker-compose-next.yml +++ b/docker-compose-next.yml @@ -10,6 +10,7 @@ volumes: backend-data: static-data: dashboard-db-data: + git-mirror: networks: public: @@ -62,6 +63,7 @@ services: - DASHBOARD_VERSION=${DASHBOARD_VERSION:-unknown} volumes: - backend-data:${BACKEND_VOLUME_DIR:-/volume_data} + - git-mirror:${GIT_MIRROR_DIR:-/var/lib/kernelci/git-mirror} restart: always networks: - private diff --git a/docker-compose.dev.yml b/docker-compose.dev.yml index 673c140ef..51e8b31d4 100644 --- a/docker-compose.dev.yml +++ b/docker-compose.dev.yml @@ -1,6 +1,7 @@ volumes: backend-data: dashboard-db-data: + git-mirror: networks: public: @@ -12,6 +13,7 @@ services: context: ./backend volumes: - backend-data:${BACKEND_VOLUME_DIR:-/volume_data} + - git-mirror:${GIT_MIRROR_DIR:-/var/lib/kernelci/git-mirror} - ./backend:/backend # live reload: source mounted over image copy env_file: [.env] environment: diff --git a/docker-compose.yml b/docker-compose.yml index fad0e3026..e7b477ea8 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -3,6 +3,7 @@ volumes: runtime-data: static-data: dashboard-db-data: + git-mirror: networks: public: @@ -15,6 +16,7 @@ services: context: ./backend volumes: - backend-data:${BACKEND_VOLUME_DIR:-/volume_data} + - git-mirror:${GIT_MIRROR_DIR:-/var/lib/kernelci/git-mirror} env_file: - .env environment: diff --git a/docs/sync-commits.md b/docs/sync-commits.md new file mode 100644 index 000000000..69467598c --- /dev/null +++ b/docs/sync-commits.md @@ -0,0 +1,245 @@ +# Commit metadata sync + +Two management commands fill `commits` and `commit_parents` from git, not from +KCIDB submissions. Author, committer, subject, message, and ordered parents +are properties of a git object. The same hash can appear on many checkouts, so +that data does not live on `checkouts`. + +Parent epic: [#2079](https://github.com/kernelci/dashboard/issues/2079). +Schema: [#2089](https://github.com/kernelci/dashboard/issues/2089). +Parse / one-shot SHA fetch: [#2090](https://github.com/kernelci/dashboard/issues/2090). +This job: [#2109](https://github.com/kernelci/dashboard/issues/2109). + +Entry points: + +- Mirror: `backend/kernelCI_app/management/commands/sync_commit_mirror.py` +- Ingest: `backend/kernelCI_app/management/commands/sync_commit_ingest.py` +- Logic: `backend/kernelCI_app/helpers/commitSync.py` +- Parse / SHA fetch: `backend/kernelCI_app/helpers/gitCommit.py` + +Postgres is the query store. Git is only an ingest cache. Request handlers +must not call git. + +## Why a mirror, not checkout hashes + +Filling only `DISTINCT checkouts.git_commit_hash` cannot walk first-parent +history when CI skipped a commit. The job therefore fetches the **ancestor +closure** of the branches we already see in checkouts, from an allowlist of +tree URLs (`tree-names.yaml`), into one persistent bare repo. + +`--fill-gaps` is the fallback for hashes the mirror still does not cover. It +uses the one-shot SHA helper from #2090 (no parents). + +The job never writes `checkouts.git_commit_message`. + +## Tables + +`commits`: surrogate `id`, unique `git_commit_hash`, author / committer / +subject / message. + +`commit_parents`: `commit_id`, `parent_id`, `ord` (`0` = first parent). No +stub rows. Inserts are topological so the parent row exists before the edge. + +Join from `checkouts.git_commit_hash` for now. `commit_id` FKs on checkouts +are a later change. + +## Storage + +One **bare** repo at `GIT_MIRROR_DIR` (default `/var/lib/kernelci/git-mirror`). +Docker volume `git-mirror`, **not** `BACKEND_VOLUME_DIR`. + +All allowlisted URLs are remotes of that one repo. Git shares objects across +remotes, so the first kernel tree is expensive and the rest mostly reuse it. + +| Kind of fetch | Typical first pack | +|---|---| +| Server with `tree:0` (googlesource, gitlab, github) | hundreds of MB of commits | +| Server without filter (`git.kernel.org`) | ~3.5 GB (trees + blobs) | +| Later trees on the same object store | tens of MB of new objects | + +Steady-state estimate for the current allowlist (~35 trees): **about 6–8 GB**, +then daily deltas in the tens of MB. Cloning each tree into a throwaway repo +and deleting it would cost that 3.5 GB **per tree per run** (no object +sharing, no incremental fetch). + +Throwaway clones for `--fill-gaps` go under `GIT_SCRATCH_DIR` (prefer tmpfs). + +Tips seen at the end of a successful **ingest** are stored in `synced-tips` next +to the mirror. The next ingest does `rev-list --remotes` with each stored tip as a `^sha` on +stdin (git's `--not --stdin` does not exclude stdin lines). +Fetch-only does not touch that file. + +### Recreate the volume + +A pack that slipped in unfiltered stays until the volume is wiped. + +```bash +docker compose -f docker-compose.dev.yml stop backend +docker compose -f docker-compose.dev.yml rm -f backend +docker volume rm dashboard_git-mirror +docker compose -f docker-compose.dev.yml up -d backend +``` + +## Job flow + +Two cron entries on the backend container: + +- `0 4 * * *` `sync_commit_mirror` — remotes into the mirror. Does not write + `commits` or `synced-tips`. +- `0 10 * * *` `sync_commit_ingest` — ingest new mirror objects (tip delta), + then upsert. Six hours later so a long first fetch is less likely to + overlap; do not run both against the same `GIT_MIRROR_DIR` at once. + +``` +tree-names.yaml URLs + | + v +sync_commit_mirror +for each remote (sequential; git locks one repo) + | + +-- ls-remote: branch list + protocol v2 fetch= capabilities + | + +-- filter advertised? git fetch --filter=tree:0 + | otherwise git fetch (no filter) + | or --skip-unfilterable → skip this remote + | + +-- pack too big / incomplete / (filter promised but trees leaked) + | → drop new packs + restore refs; skip this remote this run + | (no HEAD retry: same server, same behaviour) + v +sync_commit_ingest +rev-list new tips --not old tips (topo, reverse) + | + v +cat-file --batch in 2000-hash chunks → parse → upsert commits then edges + | + v +write synced-tips + | + v +optional --fill-gaps (one-shot SHA fetch per missing checkout hash) +``` + +`--dry-run` still fetches/parses if asked, but writes nothing to the database +or `synced-tips`, and does not regenerate `tree-names.yaml`. + +If `tree-names.yaml` is missing, a non-dry run regenerates it the same way +the ingester does (`treeproof`). Empty allowlist used to make the job a +silent no-op. + +## What is fetched + +Not `git fetch ` (every topic branch since 2015). + +For each URL, the job intersects `checkouts.git_repository_branch` with +`ls-remote --heads`. If nothing matches, it fetches `HEAD` only. + +That is enough for first-parent walks between checkouts. Untested topic +branches stay out of the mirror on purpose. + +## Filter vs full clone + +`--filter=tree:0` is a **server** feature. The client cannot force it. + +Protocol v2 capability line from a real probe: + +- `android.googlesource.com`: `fetch=filter ref-in-want ...` → treeless works +- `git.kernel.org`: `fetch=shallow wait-for-done` → **no filter**; a + `--filter=tree:0` fetch is silently a full clone + +There is no porcelain for that advertisement. The job points +`GIT_TRACE_PACKET` at a temp file during `ls-remote` and looks for `filter` +on the `fetch=` line. + +Default: still fetch unfilterable remotes. Object sharing makes the second +kernel.org tree cheap. `--skip-unfilterable` opts out (smaller disk, those +trees only covered by `--fill-gaps`). + +We do **not** rewrite `git.kernel.org` URLs to `kernel.googlesource.com`. +The googlesource host is a valid mirror, but mapping URLs would split +identity from the treeproof allowlist and from `checkouts.git_repository_url`. + +## Pack rejection + +After each fetch, only **new** pack files are inspected. + +Always reject: + +- leftover `tmp_pack_*` +- pack larger than `MAX_REMOTE_PACK_BYTES` (6 GiB; runaway backstop) + +If the server advertised filter, also reject packs that contain `tree` or +`blob` (broken promise). If it did not, trees/blobs are expected. + +On reject: restore `refs/remotes//*` to the pre-fetch snapshot and +delete the new pack files. That remote is skipped **this run** only. + +Do not retry `HEAD` after a rejected pack. The second download is the same +full clone. + +## Parse and upsert + +`git cat-file --batch` over stdin, 2000 hashes at a time. Per-commit +`rev-parse` + `cat-file` is hundreds of times slower. + +Each chunk is upserted before the next is parsed so a full-history import +does not hold every commit message in RAM. Order stays topological: +`rev-list --reverse --topo-order`, then chunks in that order. + +`bulk_create(..., ignore_conflicts=True)`. Existing hashes are left alone. +Missing parent rows skip the edge (logged); no stubs. + +## Parallel fetch + +Not implemented. `git fetch` takes a lock on a single repo, so threads +against `GIT_MIRROR_DIR` serialize or fail. + +The useful pattern would be N throwaway bares in parallel, then a serial +`git fetch` into the persistent mirror. That is extra disk and extra +pressure on git.kernel.org. Measure incremental runs first: after the first +populate, most remotes send almost nothing. + +## Commands + +```bash +poetry run python manage.py sync_commit_mirror +poetry run python manage.py sync_commit_mirror --dry-run --verbose-git +poetry run python manage.py sync_commit_mirror --skip-unfilterable +poetry run python manage.py sync_commit_mirror --mirror-dir /path --fetch-timeout 1800 + +poetry run python manage.py sync_commit_ingest +poetry run python manage.py sync_commit_ingest --dry-run +poetry run python manage.py sync_commit_ingest --fill-gaps +poetry run python manage.py sync_commit_ingest --mirror-dir /path +``` + +`sync_commit_mirror`: + +| Flag | Effect | +|---|---| +| `--dry-run` | Fetch, do not regenerate `tree-names.yaml` | +| `--skip-unfilterable` | Do not fetch servers without `fetch=filter` | +| `--verbose-git` | Git's own fetch progress on stdout (noisy in cron) | +| `--mirror-dir` | Override `GIT_MIRROR_DIR` | +| `--fetch-timeout` | Per-remote fetch timeout (default 1800s) | + +`sync_commit_ingest`: + +| Flag | Effect | +|---|---| +| `--dry-run` | No DB writes, no `synced-tips` | +| `--fill-gaps` | One-shot SHA fetch for checkout hashes still missing | +| `--mirror-dir` | Override `GIT_MIRROR_DIR` | + +Env: + +- `GIT_MIRROR_DIR` — persistent bare repo +- `GIT_SCRATCH_DIR` — ephemeral clones for `--fill-gaps` +- `BACKEND_VOLUME_DIR` — `tree-names.yaml` only + +## Out of scope (follow-ups) + +- `trees` / `named_refs` tables +- Rewriting readers of `checkouts.git_commit_message` / dropping that column +- First-parent “next checkout” queries ([#2092](https://github.com/kernelci/dashboard/issues/2092)) +- Fetching git from request handlers From e4c426a7bf90d03af16fb43aad1da62628446ce7 Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 25 Sep 2026 16:59:46 -0300 Subject: [PATCH 07/11] fix(backend): stop calling ignore-on-conflict an upsert Commit objects are immutable, so existing hashes stay as stored. Dry-run help now says the mirror fetch still updates the repo on disk, while ingest does not fetch. --- backend/kernelCI_app/helpers/commitSync.py | 26 +++++++++++-------- .../management/commands/sync_commit_ingest.py | 11 +++++--- .../management/commands/sync_commit_mirror.py | 6 ++++- .../unitTests/helpers/commitSync_test.py | 6 ++--- docs/sync-commits.md | 25 +++++++++++------- 5 files changed, 46 insertions(+), 28 deletions(-) diff --git a/backend/kernelCI_app/helpers/commitSync.py b/backend/kernelCI_app/helpers/commitSync.py index 2bdedca09..691f6bddc 100644 --- a/backend/kernelCI_app/helpers/commitSync.py +++ b/backend/kernelCI_app/helpers/commitSync.py @@ -34,7 +34,7 @@ logger = logging.getLogger(__name__) TIPS_FILENAME = "synced-tips" -UPSERT_BATCH_SIZE = 1000 +INSERT_BATCH_SIZE = 1000 DEFAULT_FETCH_TIMEOUT_SECONDS = 1800 REV_LIST_TIMEOUT_SECONDS = 3600 # One `cat-file --batch` process per chunk instead of two per commit. Chunked so a @@ -266,13 +266,17 @@ def parse_commits(repo_dir: Path, hashes: Sequence[str]) -> list[CommitMetadata] ] -def upsert_commits(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: - """Insert commits then parent edges. No stubs. Callers pass topo order.""" +def insert_commits(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: + """Insert commits then parent edges. Existing rows are left alone. + + A git object is immutable, so a hash that is already stored is not updated. + No stubs. Callers pass topo order. + """ commit_count = 0 edge_count = 0 - for start in range(0, len(metadatas), UPSERT_BATCH_SIZE): - batch = metadatas[start : start + UPSERT_BATCH_SIZE] - commits, edges = _upsert_batch(batch) + for start in range(0, len(metadatas), INSERT_BATCH_SIZE): + batch = metadatas[start : start + INSERT_BATCH_SIZE] + commits, edges = _insert_batch(batch) commit_count += commits edge_count += edges return commit_count, edge_count @@ -304,7 +308,7 @@ def fill_checkout_gaps(*, dry_run: bool = False) -> tuple[int, int]: continue fetched += 1 if not dry_run: - commits, _edges = upsert_commits([metadata]) + commits, _edges = insert_commits([metadata]) written += commits if index % 100 == 0 or index == len(gaps): out(f"gaps {index}/{len(gaps)}: {fetched} fetched, {written} written") @@ -375,7 +379,7 @@ def _ingest_new_commits( for batch in iter_parsed_commits(repo_dir, hashes): parsed += len(batch) if not dry_run: - commits, edges = upsert_commits(batch) + commits, edges = insert_commits(batch) commits_written += commits edges_written += edges if parsed - logged_at >= PROGRESS_LOG_EVERY or parsed == len(hashes): @@ -758,7 +762,7 @@ def _remote_urls(repo_dir: Path) -> dict[str, str]: return urls -def _upsert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: +def _insert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: rows = [ Commits( git_commit_hash=metadata.git_commit_hash, @@ -774,7 +778,7 @@ def _upsert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: for metadata in metadatas ] Commits.objects.bulk_create( - rows, ignore_conflicts=True, batch_size=UPSERT_BATCH_SIZE + rows, ignore_conflicts=True, batch_size=INSERT_BATCH_SIZE ) hashes = {metadata.git_commit_hash for metadata in metadatas} @@ -806,6 +810,6 @@ def _upsert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: if edges: CommitParents.objects.bulk_create( - edges, ignore_conflicts=True, batch_size=UPSERT_BATCH_SIZE + edges, ignore_conflicts=True, batch_size=INSERT_BATCH_SIZE ) return len(metadatas), len(edges) diff --git a/backend/kernelCI_app/management/commands/sync_commit_ingest.py b/backend/kernelCI_app/management/commands/sync_commit_ingest.py index b1119bcbf..03cb8a2ae 100644 --- a/backend/kernelCI_app/management/commands/sync_commit_ingest.py +++ b/backend/kernelCI_app/management/commands/sync_commit_ingest.py @@ -9,15 +9,20 @@ class Command(BaseCommand): help = ( - "Upsert commits / commit_parents from objects already in the persistent " - "mirror (tip delta). Does not write checkouts.git_commit_message." + "Insert commits / commit_parents from objects already in the persistent " + "mirror (tip delta). Existing hashes are left alone. Does not write " + "checkouts.git_commit_message." ) def add_arguments(self, parser): parser.add_argument( "--dry-run", action="store_true", - help="Parse as requested, write nothing to the database or tips file.", + help=( + "Parse objects already in the mirror. Writes nothing to the " + "database or synced-tips, and does not fetch. --fill-gaps still " + "fetches missing SHAs and then discards them." + ), ) parser.add_argument( "--fill-gaps", diff --git a/backend/kernelCI_app/management/commands/sync_commit_mirror.py b/backend/kernelCI_app/management/commands/sync_commit_mirror.py index 1f4306c50..99ddbfb22 100644 --- a/backend/kernelCI_app/management/commands/sync_commit_mirror.py +++ b/backend/kernelCI_app/management/commands/sync_commit_mirror.py @@ -20,7 +20,11 @@ def add_arguments(self, parser): parser.add_argument( "--dry-run", action="store_true", - help="Fetch as requested, do not regenerate tree-names.yaml.", + help=( + "Still fetches into the persistent mirror, so the git repo on " + "disk is updated. Does not regenerate tree-names.yaml or write " + "the database." + ), ) parser.add_argument( "--mirror-dir", diff --git a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py index d74a0e76f..418c27336 100644 --- a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py +++ b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py @@ -19,7 +19,7 @@ new_commit_hashes, parse_commits, sync_commit_metadata, - upsert_commits, + insert_commits, ) from kernelCI_app.helpers.gitCommit import CommitMetadata @@ -202,7 +202,7 @@ def test_topo_order_first_parent_edges(self, commit_store): skipped = _metadata("bb" * 20, "aa" * 20, subject="skipped") tip = _metadata("cc" * 20, "bb" * 20, subject="tip") - commit_count, edge_count = upsert_commits([root, skipped, tip]) + commit_count, edge_count = insert_commits([root, skipped, tip]) assert commit_count == 3 assert edge_count == 2 @@ -217,7 +217,7 @@ def test_skips_edge_when_parent_row_missing(self, commit_store): _commits_by_hash, parent_rows = commit_store child = _metadata("dd" * 20, "ee" * 20, subject="orphan-parent") - commit_count, edge_count = upsert_commits([child]) + commit_count, edge_count = insert_commits([child]) assert commit_count == 1 assert edge_count == 0 diff --git a/docs/sync-commits.md b/docs/sync-commits.md index 69467598c..08c2c3dad 100644 --- a/docs/sync-commits.md +++ b/docs/sync-commits.md @@ -87,7 +87,7 @@ Two cron entries on the backend container: - `0 4 * * *` `sync_commit_mirror` — remotes into the mirror. Does not write `commits` or `synced-tips`. - `0 10 * * *` `sync_commit_ingest` — ingest new mirror objects (tip delta), - then upsert. Six hours later so a long first fetch is less likely to + then insert. Six hours later so a long first fetch is less likely to overlap; do not run both against the same `GIT_MIRROR_DIR` at once. ``` @@ -111,7 +111,7 @@ sync_commit_ingest rev-list new tips --not old tips (topo, reverse) | v -cat-file --batch in 2000-hash chunks → parse → upsert commits then edges +cat-file --batch in 2000-hash chunks → parse → insert commits then edges | v write synced-tips @@ -120,8 +120,12 @@ write synced-tips optional --fill-gaps (one-shot SHA fetch per missing checkout hash) ``` -`--dry-run` still fetches/parses if asked, but writes nothing to the database -or `synced-tips`, and does not regenerate `tree-names.yaml`. +`sync_commit_mirror --dry-run` still fetches, so the mirror on disk changes. +It does not regenerate `tree-names.yaml` or write the database. + +`sync_commit_ingest --dry-run` parses objects already in the mirror and writes +nothing to the database or `synced-tips`. It does not fetch. `--fill-gaps` +still fetches missing checkout SHAs and then discards them. If `tree-names.yaml` is missing, a non-dry run regenerates it the same way the ingester does (`treeproof`). Empty allowlist used to make the job a @@ -177,17 +181,18 @@ delete the new pack files. That remote is skipped **this run** only. Do not retry `HEAD` after a rejected pack. The second download is the same full clone. -## Parse and upsert +## Parse and insert `git cat-file --batch` over stdin, 2000 hashes at a time. Per-commit `rev-parse` + `cat-file` is hundreds of times slower. -Each chunk is upserted before the next is parsed so a full-history import +Each chunk is inserted before the next is parsed so a full-history import does not hold every commit message in RAM. Order stays topological: `rev-list --reverse --topo-order`, then chunks in that order. -`bulk_create(..., ignore_conflicts=True)`. Existing hashes are left alone. -Missing parent rows skip the edge (logged); no stubs. +`bulk_create(..., ignore_conflicts=True)`. A git object is immutable, so a +hash that is already stored is left alone rather than updated. Missing parent +rows skip the edge (logged); no stubs. ## Parallel fetch @@ -217,7 +222,7 @@ poetry run python manage.py sync_commit_ingest --mirror-dir /path | Flag | Effect | |---|---| -| `--dry-run` | Fetch, do not regenerate `tree-names.yaml` | +| `--dry-run` | Still fetches into the mirror. Does not regenerate `tree-names.yaml` | | `--skip-unfilterable` | Do not fetch servers without `fetch=filter` | | `--verbose-git` | Git's own fetch progress on stdout (noisy in cron) | | `--mirror-dir` | Override `GIT_MIRROR_DIR` | @@ -227,7 +232,7 @@ poetry run python manage.py sync_commit_ingest --mirror-dir /path | Flag | Effect | |---|---| -| `--dry-run` | No DB writes, no `synced-tips` | +| `--dry-run` | No fetch, no DB writes, no `synced-tips`. `--fill-gaps` still fetches | | `--fill-gaps` | One-shot SHA fetch for checkout hashes still missing | | `--mirror-dir` | Override `GIT_MIRROR_DIR` | From 6fe41d3fccf6928caee1bf8bb93740d991768070 Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 25 Sep 2026 17:05:21 -0300 Subject: [PATCH 08/11] fix(backend): sort commit sync test imports --- backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py index 418c27336..3589b0b8e 100644 --- a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py +++ b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py @@ -15,11 +15,11 @@ allowlisted_tree_urls, ensure_mirror, fetch_remote, + insert_commits, list_tips, new_commit_hashes, parse_commits, sync_commit_metadata, - insert_commits, ) from kernelCI_app.helpers.gitCommit import CommitMetadata From 07d64ab4a7dfb058e75e9867e988dfc3718e6756 Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 25 Sep 2026 18:00:31 -0300 Subject: [PATCH 09/11] fix(backend): write synced commits onto the split schema Author and committer now live on commit_identity, and subject and message on commit_message. --- backend/kernelCI_app/helpers/commitSync.py | 58 ++++++++++++++++--- .../unitTests/helpers/commitSync_test.py | 39 +++++++++++-- docs/sync-commits.md | 17 ++++-- 3 files changed, 95 insertions(+), 19 deletions(-) diff --git a/backend/kernelCI_app/helpers/commitSync.py b/backend/kernelCI_app/helpers/commitSync.py index 691f6bddc..c3821bf21 100644 --- a/backend/kernelCI_app/helpers/commitSync.py +++ b/backend/kernelCI_app/helpers/commitSync.py @@ -17,6 +17,7 @@ from django.conf import settings from kernelCI_app.constants.tree_names import TREE_NAMES_FILENAME +from kernelCI_app.helpers.commit_message_compression import message_to_bytea from kernelCI_app.helpers.gitCommit import ( CommitMetadata, CommitMetadataError, @@ -29,7 +30,13 @@ from kernelCI_app.helpers.logger import out from kernelCI_app.helpers.trees import get_tree_file_data from kernelCI_app.management.commands.treeproof import Command as TreeproofCommand -from kernelCI_app.models import Checkouts, CommitParents, Commits +from kernelCI_app.models import ( + Checkouts, + CommitIdentity, + CommitMessage, + CommitParents, + Commits, +) logger = logging.getLogger(__name__) @@ -266,17 +273,35 @@ def parse_commits(repo_dir: Path, hashes: Sequence[str]) -> list[CommitMetadata] ] +def _identity_id( + email: str | None, + name: str | None, + cache: dict[tuple[str, str], int], +) -> int: + key = (email or "", name or "") + identity_id = cache.get(key) + if identity_id is not None: + return identity_id + identity, _created = CommitIdentity.objects.get_or_create( + email=key[0], + name=key[1], + ) + cache[key] = identity.id + return identity.id + + def insert_commits(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: """Insert commits then parent edges. Existing rows are left alone. A git object is immutable, so a hash that is already stored is not updated. No stubs. Callers pass topo order. """ + identity_ids: dict[tuple[str, str], int] = {} commit_count = 0 edge_count = 0 for start in range(0, len(metadatas), INSERT_BATCH_SIZE): batch = metadatas[start : start + INSERT_BATCH_SIZE] - commits, edges = _insert_batch(batch) + commits, edges = _insert_batch(batch, identity_ids) commit_count += commits edge_count += edges return commit_count, edge_count @@ -762,18 +787,21 @@ def _remote_urls(repo_dir: Path) -> dict[str, str]: return urls -def _insert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: +def _insert_batch( + metadatas: Sequence[CommitMetadata], + identity_ids: dict[tuple[str, str], int], +) -> tuple[int, int]: rows = [ Commits( git_commit_hash=metadata.git_commit_hash, - author_name=metadata.author_name, - author_email=metadata.author_email, + author_identity_id=_identity_id( + metadata.author_email, metadata.author_name, identity_ids + ), author_date=metadata.author_date, - committer_name=metadata.committer_name, - committer_email=metadata.committer_email, + committer_identity_id=_identity_id( + metadata.committer_email, metadata.committer_name, identity_ids + ), committer_date=metadata.committer_date, - subject=metadata.subject, - message=metadata.message, ) for metadata in metadatas ] @@ -791,10 +819,18 @@ def _insert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: ) edges: list[CommitParents] = [] + messages: list[CommitMessage] = [] for metadata in metadatas: commit_id = hash_to_id.get(metadata.git_commit_hash) if commit_id is None: continue + messages.append( + CommitMessage( + commit_id=commit_id, + subject=metadata.subject, + message=message_to_bytea(metadata.message), + ) + ) for ord_, parent_hash in enumerate(metadata.parent_hashes): parent_id = hash_to_id.get(parent_hash) if parent_id is None: @@ -808,6 +844,10 @@ def _insert_batch(metadatas: Sequence[CommitMetadata]) -> tuple[int, int]: CommitParents(commit_id=commit_id, parent_id=parent_id, ord=ord_) ) + if messages: + CommitMessage.objects.bulk_create( + messages, ignore_conflicts=True, batch_size=INSERT_BATCH_SIZE + ) if edges: CommitParents.objects.bulk_create( edges, ignore_conflicts=True, batch_size=INSERT_BATCH_SIZE diff --git a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py index 3589b0b8e..646fbf85b 100644 --- a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py +++ b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py @@ -3,6 +3,7 @@ import subprocess from datetime import datetime, timezone from pathlib import Path +from types import SimpleNamespace import pytest from django.core.management import call_command @@ -86,7 +87,10 @@ def _metadata( def commit_store(monkeypatch): commits_by_hash: dict[str, object] = {} parent_rows: list[object] = [] + message_rows: list[object] = [] + identities: dict[tuple[str, str], SimpleNamespace] = {} next_id = {"n": 1} + next_identity_id = {"n": 1} class _Filter: def __init__(self, hashes: set[str]): @@ -120,11 +124,33 @@ def bulk_create(self, rows, **_kwargs): parent_rows.extend(rows) return rows + class _Identities: + def get_or_create(self, *, email: str, name: str): + key = (email, name) + row = identities.get(key) + if row is not None: + return row, False + row = SimpleNamespace(id=next_identity_id["n"], email=email, name=name) + next_identity_id["n"] += 1 + identities[key] = row + return row, True + + class _Messages: + def bulk_create(self, rows, **_kwargs): + message_rows.extend(rows) + return rows + monkeypatch.setattr("kernelCI_app.helpers.commitSync.Commits.objects", _Commits()) monkeypatch.setattr( "kernelCI_app.helpers.commitSync.CommitParents.objects", _Parents() ) - return commits_by_hash, parent_rows + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.CommitIdentity.objects", _Identities() + ) + monkeypatch.setattr( + "kernelCI_app.helpers.commitSync.CommitMessage.objects", _Messages() + ) + return commits_by_hash, parent_rows, message_rows _TREES_FILE = { @@ -197,7 +223,7 @@ def test_payload_with_newlines_does_not_desync_records(self): class TestUpsertCommits: def test_topo_order_first_parent_edges(self, commit_store): - commits_by_hash, parent_rows = commit_store + commits_by_hash, parent_rows, message_rows = commit_store root = _metadata("aa" * 20, subject="root") skipped = _metadata("bb" * 20, "aa" * 20, subject="skipped") tip = _metadata("cc" * 20, "bb" * 20, subject="tip") @@ -212,9 +238,14 @@ def test_topo_order_first_parent_edges(self, commit_store): (commits_by_hash["bb" * 20].id, commits_by_hash["aa" * 20].id, 0), (commits_by_hash["cc" * 20].id, commits_by_hash["bb" * 20].id, 0), } + assert len({row.author_identity_id for row in commits_by_hash.values()}) == 1 + root_id = commits_by_hash["aa" * 20].id + stored = {row.commit_id: row for row in message_rows} + assert stored[root_id].subject == "root" + assert stored[root_id].message == b"root\n" def test_skips_edge_when_parent_row_missing(self, commit_store): - _commits_by_hash, parent_rows = commit_store + _commits_by_hash, parent_rows, _message_rows = commit_store child = _metadata("dd" * 20, "ee" * 20, subject="orphan-parent") commit_count, edge_count = insert_commits([child]) @@ -241,7 +272,7 @@ def test_skipped_intermediate_commit_is_ingested( ) stats = sync_commit_metadata(mirror_dir=mirror) - commits_by_hash, parent_rows = commit_store + commits_by_hash, parent_rows, _message_rows = commit_store assert stats["remotes_ok"] == 1 assert stats["remotes_failed"] == 0 diff --git a/docs/sync-commits.md b/docs/sync-commits.md index 08c2c3dad..1d946320e 100644 --- a/docs/sync-commits.md +++ b/docs/sync-commits.md @@ -1,9 +1,9 @@ # Commit metadata sync -Two management commands fill `commits` and `commit_parents` from git, not from -KCIDB submissions. Author, committer, subject, message, and ordered parents -are properties of a git object. The same hash can appear on many checkouts, so -that data does not live on `checkouts`. +Two management commands fill `commits`, `commit_identity`, `commit_message`, +and `commit_parents` from git, not from KCIDB submissions. Author, committer, +subject, message, and ordered parents are properties of a git object. The same +hash can appear on many checkouts, so that data does not live on `checkouts`. Parent epic: [#2079](https://github.com/kernelci/dashboard/issues/2079). Schema: [#2089](https://github.com/kernelci/dashboard/issues/2089). @@ -34,8 +34,13 @@ The job never writes `checkouts.git_commit_message`. ## Tables -`commits`: surrogate `id`, unique `git_commit_hash`, author / committer / -subject / message. +`commit_identity`: unique `(email, name)`. Missing name or email is stored as +`""`. + +`commits`: surrogate `id`, unique `git_commit_hash`, `author_identity_id`, +`committer_identity_id`, author / committer dates, optional `fetched_from_url`. + +`commit_message`: `commit_id` primary key, `subject`, `message` as UTF-8 bytes. `commit_parents`: `commit_id`, `parent_id`, `ord` (`0` = first parent). No stub rows. Inserts are topological so the parent row exists before the edge. From 992b1fa5da549cd7092e1301d2cfa2b43cdce8b6 Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Fri, 25 Sep 2026 18:10:59 -0300 Subject: [PATCH 10/11] fix(backend): move commit sync fakes out of the fixture Nested identity and message stores pushed commit_store over the complexity limit. --- .../unitTests/helpers/commitSync_test.py | 48 +++++++++++-------- 1 file changed, 28 insertions(+), 20 deletions(-) diff --git a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py index 646fbf85b..1ea532d72 100644 --- a/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py +++ b/backend/kernelCI_app/tests/unitTests/helpers/commitSync_test.py @@ -83,14 +83,37 @@ def _metadata( ) +class _IdentityStore: + def __init__(self) -> None: + self.rows: dict[tuple[str, str], SimpleNamespace] = {} + self.next_id = 1 + + def get_or_create(self, *, email: str, name: str): + key = (email, name) + row = self.rows.get(key) + if row is not None: + return row, False + row = SimpleNamespace(id=self.next_id, email=email, name=name) + self.next_id += 1 + self.rows[key] = row + return row, True + + +class _MessageStore: + def __init__(self, rows: list[object]) -> None: + self.rows = rows + + def bulk_create(self, rows, **_kwargs): + self.rows.extend(rows) + return rows + + @pytest.fixture def commit_store(monkeypatch): commits_by_hash: dict[str, object] = {} parent_rows: list[object] = [] message_rows: list[object] = [] - identities: dict[tuple[str, str], SimpleNamespace] = {} next_id = {"n": 1} - next_identity_id = {"n": 1} class _Filter: def __init__(self, hashes: set[str]): @@ -124,31 +147,16 @@ def bulk_create(self, rows, **_kwargs): parent_rows.extend(rows) return rows - class _Identities: - def get_or_create(self, *, email: str, name: str): - key = (email, name) - row = identities.get(key) - if row is not None: - return row, False - row = SimpleNamespace(id=next_identity_id["n"], email=email, name=name) - next_identity_id["n"] += 1 - identities[key] = row - return row, True - - class _Messages: - def bulk_create(self, rows, **_kwargs): - message_rows.extend(rows) - return rows - monkeypatch.setattr("kernelCI_app.helpers.commitSync.Commits.objects", _Commits()) monkeypatch.setattr( "kernelCI_app.helpers.commitSync.CommitParents.objects", _Parents() ) monkeypatch.setattr( - "kernelCI_app.helpers.commitSync.CommitIdentity.objects", _Identities() + "kernelCI_app.helpers.commitSync.CommitIdentity.objects", _IdentityStore() ) monkeypatch.setattr( - "kernelCI_app.helpers.commitSync.CommitMessage.objects", _Messages() + "kernelCI_app.helpers.commitSync.CommitMessage.objects", + _MessageStore(message_rows), ) return commits_by_hash, parent_rows, message_rows From bfa9a4546be48a1bca949de1845934107dca97da Mon Sep 17 00:00:00 2001 From: Felipe Bergamin Date: Tue, 29 Sep 2026 16:58:25 -0300 Subject: [PATCH 11/11] feat(backend): expose git mirror size after each fetch Record the mirror's on-disk size when sync_commit_mirror finishes so Prometheus can alert above 10 GB. Co-authored-by: Cursor Signed-off-by: Felipe Bergamin --- .../management/commands/sync_commit_mirror.py | 42 +++++--- .../unitTests/utils/git_mirror_size_test.py | 62 +++++++++++ backend/utils/git_mirror_size.py | 101 ++++++++++++++++++ backend/utils/prometheus_aggregator.py | 5 + docs/monitoring.md | 14 +++ docs/sync-commits.md | 8 ++ monitoring/django.rules | 9 ++ 7 files changed, 227 insertions(+), 14 deletions(-) create mode 100644 backend/kernelCI_app/tests/unitTests/utils/git_mirror_size_test.py create mode 100644 backend/utils/git_mirror_size.py diff --git a/backend/kernelCI_app/management/commands/sync_commit_mirror.py b/backend/kernelCI_app/management/commands/sync_commit_mirror.py index 99ddbfb22..a39a34d40 100644 --- a/backend/kernelCI_app/management/commands/sync_commit_mirror.py +++ b/backend/kernelCI_app/management/commands/sync_commit_mirror.py @@ -7,7 +7,8 @@ DEFAULT_FETCH_TIMEOUT_SECONDS, sync_commit_metadata, ) -from kernelCI_app.helpers.logger import out +from kernelCI_app.helpers.logger import log_message, out +from utils.git_mirror_size import record_mirror_size class Command(BaseCommand): @@ -51,16 +52,29 @@ def add_arguments(self, parser): ) def handle(self, *args, **options): - mirror_dir = options["mirror_dir"] or settings.GIT_MIRROR_DIR - stats = sync_commit_metadata( - mirror_dir=Path(mirror_dir), - dry_run=options["dry_run"], - skip_ingest=True, - fetch_timeout=options["fetch_timeout"], - verbose_git=options["verbose_git"], - skip_unfilterable=options["skip_unfilterable"], - ) - out( - "sync_commit_mirror remotes_ok=%(remotes_ok)s remotes_failed=%(remotes_failed)s" - % stats - ) + mirror_dir = Path(options["mirror_dir"] or settings.GIT_MIRROR_DIR) + try: + stats = sync_commit_metadata( + mirror_dir=mirror_dir, + dry_run=options["dry_run"], + skip_ingest=True, + fetch_timeout=options["fetch_timeout"], + verbose_git=options["verbose_git"], + skip_unfilterable=options["skip_unfilterable"], + ) + out( + "sync_commit_mirror remotes_ok=%(remotes_ok)s remotes_failed=%(remotes_failed)s" + % stats + ) + finally: + self._record_mirror_size(mirror_dir) + + def _record_mirror_size(self, mirror_dir: Path) -> None: + try: + size = record_mirror_size(mirror_dir) + except OSError as exc: + log_message("sync_commit_mirror failed to record mirror size: %s" % exc) + return + if size is None: + return + out("sync_commit_mirror mirror_bytes=%s" % size) diff --git a/backend/kernelCI_app/tests/unitTests/utils/git_mirror_size_test.py b/backend/kernelCI_app/tests/unitTests/utils/git_mirror_size_test.py new file mode 100644 index 000000000..4b711091a --- /dev/null +++ b/backend/kernelCI_app/tests/unitTests/utils/git_mirror_size_test.py @@ -0,0 +1,62 @@ +from pathlib import Path + +from utils.git_mirror_size import ( + SIZE_FILENAME, + MirrorSizeCollector, + directory_size_bytes, + record_mirror_size, +) + + +def test_directory_size_bytes_sums_files(tmp_path: Path) -> None: + (tmp_path / "a").write_bytes(b"abc") + nested = tmp_path / "sub" + nested.mkdir() + (nested / "b").write_bytes(b"xy") + + assert directory_size_bytes(tmp_path) == 5 + + +def test_directory_size_bytes_missing_path_is_zero(tmp_path: Path) -> None: + assert directory_size_bytes(tmp_path / "missing") == 0 + + +def test_record_mirror_size_writes_integer_without_temp(tmp_path: Path) -> None: + (tmp_path / "pack").write_bytes(b"hello") + + size = record_mirror_size(tmp_path) + + written = tmp_path / SIZE_FILENAME + assert size == 5 + assert written.read_text() == "5\n" + temps = [ + path + for path in tmp_path.iterdir() + if path.name.startswith(f".{SIZE_FILENAME}.") + ] + assert temps == [] + + +def test_record_mirror_size_missing_dir_leaves_nothing(tmp_path: Path) -> None: + missing = tmp_path / "missing" + + assert record_mirror_size(missing) is None + assert not missing.exists() + + +def test_collector_emits_size_and_mtime(tmp_path: Path) -> None: + size_file = tmp_path / SIZE_FILENAME + size_file.write_text("42\n") + mtime = size_file.stat().st_mtime + + families = list(MirrorSizeCollector(size_file).collect()) + + by_name = {family.name: family.samples[0].value for family in families} + assert by_name["git_mirror_size_bytes"] == 42 + assert by_name["git_mirror_size_mtime_seconds"] == mtime + + +def test_collector_emits_nothing_when_file_is_missing(tmp_path: Path) -> None: + families = list(MirrorSizeCollector(tmp_path / SIZE_FILENAME).collect()) + + assert families == [] diff --git a/backend/utils/git_mirror_size.py b/backend/utils/git_mirror_size.py new file mode 100644 index 000000000..b0a15f4d1 --- /dev/null +++ b/backend/utils/git_mirror_size.py @@ -0,0 +1,101 @@ +"""On-disk size of the commit-metadata git mirror. + +sync_commit_mirror writes one integer at the end of a run. The long-lived +metrics process reads that file; the cron must not publish a gauge itself. +""" + +from __future__ import annotations + +import os +import tempfile +from pathlib import Path + +from prometheus_client.core import GaugeMetricFamily +from prometheus_client.registry import Collector + +SIZE_FILENAME = "mirror-size-bytes" +DEFAULT_MIRROR_DIR = "/var/lib/kernelci/git-mirror" + + +def directory_size_bytes(path: Path) -> int: + """Apparent size of files under path. Missing path is 0. + + A file that vanishes mid-walk is skipped. A directory that cannot be + listed raises, so the caller can keep the previous size file. + """ + root = Path(path) + if not root.is_dir(): + return 0 + + total = 0 + + def onerror(err: OSError) -> None: + raise err + + for dirpath, _, filenames in os.walk(root, onerror=onerror): + for name in filenames: + file_path = Path(dirpath) / name + try: + total += file_path.stat(follow_symlinks=False).st_size + except OSError: + continue + return total + + +def record_mirror_size(repo_dir: Path) -> int | None: + """Write repo_dir/mirror-size-bytes. None if there is no directory to measure. + + The write is atomic: temp file in the same directory, fsync, then replace. + A listing error leaves the previous file in place. + """ + root = Path(repo_dir) + if not root.is_dir(): + return None + size = directory_size_bytes(root) + _atomic_write(root / SIZE_FILENAME, f"{size}\n".encode()) + return size + + +def _atomic_write(path: Path, payload: bytes) -> None: + fd, tmp_name = tempfile.mkstemp(dir=path.parent, prefix=f".{path.name}.") + tmp = Path(tmp_name) + try: + view = memoryview(payload) + while view: + written = os.write(fd, view) + view = view[written:] + os.fsync(fd) + except BaseException: + os.close(fd) + tmp.unlink(missing_ok=True) + raise + os.close(fd) + try: + os.replace(tmp, path) + except BaseException: + tmp.unlink(missing_ok=True) + raise + + +class MirrorSizeCollector(Collector): + """Expose the size file written by sync_commit_mirror. Absent file emits nothing.""" + + def __init__(self, path: Path) -> None: + self.path = Path(path) + + def collect(self): + try: + st = self.path.stat() + size = int(self.path.read_text().strip()) + except (OSError, ValueError): + return + yield GaugeMetricFamily( + "git_mirror_size_bytes", + "Apparent size in bytes of the git mirror after the last sync_commit_mirror run", + value=size, + ) + yield GaugeMetricFamily( + "git_mirror_size_mtime_seconds", + "Unix mtime of the git mirror size file, set when sync_commit_mirror finishes", + value=st.st_mtime, + ) diff --git a/backend/utils/prometheus_aggregator.py b/backend/utils/prometheus_aggregator.py index ab99c545d..3916cba5a 100644 --- a/backend/utils/prometheus_aggregator.py +++ b/backend/utils/prometheus_aggregator.py @@ -1,6 +1,9 @@ import os import time +from pathlib import Path +# Entrypoint runs this file as a script, so its directory is sys.path[0]. +from git_mirror_size import DEFAULT_MIRROR_DIR, SIZE_FILENAME, MirrorSizeCollector from prometheus_client import REGISTRY, start_http_server from prometheus_client.multiprocess import MultiProcessCollector @@ -14,6 +17,8 @@ # Register the multi-process collector REGISTRY.register(MultiProcessCollector(REGISTRY)) +mirror_dir = Path(os.environ.get("GIT_MIRROR_DIR", DEFAULT_MIRROR_DIR)) +REGISTRY.register(MirrorSizeCollector(mirror_dir / SIZE_FILENAME)) start_http_server(port) diff --git a/docs/monitoring.md b/docs/monitoring.md index 0a6a16a2c..dc7489e20 100644 --- a/docs/monitoring.md +++ b/docs/monitoring.md @@ -99,6 +99,20 @@ The monitoring system supports multi-worker Gunicorn deployments using Prometheu - `PROMETHEUS_METRICS_PORT`: Port for the metrics aggregator (default: `8001`) - `PROMETHEUS_MULTIPROC_DIR`: Directory for multiprocess metric files (default: `/tmp/prometheus_multiproc_dir`) +### Git mirror size + +`sync_commit_mirror` writes `GIT_MIRROR_DIR/mirror-size-bytes` when the fetch +finishes. The aggregator (not `runserver`, and not the cron process) reads +that file on each scrape: + +- `git_mirror_size_bytes` — apparent size of the mirror, in bytes +- `git_mirror_size_mtime_seconds` — Unix mtime of the size file, so + `time() - git_mirror_size_mtime_seconds` is how long ago it was recorded + +Both series exist only when `PROMETHEUS_METRICS_ENABLED=true` and a fetch has +completed. `GitMirrorSizeHigh` in `monitoring/django.rules` fires when the +size stays above 10 GB for 30 minutes. Steady state is about 6–8 GB. + ### Cronjob Healthchecks The backend can ping healthcheck.io for cronjobs that run Django management commands. diff --git a/docs/sync-commits.md b/docs/sync-commits.md index 1d946320e..632b6e6da 100644 --- a/docs/sync-commits.md +++ b/docs/sync-commits.md @@ -74,6 +74,14 @@ to the mirror. The next ingest does `rev-list --remotes` with each stored tip as stdin (git's `--not --stdin` does not exclude stdin lines). Fetch-only does not touch that file. +When `sync_commit_mirror` finishes (including a failed run, and `--dry-run`), +it writes `GIT_MIRROR_DIR/mirror-size-bytes`: the apparent size of the mirror +in bytes. The write is a temp file in that directory, `fsync`, then replace. +With `PROMETHEUS_METRICS_ENABLED=true`, the metrics process exposes +`git_mirror_size_bytes` and `git_mirror_size_mtime_seconds` (Unix mtime of +that file). Prometheus alerts `GitMirrorSizeHigh` when the size stays above +10 GB for 30 minutes. The series stay absent until the first completed fetch. + ### Recreate the volume A pack that slipped in unfiltered stays until the volume is wiped. diff --git a/monitoring/django.rules b/monitoring/django.rules index 85ad6188b..26890d9ed 100644 --- a/monitoring/django.rules +++ b/monitoring/django.rules @@ -36,3 +36,12 @@ groups: annotations: summary: "High database query time detected" description: "Django average database query time is {{ $value }}s" + + - alert: GitMirrorSizeHigh + expr: git_mirror_size_bytes > 10 * 1000 * 1000 * 1000 + for: 30m + labels: + severity: warning + annotations: + summary: "Git commit mirror is larger than 10 GB" + description: "GIT_MIRROR_DIR is {{ $value | humanize1024 }}B after the last sync_commit_mirror run. Steady state is about 6–8 GB."