From 0c6b034521ec6f21dd85d471b4ed168420451891 Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Wed, 26 Aug 2026 11:55:08 +0530 Subject: [PATCH 1/9] feat(facade): check repo size and enforce clone limit before cloning (#458) Signed-off-by: Diptesh Roy --- collectoss/application/config.py | 2 + .../facade_worker/facade_worker/config.py | 2 + .../facade_worker/facade_worker/repofetch.py | 65 ++++++++++++++++++ .../test_git/test_repo_size_limit.py | 68 +++++++++++++++++++ 4 files changed, 137 insertions(+) create mode 100644 tests/test_tasks/test_git/test_repo_size_limit.py diff --git a/collectoss/application/config.py b/collectoss/application/config.py index 43bef8b49..8375ec29f 100644 --- a/collectoss/application/config.py +++ b/collectoss/application/config.py @@ -58,6 +58,8 @@ def redact_setting_value(section_name, setting_name, value): "run_analysis": 1, "run_facade_contributors": 1, "commit_messages": 1, + "max_clone_size_kb": 0, + "clone_size_safety_margin": 0.5, }, "Server": { "cache_expire": "3600", diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py index 2b536a3a4..40357afc4 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py @@ -134,6 +134,8 @@ def __init__(self,logger: Logger): self.multithreaded = worker_options["multithreaded"] self.create_xlsx_summary_files = worker_options["create_xlsx_summary_files"] self.commit_messages = worker_options["commit_messages"] + self.max_clone_size_kb = worker_options.get("max_clone_size_kb", 0) + self.clone_size_safety_margin = float(worker_options.get("clone_size_safety_margin", 0.5)) self.tool_source = "Facade" self.data_source = "Git Log" diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index 49f7fae21..b4544d979 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -25,6 +25,7 @@ # and checks for any parents of HEAD that aren't already accounted for in the # repos. It also rebuilds analysis data, checks any changed affiliations and # aliases, and caches data for display. +import logging import html.parser import subprocess import os @@ -38,9 +39,63 @@ from collectoss.application.db.lib import execute_sql, get_repo_by_repo_git from typing_extensions import deprecated +logger = logging.getLogger(__name__) + class GitCloneError(Exception): pass + +def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, clone_size_safety_margin: float = 0.5, logger=None): + """ + Checks if a repository's estimated clone size exceeds max_clone_size_kb. + Returns (allowed: bool, reported_size_kb: Optional[int], estimated_size_kb: Optional[float]). + """ + if not max_clone_size_kb or max_clone_size_kb <= 0: + return True, None, None + + reported_size_kb = None + + try: + if "github.com" in repo_git.lower(): + from collectoss.tasks.github.util.util import get_owner_repo + from collectoss.tasks.github.util.github_data_access import GithubDataAccess + owner, repo = get_owner_repo(repo_git) + url = f"https://api.github.com/repos/{owner}/{repo}" + github_data_access = GithubDataAccess(None, logger) + result = github_data_access.get_resource(url) + if result and isinstance(result, dict) and "size" in result: + reported_size_kb = result["size"] + elif "gitlab.com" in repo_git.lower(): + import httpx + from urllib.parse import quote_plus + git_clean = repo_git.rstrip('/') + if git_clean.endswith('.git'): + git_clean = git_clean[:-4] + parts = git_clean.split("gitlab.com/") + if len(parts) > 1: + project_path = parts[1] + encoded_path = quote_plus(project_path) + url = f"https://gitlab.com/api/v4/projects/{encoded_path}?statistics=true" + response = httpx.get(url, timeout=10.0) + if response.status_code == 200: + data = response.json() + stats = data.get("statistics", {}) + bytes_size = stats.get("repository_size") or data.get("repository_size") + if bytes_size is not None: + reported_size_kb = int(bytes_size / 1024) + except Exception as e: + if logger: + logger.warning(f"Could not retrieve repo size for {repo_git} via API: {e}") + return True, None, None + + if reported_size_kb is not None: + estimated_size_kb = reported_size_kb * (1.0 + float(clone_size_safety_margin)) + if estimated_size_kb > max_clone_size_kb: + return False, reported_size_kb, estimated_size_kb + + return True, reported_size_kb, (reported_size_kb * (1.0 + float(clone_size_safety_margin))) if reported_size_kb is not None else None + + def git_repo_initialize(facade_helper, session, repo_git): # Select any new git repos so we can set up their locations and git clone @@ -125,6 +180,16 @@ def git_repo_initialize(facade_helper, session, repo_git): execute_sql(query) return + max_limit = getattr(facade_helper, 'max_clone_size_kb', 0) + safety_margin = getattr(facade_helper, 'clone_size_safety_margin', 0.5) + if max_limit > 0: + allowed, reported_kb, estimated_kb = check_repo_size_limit(git, max_limit, safety_margin, logger) + if not allowed: + msg = f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) exceeds maximum clone size limit ({max_limit} KB)" + update_repo_log(logger, facade_helper, row.repo_id, 'Failed (size limit)') + facade_helper.log_activity('Error', msg) + raise GitCloneError(msg) + # Create the prerequisite directories try: pathlib.Path(repo_path).mkdir(parents=True, exist_ok=True) diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py new file mode 100644 index 000000000..c2d6134f0 --- /dev/null +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -0,0 +1,68 @@ +import unittest +from unittest.mock import MagicMock, patch +import collectoss.tasks.github.util.github_data_access +from collectoss.tasks.git.util.facade_worker.facade_worker.repofetch import check_repo_size_limit, GitCloneError + + +class TestRepoSizeLimit(unittest.TestCase): + + def setUp(self): + self.github_access_patcher = patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.__init__", return_value=None) + self.mock_github_init = self.github_access_patcher.start() + + def tearDown(self): + self.github_access_patcher.stop() + + def test_size_limit_disabled(self): + allowed, reported_kb, estimated_kb = check_repo_size_limit("https://github.com/chaoss/CollectOSS", 0) + self.assertTrue(allowed) + self.assertIsNone(reported_kb) + self.assertIsNone(estimated_kb) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_repo_under_limit(self, mock_get_resource): + mock_get_resource.return_value = {"size": 1000} + allowed, reported_kb, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000, clone_size_safety_margin=0.5 + ) + self.assertTrue(allowed) + self.assertEqual(reported_kb, 1000) + self.assertEqual(estimated_kb, 1500.0) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_repo_exceeds_limit(self, mock_get_resource): + mock_get_resource.return_value = {"size": 2000} + allowed, reported_kb, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2500, clone_size_safety_margin=0.5 + ) + self.assertFalse(allowed) + self.assertEqual(reported_kb, 2000) + self.assertEqual(estimated_kb, 3000.0) + + @patch("httpx.get") + def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.json.return_value = {"statistics": {"repository_size": 5242880}} # 5120 KB + mock_httpx_get.return_value = mock_response + + allowed, reported_kb, estimated_kb = check_repo_size_limit( + "https://gitlab.com/group/project", max_clone_size_kb=5000, clone_size_safety_margin=0.5 + ) + self.assertFalse(allowed) + self.assertEqual(reported_kb, 5120) + self.assertEqual(estimated_kb, 7680.0) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_api_error_fallback(self, mock_get_resource): + mock_get_resource.side_effect = Exception("API rate limit exceeded") + allowed, reported_kb, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000, clone_size_safety_margin=0.5 + ) + self.assertTrue(allowed) + self.assertIsNone(reported_kb) + self.assertIsNone(estimated_kb) + + +if __name__ == "__main__": + unittest.main() From 6d41dacf0e1e699564e5dde7d95418a34bdfd6da Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Thu, 27 Aug 2026 11:16:41 +0530 Subject: [PATCH 2/9] fix(facade): skip API lookups for non-searchable local emails to prevent E2E worker timeout Signed-off-by: Diptesh Roy --- .../facade_worker/facade_worker/config.py | 2 +- .../contributor_interface.py | 28 +++++++++++++++++-- .../tasks/github/facade_github/tasks.py | 10 +++++++ .../test_git/test_repo_size_limit.py | 12 ++++++++ 4 files changed, 48 insertions(+), 4 deletions(-) diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py index 40357afc4..ca4299607 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py @@ -134,7 +134,7 @@ def __init__(self,logger: Logger): self.multithreaded = worker_options["multithreaded"] self.create_xlsx_summary_files = worker_options["create_xlsx_summary_files"] self.commit_messages = worker_options["commit_messages"] - self.max_clone_size_kb = worker_options.get("max_clone_size_kb", 0) + self.max_clone_size_kb = int(worker_options.get("max_clone_size_kb", 0)) self.clone_size_safety_margin = float(worker_options.get("clone_size_safety_margin", 0.5)) self.tool_source = "Facade" diff --git a/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py b/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py index b1b163a2b..b211573c5 100644 --- a/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py +++ b/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py @@ -261,6 +261,28 @@ def update_contributor(self, cntrb, max_attempts=3): +def is_valid_searchable_email(email: str) -> bool: + """Check if an email address is valid and searchable via GitHub API.""" + if not email or not isinstance(email, str): + return False + email = email.strip().lower() + if len(email) < 5 or "@" not in email: + return False + parts = email.rsplit("@", 1) + if len(parts) != 2: + return False + user, domain = parts[0], parts[1] + if not user or not domain or "." not in domain: + return False + invalid_domains = {"localhost", "augur", "none", "local", "internal", "test", "example", "invalid"} + if domain in invalid_domains: + return False + for suffix in [".local", ".internal", ".lan", ".dhcp.missouri.edu"]: + if domain.endswith(suffix): + return False + return True + + def fetch_username_from_email(logger, auth, commit) -> dict | None: """Try every distinct email found within a commit for possible username resolution. Add email to garbage table if can't be resolved. @@ -283,9 +305,9 @@ def fetch_username_from_email(logger, auth, commit) -> dict | None: logger.info(f"Here is the commit: {commit}") email_raw = commit.get('email_raw') - if not email_raw or not isinstance(email_raw, str) or len(email_raw.strip()) <= 2: - logger.warning("Commit does not contain a valid 'email_raw' value.") - return login_json # Don't bother with emails that are blank or less than 2 characters + if not is_valid_searchable_email(email_raw): + logger.warning(f"Commit contains non-searchable or local email format '{email_raw}'. Skipping API lookup.") + return login_json try: url = create_endpoint_from_email(email_raw) diff --git a/collectoss/tasks/github/facade_github/tasks.py b/collectoss/tasks/github/facade_github/tasks.py index cc380d497..4a1149653 100644 --- a/collectoss/tasks/github/facade_github/tasks.py +++ b/collectoss/tasks/github/facade_github/tasks.py @@ -11,6 +11,7 @@ from collectoss.tasks.git.util.facade_worker.facade_worker.facade00mainprogram import * from collectoss.application.db.lib import bulk_insert_dicts from collectoss.application.db.data_parse import extract_needed_contributor_data as extract_github_contributor +from collectoss.tasks.github.facade_github.contributor_interfaceable.contributor_interface import is_valid_searchable_email @@ -46,6 +47,15 @@ def process_commit_metadata(logger, auth, contributorQueue, repo_id, platform_id logger.debug(f"Commit data with email {email} has been unresolved in the past, skipping...") continue + if not is_valid_searchable_email(email): + logger.debug(f"Email '{email}' is non-searchable or local format. Marking as unresolved and skipping...") + unresolved = {"email": email, "name": name} + try: + bulk_insert_dicts(logger, unresolved, UnresolvedCommitEmail, ['email']) + except Exception as e: + logger.error(f"Could not insert non-searchable email {email} into unresolved_commit_emails: {e}") + continue + login = None #Check the contributors table for a login for the given name diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py index c2d6134f0..afd85db8d 100644 --- a/tests/test_tasks/test_git/test_repo_size_limit.py +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -63,6 +63,18 @@ def test_api_error_fallback(self, mock_get_resource): self.assertIsNone(reported_kb) self.assertIsNone(estimated_kb) + def test_is_valid_searchable_email(self): + from collectoss.tasks.github.facade_github.contributor_interfaceable.contributor_interface import is_valid_searchable_email + self.assertTrue(is_valid_searchable_email("user@example.org")) + self.assertTrue(is_valid_searchable_email("john.doe@company.co.uk")) + + self.assertFalse(is_valid_searchable_email("root@augur")) + self.assertFalse(is_valid_searchable_email("michaelwoodruff@mwc-021001.dhcp.missouri.edu")) + self.assertFalse(is_valid_searchable_email("user@localhost")) + self.assertFalse(is_valid_searchable_email("invalid_email")) + self.assertFalse(is_valid_searchable_email("")) + self.assertFalse(is_valid_searchable_email(None)) + if __name__ == "__main__": unittest.main() From 4cc4d189a4513c2dde2e08337ac15d2f982d3855 Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Sun, 6 Sep 2026 02:52:38 +0530 Subject: [PATCH 3/9] refactor: remove safety margin, use raw forge-reported size directly Per MoralCode's review feedback, the safety margin concept was confusing because the GitHub/GitLab API size field is already an approximation (measured from a bare repo, not an actual clone). Adding a multiplier on top of an already-inaccurate value made the limit unpredictable. Simplified to compare the raw reported size directly against max_clone_size_kb. Users set the limit knowing it's an approximation of the bare repo size, which is the most honest and predictable behavior. Also removes clone_size_safety_margin from config and FacadeHelper. Signed-off-by: Diptesh Roy --- collectoss/application/config.py | 1 - .../facade_worker/facade_worker/config.py | 1 - .../facade_worker/facade_worker/repofetch.py | 32 +++++++++++-------- .../test_git/test_repo_size_limit.py | 29 +++++++---------- 4 files changed, 31 insertions(+), 32 deletions(-) diff --git a/collectoss/application/config.py b/collectoss/application/config.py index 8375ec29f..66668ebff 100644 --- a/collectoss/application/config.py +++ b/collectoss/application/config.py @@ -59,7 +59,6 @@ def redact_setting_value(section_name, setting_name, value): "run_facade_contributors": 1, "commit_messages": 1, "max_clone_size_kb": 0, - "clone_size_safety_margin": 0.5, }, "Server": { "cache_expire": "3600", diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py index ca4299607..32cf6ee15 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py @@ -135,7 +135,6 @@ def __init__(self,logger: Logger): self.create_xlsx_summary_files = worker_options["create_xlsx_summary_files"] self.commit_messages = worker_options["commit_messages"] self.max_clone_size_kb = int(worker_options.get("max_clone_size_kb", 0)) - self.clone_size_safety_margin = float(worker_options.get("clone_size_safety_margin", 0.5)) self.tool_source = "Facade" self.data_source = "Git Log" diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index b4544d979..474b67ccf 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -45,13 +45,22 @@ class GitCloneError(Exception): pass -def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, clone_size_safety_margin: float = 0.5, logger=None): +def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, logger=None): """ - Checks if a repository's estimated clone size exceeds max_clone_size_kb. - Returns (allowed: bool, reported_size_kb: Optional[int], estimated_size_kb: Optional[float]). + Checks if a repository's reported size exceeds max_clone_size_kb. + + Uses the forge API's reported size (GitHub: size field in KB, GitLab: + repository_size in bytes) as a best-effort estimate. Note that forge-reported + sizes are measured from a bare repo and may differ from the actual on-disk + size after a full clone. Users should set max_clone_size_kb with this in mind. + + If the size cannot be determined (API error, unsupported forge), cloning is + allowed to proceed. + + Returns (allowed: bool, reported_size_kb: Optional[int]). """ if not max_clone_size_kb or max_clone_size_kb <= 0: - return True, None, None + return True, None reported_size_kb = None @@ -86,14 +95,12 @@ def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, clone_size_safe except Exception as e: if logger: logger.warning(f"Could not retrieve repo size for {repo_git} via API: {e}") - return True, None, None + return True, None - if reported_size_kb is not None: - estimated_size_kb = reported_size_kb * (1.0 + float(clone_size_safety_margin)) - if estimated_size_kb > max_clone_size_kb: - return False, reported_size_kb, estimated_size_kb + if reported_size_kb is not None and reported_size_kb > max_clone_size_kb: + return False, reported_size_kb - return True, reported_size_kb, (reported_size_kb * (1.0 + float(clone_size_safety_margin))) if reported_size_kb is not None else None + return True, reported_size_kb def git_repo_initialize(facade_helper, session, repo_git): @@ -181,11 +188,10 @@ def git_repo_initialize(facade_helper, session, repo_git): return max_limit = getattr(facade_helper, 'max_clone_size_kb', 0) - safety_margin = getattr(facade_helper, 'clone_size_safety_margin', 0.5) if max_limit > 0: - allowed, reported_kb, estimated_kb = check_repo_size_limit(git, max_limit, safety_margin, logger) + allowed, reported_kb = check_repo_size_limit(git, max_limit, logger) if not allowed: - msg = f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) exceeds maximum clone size limit ({max_limit} KB)" + msg = f"Repo '{git}' reported size ({reported_kb} KB) exceeds maximum clone size limit ({max_limit} KB)" update_repo_log(logger, facade_helper, row.repo_id, 'Failed (size limit)') facade_helper.log_activity('Error', msg) raise GitCloneError(msg) diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py index afd85db8d..60dd813c9 100644 --- a/tests/test_tasks/test_git/test_repo_size_limit.py +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -14,54 +14,49 @@ def tearDown(self): self.github_access_patcher.stop() def test_size_limit_disabled(self): - allowed, reported_kb, estimated_kb = check_repo_size_limit("https://github.com/chaoss/CollectOSS", 0) + allowed, reported_kb = check_repo_size_limit("https://github.com/chaoss/CollectOSS", 0) self.assertTrue(allowed) self.assertIsNone(reported_kb) - self.assertIsNone(estimated_kb) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") def test_github_repo_under_limit(self, mock_get_resource): mock_get_resource.return_value = {"size": 1000} - allowed, reported_kb, estimated_kb = check_repo_size_limit( - "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000, clone_size_safety_margin=0.5 + allowed, reported_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 ) self.assertTrue(allowed) self.assertEqual(reported_kb, 1000) - self.assertEqual(estimated_kb, 1500.0) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") def test_github_repo_exceeds_limit(self, mock_get_resource): - mock_get_resource.return_value = {"size": 2000} - allowed, reported_kb, estimated_kb = check_repo_size_limit( - "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2500, clone_size_safety_margin=0.5 + mock_get_resource.return_value = {"size": 2001} + allowed, reported_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 ) self.assertFalse(allowed) - self.assertEqual(reported_kb, 2000) - self.assertEqual(estimated_kb, 3000.0) + self.assertEqual(reported_kb, 2001) @patch("httpx.get") def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): mock_response = MagicMock() mock_response.status_code = 200 - mock_response.json.return_value = {"statistics": {"repository_size": 5242880}} # 5120 KB + mock_response.json.return_value = {"statistics": {"repository_size": 5242880}} # 5120 KB mock_httpx_get.return_value = mock_response - allowed, reported_kb, estimated_kb = check_repo_size_limit( - "https://gitlab.com/group/project", max_clone_size_kb=5000, clone_size_safety_margin=0.5 + allowed, reported_kb = check_repo_size_limit( + "https://gitlab.com/group/project", max_clone_size_kb=5000 ) self.assertFalse(allowed) self.assertEqual(reported_kb, 5120) - self.assertEqual(estimated_kb, 7680.0) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") def test_api_error_fallback(self, mock_get_resource): mock_get_resource.side_effect = Exception("API rate limit exceeded") - allowed, reported_kb, estimated_kb = check_repo_size_limit( - "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000, clone_size_safety_margin=0.5 + allowed, reported_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 ) self.assertTrue(allowed) self.assertIsNone(reported_kb) - self.assertIsNone(estimated_kb) def test_is_valid_searchable_email(self): from collectoss.tasks.github.facade_github.contributor_interfaceable.contributor_interface import is_valid_searchable_email From 75ab567e88483f5838531d4f5bd6ad1975508b5f Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Fri, 25 Sep 2026 02:35:06 +0530 Subject: [PATCH 4/9] feat: use combined bare+tree size for more precise clone size estimate Per MoralCode's research, the GitHub API 'size' field alone underestimates by ~4% because it only measures the bare repo. Adding the working tree file size (sum of all blob sizes from /git/trees/HEAD?recursive=1) gives a much more accurate estimate of actual on-disk clone size. If the tree is truncated (very large repo), falls back to bare size only. Updated tests to cover the combined estimation logic. Signed-off-by: Diptesh Roy --- .../facade_worker/facade_worker/repofetch.py | 64 +++++++++++++------ .../test_git/test_repo_size_limit.py | 59 ++++++++++++----- 2 files changed, 89 insertions(+), 34 deletions(-) diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index 474b67ccf..0592cbfc9 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -47,33 +47,58 @@ class GitCloneError(Exception): def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, logger=None): """ - Checks if a repository's reported size exceeds max_clone_size_kb. + Estimates a repository's checkout size and checks if it exceeds max_clone_size_kb. - Uses the forge API's reported size (GitHub: size field in KB, GitLab: - repository_size in bytes) as a best-effort estimate. Note that forge-reported - sizes are measured from a bare repo and may differ from the actual on-disk - size after a full clone. Users should set max_clone_size_kb with this in mind. + For GitHub repos, uses a two-part estimate that combines: + 1. The bare repo size (GitHub API 'size' field, in KB) — captures git object storage + 2. The working tree file size (sum of all blob sizes from the git tree API, in bytes) + via GET /repos/{owner}/{repo}/git/trees/HEAD?recursive=1 - If the size cannot be determined (API error, unsupported forge), cloning is - allowed to proceed. + This combined estimate closely approximates the actual on-disk size of a fresh clone + (bare + checkout). Testing shows ~4% error vs actual clone size, which is much more + accurate than using the bare repo size alone. - Returns (allowed: bool, reported_size_kb: Optional[int]). + For GitLab repos, falls back to the repository_size from the statistics API. + + If the size cannot be determined (API error, truncated tree, unsupported forge), + cloning is allowed to proceed. + + Returns (allowed: bool, estimated_size_kb: Optional[float]). """ if not max_clone_size_kb or max_clone_size_kb <= 0: return True, None - reported_size_kb = None + estimated_size_kb = None try: if "github.com" in repo_git.lower(): from collectoss.tasks.github.util.util import get_owner_repo from collectoss.tasks.github.util.github_data_access import GithubDataAccess owner, repo = get_owner_repo(repo_git) - url = f"https://api.github.com/repos/{owner}/{repo}" github_data_access = GithubDataAccess(None, logger) - result = github_data_access.get_resource(url) - if result and isinstance(result, dict) and "size" in result: - reported_size_kb = result["size"] + + # Part 1: bare repo size from the repo metadata endpoint + bare_size_kb = 0 + repo_url = f"https://api.github.com/repos/{owner}/{repo}" + repo_info = github_data_access.get_resource(repo_url) + if repo_info and isinstance(repo_info, dict) and "size" in repo_info: + bare_size_kb = repo_info["size"] + + # Part 2: working tree file size from the git tree API + # Sum of all blob (file) sizes gives the checkout size + file_tree_bytes = 0 + tree_url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/HEAD?recursive=1" + tree_data = github_data_access.get_resource(tree_url) + if tree_data and isinstance(tree_data, dict): + # If truncated=True the tree is too large to enumerate; skip file size + if not tree_data.get("truncated", False): + for item in tree_data.get("tree", []): + if item.get("type") == "blob" and item.get("size") is not None: + file_tree_bytes += item["size"] + + if bare_size_kb > 0 or file_tree_bytes > 0: + estimated_size_kb = bare_size_kb + (file_tree_bytes / 1024) + elif "gitlab.com" in repo_git.lower(): import httpx from urllib.parse import quote_plus @@ -91,16 +116,17 @@ def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, logger=None): stats = data.get("statistics", {}) bytes_size = stats.get("repository_size") or data.get("repository_size") if bytes_size is not None: - reported_size_kb = int(bytes_size / 1024) + estimated_size_kb = bytes_size / 1024 + except Exception as e: if logger: logger.warning(f"Could not retrieve repo size for {repo_git} via API: {e}") return True, None - if reported_size_kb is not None and reported_size_kb > max_clone_size_kb: - return False, reported_size_kb + if estimated_size_kb is not None and estimated_size_kb > max_clone_size_kb: + return False, estimated_size_kb - return True, reported_size_kb + return True, estimated_size_kb def git_repo_initialize(facade_helper, session, repo_git): @@ -189,9 +215,9 @@ def git_repo_initialize(facade_helper, session, repo_git): max_limit = getattr(facade_helper, 'max_clone_size_kb', 0) if max_limit > 0: - allowed, reported_kb = check_repo_size_limit(git, max_limit, logger) + allowed, estimated_kb = check_repo_size_limit(git, max_limit, logger) if not allowed: - msg = f"Repo '{git}' reported size ({reported_kb} KB) exceeds maximum clone size limit ({max_limit} KB)" + msg = f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) exceeds maximum clone size limit ({max_limit} KB)" update_repo_log(logger, facade_helper, row.repo_id, 'Failed (size limit)') facade_helper.log_activity('Error', msg) raise GitCloneError(msg) diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py index 60dd813c9..716e3897f 100644 --- a/tests/test_tasks/test_git/test_repo_size_limit.py +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -1,5 +1,5 @@ import unittest -from unittest.mock import MagicMock, patch +from unittest.mock import MagicMock, patch, call import collectoss.tasks.github.util.github_data_access from collectoss.tasks.git.util.facade_worker.facade_worker.repofetch import check_repo_size_limit, GitCloneError @@ -14,27 +14,56 @@ def tearDown(self): self.github_access_patcher.stop() def test_size_limit_disabled(self): - allowed, reported_kb = check_repo_size_limit("https://github.com/chaoss/CollectOSS", 0) + allowed, estimated_kb = check_repo_size_limit("https://github.com/chaoss/CollectOSS", 0) self.assertTrue(allowed) - self.assertIsNone(reported_kb) + self.assertIsNone(estimated_kb) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") - def test_github_repo_under_limit(self, mock_get_resource): - mock_get_resource.return_value = {"size": 1000} - allowed, reported_kb = check_repo_size_limit( + def test_github_repo_under_limit_combined(self, mock_get_resource): + # bare size: 1000 KB, file tree: 500,000 bytes (~488 KB) + # estimated = 1000 + 488 = 1488 KB < 2000 KB limit + mock_get_resource.side_effect = [ + {"size": 1000}, # repo info call + {"truncated": False, "tree": [ + {"type": "blob", "size": 300000}, + {"type": "blob", "size": 200000}, + {"type": "tree", "size": None}, # directories are ignored + ]} # git tree call + ] + allowed, estimated_kb = check_repo_size_limit( "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 ) self.assertTrue(allowed) - self.assertEqual(reported_kb, 1000) + self.assertAlmostEqual(estimated_kb, 1000 + (500000 / 1024), places=1) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") - def test_github_repo_exceeds_limit(self, mock_get_resource): - mock_get_resource.return_value = {"size": 2001} - allowed, reported_kb = check_repo_size_limit( + def test_github_repo_exceeds_limit_combined(self, mock_get_resource): + # bare size: 1000 KB, file tree: 1,500,000 bytes (~1465 KB) + # estimated = 1000 + 1465 = 2465 KB > 2000 KB limit + mock_get_resource.side_effect = [ + {"size": 1000}, + {"truncated": False, "tree": [ + {"type": "blob", "size": 1500000}, + ]} + ] + allowed, estimated_kb = check_repo_size_limit( "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 ) self.assertFalse(allowed) - self.assertEqual(reported_kb, 2001) + self.assertAlmostEqual(estimated_kb, 1000 + (1500000 / 1024), places=1) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_truncated_tree_uses_bare_size_only(self, mock_get_resource): + # When tree is truncated (large repo), fall back to bare size only + mock_get_resource.side_effect = [ + {"size": 2500}, + {"truncated": True, "tree": []} # truncated — can't sum blobs + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 + ) + self.assertFalse(allowed) + self.assertEqual(estimated_kb, 2500.0) @patch("httpx.get") def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): @@ -43,20 +72,20 @@ def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): mock_response.json.return_value = {"statistics": {"repository_size": 5242880}} # 5120 KB mock_httpx_get.return_value = mock_response - allowed, reported_kb = check_repo_size_limit( + allowed, estimated_kb = check_repo_size_limit( "https://gitlab.com/group/project", max_clone_size_kb=5000 ) self.assertFalse(allowed) - self.assertEqual(reported_kb, 5120) + self.assertAlmostEqual(estimated_kb, 5120.0, places=1) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") def test_api_error_fallback(self, mock_get_resource): mock_get_resource.side_effect = Exception("API rate limit exceeded") - allowed, reported_kb = check_repo_size_limit( + allowed, estimated_kb = check_repo_size_limit( "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 ) self.assertTrue(allowed) - self.assertIsNone(reported_kb) + self.assertIsNone(estimated_kb) def test_is_valid_searchable_email(self): from collectoss.tasks.github.facade_github.contributor_interfaceable.contributor_interface import is_valid_searchable_email From 4ec0dcd728172cb6f80140a813180591798e8144 Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Fri, 25 Sep 2026 23:14:39 +0530 Subject: [PATCH 5/9] fix: set default max_clone_size_kb to 5GB, remove unrelated changes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Default max_clone_size_kb changed from 0 (disabled) to 5242880 KB (5 GB) as suggested by MoralCode — provides a sensible out-of-the-box limit - Reverted contributor_interface.py and tasks.py to upstream — those changes are unrelated to the repo size limit feature Signed-off-by: Diptesh Roy --- collectoss/application/config.py | 2 +- .../facade_worker/facade_worker/config.py | 2 +- .../contributor_interface.py | 28 ++----------------- .../tasks/github/facade_github/tasks.py | 10 ------- 4 files changed, 5 insertions(+), 37 deletions(-) diff --git a/collectoss/application/config.py b/collectoss/application/config.py index 66668ebff..8ac04ca02 100644 --- a/collectoss/application/config.py +++ b/collectoss/application/config.py @@ -58,7 +58,7 @@ def redact_setting_value(section_name, setting_name, value): "run_analysis": 1, "run_facade_contributors": 1, "commit_messages": 1, - "max_clone_size_kb": 0, + "max_clone_size_kb": 5242880, }, "Server": { "cache_expire": "3600", diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py index 32cf6ee15..b7a14d8a3 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py @@ -134,7 +134,7 @@ def __init__(self,logger: Logger): self.multithreaded = worker_options["multithreaded"] self.create_xlsx_summary_files = worker_options["create_xlsx_summary_files"] self.commit_messages = worker_options["commit_messages"] - self.max_clone_size_kb = int(worker_options.get("max_clone_size_kb", 0)) + self.max_clone_size_kb = int(worker_options.get("max_clone_size_kb", 5242880)) self.tool_source = "Facade" self.data_source = "Git Log" diff --git a/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py b/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py index b211573c5..b1b163a2b 100644 --- a/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py +++ b/collectoss/tasks/github/facade_github/contributor_interfaceable/contributor_interface.py @@ -261,28 +261,6 @@ def update_contributor(self, cntrb, max_attempts=3): -def is_valid_searchable_email(email: str) -> bool: - """Check if an email address is valid and searchable via GitHub API.""" - if not email or not isinstance(email, str): - return False - email = email.strip().lower() - if len(email) < 5 or "@" not in email: - return False - parts = email.rsplit("@", 1) - if len(parts) != 2: - return False - user, domain = parts[0], parts[1] - if not user or not domain or "." not in domain: - return False - invalid_domains = {"localhost", "augur", "none", "local", "internal", "test", "example", "invalid"} - if domain in invalid_domains: - return False - for suffix in [".local", ".internal", ".lan", ".dhcp.missouri.edu"]: - if domain.endswith(suffix): - return False - return True - - def fetch_username_from_email(logger, auth, commit) -> dict | None: """Try every distinct email found within a commit for possible username resolution. Add email to garbage table if can't be resolved. @@ -305,9 +283,9 @@ def fetch_username_from_email(logger, auth, commit) -> dict | None: logger.info(f"Here is the commit: {commit}") email_raw = commit.get('email_raw') - if not is_valid_searchable_email(email_raw): - logger.warning(f"Commit contains non-searchable or local email format '{email_raw}'. Skipping API lookup.") - return login_json + if not email_raw or not isinstance(email_raw, str) or len(email_raw.strip()) <= 2: + logger.warning("Commit does not contain a valid 'email_raw' value.") + return login_json # Don't bother with emails that are blank or less than 2 characters try: url = create_endpoint_from_email(email_raw) diff --git a/collectoss/tasks/github/facade_github/tasks.py b/collectoss/tasks/github/facade_github/tasks.py index 4a1149653..cc380d497 100644 --- a/collectoss/tasks/github/facade_github/tasks.py +++ b/collectoss/tasks/github/facade_github/tasks.py @@ -11,7 +11,6 @@ from collectoss.tasks.git.util.facade_worker.facade_worker.facade00mainprogram import * from collectoss.application.db.lib import bulk_insert_dicts from collectoss.application.db.data_parse import extract_needed_contributor_data as extract_github_contributor -from collectoss.tasks.github.facade_github.contributor_interfaceable.contributor_interface import is_valid_searchable_email @@ -47,15 +46,6 @@ def process_commit_metadata(logger, auth, contributorQueue, repo_id, platform_id logger.debug(f"Commit data with email {email} has been unresolved in the past, skipping...") continue - if not is_valid_searchable_email(email): - logger.debug(f"Email '{email}' is non-searchable or local format. Marking as unresolved and skipping...") - unresolved = {"email": email, "name": name} - try: - bulk_insert_dicts(logger, unresolved, UnresolvedCommitEmail, ['email']) - except Exception as e: - logger.error(f"Could not insert non-searchable email {email} into unresolved_commit_emails: {e}") - continue - login = None #Check the contributors table for a login for the given name From 1727dd17d7203738a8517b1f680098b40c8d9596 Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Sat, 26 Sep 2026 21:35:23 +0530 Subject: [PATCH 6/9] fix: remove unrelated email validation test from repo size limit tests Signed-off-by: Diptesh Roy --- tests/test_tasks/test_git/test_repo_size_limit.py | 12 ------------ 1 file changed, 12 deletions(-) diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py index 716e3897f..3aa7c1686 100644 --- a/tests/test_tasks/test_git/test_repo_size_limit.py +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -87,18 +87,6 @@ def test_api_error_fallback(self, mock_get_resource): self.assertTrue(allowed) self.assertIsNone(estimated_kb) - def test_is_valid_searchable_email(self): - from collectoss.tasks.github.facade_github.contributor_interfaceable.contributor_interface import is_valid_searchable_email - self.assertTrue(is_valid_searchable_email("user@example.org")) - self.assertTrue(is_valid_searchable_email("john.doe@company.co.uk")) - - self.assertFalse(is_valid_searchable_email("root@augur")) - self.assertFalse(is_valid_searchable_email("michaelwoodruff@mwc-021001.dhcp.missouri.edu")) - self.assertFalse(is_valid_searchable_email("user@localhost")) - self.assertFalse(is_valid_searchable_email("invalid_email")) - self.assertFalse(is_valid_searchable_email("")) - self.assertFalse(is_valid_searchable_email(None)) - if __name__ == "__main__": unittest.main() From d359e8eeaeb09422c5f4be3e1572efcf5ff1427f Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Sun, 27 Sep 2026 19:22:59 +0530 Subject: [PATCH 7/9] refactor(repofetch): separate size fetchers, fail closed on error Address MoralCode's design review: - Split check_repo_size_limit into three functions: _get_github_repo_size_kb: raises on API error or non-dict response _get_gitlab_repo_size_kb: calls raise_for_status, raises on missing field check_repo_size_limit: policy-only, calls forge-specific fetchers - Fail CLOSED: API error, malformed response, or unsupported forge with limit>0 returns (False, None) to block the clone rather than allow it - Truncated GitHub tree: partial blob data is counted even when truncated=True, since the endpoint is not paginated and partial is better than zero (per MoralCode's feedback) - Fix TypeError crash in git_repo_initialize: estimated_kb can be None when fail-closed, so the error message now handles both cases - Tests: 12 tests covering disabled limit, under/over/exact, truncated tree (blocked and allowed), API error, malformed response, unsupported forge, and GitLab path Signed-off-by: Diptesh Roy --- .../facade_worker/facade_worker/repofetch.py | 182 +++++++++++------- .../test_git/test_repo_size_limit.py | 163 +++++++++++++--- 2 files changed, 256 insertions(+), 89 deletions(-) diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index 0592cbfc9..799ce5ab9 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -45,88 +45,131 @@ class GitCloneError(Exception): pass -def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, logger=None): +def _get_github_repo_size_kb(repo_git: str, logger=None): """ - Estimates a repository's checkout size and checks if it exceeds max_clone_size_kb. + Fetches the estimated size of a GitHub repository in KB. - For GitHub repos, uses a two-part estimate that combines: - 1. The bare repo size (GitHub API 'size' field, in KB) — captures git object storage - 2. The working tree file size (sum of all blob sizes from the git tree API, in bytes) - via GET /repos/{owner}/{repo}/git/trees/HEAD?recursive=1 + Combines two signals: + 1. The bare repo size from the repo metadata API (GitHub 'size' field, in KB) + 2. The working tree blob sizes from /git/trees/HEAD?recursive=1 (in bytes) - This combined estimate closely approximates the actual on-disk size of a fresh clone - (bare + checkout). Testing shows ~4% error vs actual clone size, which is much more - accurate than using the bare repo size alone. + If the tree response is truncated (too many entries for one page), the partial + blob data from the response is still used — GitHub does not paginate this endpoint, + so partial is all we can get. This is still more accurate than ignoring it. - For GitLab repos, falls back to the repository_size from the statistics API. + Returns the estimated size in KB, or raises an exception if the API is unreachable + or returns unexpected data. + """ + from collectoss.tasks.github.util.util import get_owner_repo + from collectoss.tasks.github.util.github_data_access import GithubDataAccess + owner, repo = get_owner_repo(repo_git) + github_data_access = GithubDataAccess(None, logger) + + # Part 1: bare repo size + repo_url = f"https://api.github.com/repos/{owner}/{repo}" + repo_info = github_data_access.get_resource(repo_url) + if not isinstance(repo_info, dict): + raise ValueError( + f"GitHub API returned unexpected response for repo metadata: {type(repo_info)}" + ) + bare_size_kb = repo_info.get("size", 0) or 0 + + # Part 2: working tree blob sizes + # Note: this endpoint is NOT paginated. truncated=True means the response was + # cut off due to size, but the entries that ARE present are still valid and + # should be counted. We use whatever partial data we received. + tree_url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/HEAD?recursive=1" + tree_data = github_data_access.get_resource(tree_url) + if not isinstance(tree_data, dict): + raise ValueError( + f"GitHub API returned unexpected response for tree data: {type(tree_data)}" + ) + file_tree_bytes = 0 + for item in tree_data.get("tree", []): + if item.get("type") == "blob" and item.get("size") is not None: + file_tree_bytes += item["size"] + if tree_data.get("truncated", False) and logger: + logger.warning( + f"Git tree response for {repo_git} was truncated — " + f"file size estimate uses partial data ({file_tree_bytes} bytes counted so far)" + ) + + return bare_size_kb + (file_tree_bytes / 1024) - If the size cannot be determined (API error, truncated tree, unsupported forge), - cloning is allowed to proceed. - Returns (allowed: bool, estimated_size_kb: Optional[float]). +def _get_gitlab_repo_size_kb(repo_git: str, logger=None): + """ + Fetches the estimated size of a GitLab repository in KB using the statistics API. + + Returns the repository size in KB, or raises an exception if the API + is unreachable or returns unexpected data. + """ + import httpx + from urllib.parse import quote_plus + git_clean = repo_git.rstrip('/') + if git_clean.endswith('.git'): + git_clean = git_clean[:-4] + parts = git_clean.split("gitlab.com/") + if len(parts) < 2: + raise ValueError(f"Could not parse GitLab project path from: {repo_git}") + project_path = parts[1] + encoded_path = quote_plus(project_path) + url = f"https://gitlab.com/api/v4/projects/{encoded_path}?statistics=true" + response = httpx.get(url, timeout=10.0) + response.raise_for_status() + data = response.json() + stats = data.get("statistics", {}) + bytes_size = stats.get("repository_size") or data.get("repository_size") + if bytes_size is None: + raise ValueError(f"No repository_size field in GitLab response for {repo_git}") + return bytes_size / 1024 + + +def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, logger=None): + """ + Checks whether cloning a repository is permitted under the configured size limit. + + Retrieves the estimated repository size using forge-specific logic, then + compares it against max_clone_size_kb. + + If the size CANNOT be determined (API error, network failure, unsupported forge), + cloning is BLOCKED. This is fail-closed behavior: a configured limit must not + be bypassed silently because of an error. + + Returns: + (True, estimated_kb) — clone is permitted + (False, estimated_kb) — clone is blocked because limit is exceeded + (False, None) — clone is blocked because size could not be determined """ if not max_clone_size_kb or max_clone_size_kb <= 0: return True, None - estimated_size_kb = None - try: if "github.com" in repo_git.lower(): - from collectoss.tasks.github.util.util import get_owner_repo - from collectoss.tasks.github.util.github_data_access import GithubDataAccess - owner, repo = get_owner_repo(repo_git) - github_data_access = GithubDataAccess(None, logger) - - # Part 1: bare repo size from the repo metadata endpoint - bare_size_kb = 0 - repo_url = f"https://api.github.com/repos/{owner}/{repo}" - repo_info = github_data_access.get_resource(repo_url) - if repo_info and isinstance(repo_info, dict) and "size" in repo_info: - bare_size_kb = repo_info["size"] - - # Part 2: working tree file size from the git tree API - # Sum of all blob (file) sizes gives the checkout size - file_tree_bytes = 0 - tree_url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/HEAD?recursive=1" - tree_data = github_data_access.get_resource(tree_url) - if tree_data and isinstance(tree_data, dict): - # If truncated=True the tree is too large to enumerate; skip file size - if not tree_data.get("truncated", False): - for item in tree_data.get("tree", []): - if item.get("type") == "blob" and item.get("size") is not None: - file_tree_bytes += item["size"] - - if bare_size_kb > 0 or file_tree_bytes > 0: - estimated_size_kb = bare_size_kb + (file_tree_bytes / 1024) - + estimated_kb = _get_github_repo_size_kb(repo_git, logger) elif "gitlab.com" in repo_git.lower(): - import httpx - from urllib.parse import quote_plus - git_clean = repo_git.rstrip('/') - if git_clean.endswith('.git'): - git_clean = git_clean[:-4] - parts = git_clean.split("gitlab.com/") - if len(parts) > 1: - project_path = parts[1] - encoded_path = quote_plus(project_path) - url = f"https://gitlab.com/api/v4/projects/{encoded_path}?statistics=true" - response = httpx.get(url, timeout=10.0) - if response.status_code == 200: - data = response.json() - stats = data.get("statistics", {}) - bytes_size = stats.get("repository_size") or data.get("repository_size") - if bytes_size is not None: - estimated_size_kb = bytes_size / 1024 + estimated_kb = _get_gitlab_repo_size_kb(repo_git, logger) + else: + # Unsupported forge — cannot determine size, fail closed + if logger: + logger.warning( + f"Cannot determine repo size for unsupported forge: {repo_git}. " + f"Blocking clone to enforce configured limit of {max_clone_size_kb} KB." + ) + return False, None except Exception as e: if logger: - logger.warning(f"Could not retrieve repo size for {repo_git} via API: {e}") - return True, None + logger.warning( + f"Could not retrieve repo size for {repo_git}: {e}. " + f"Blocking clone to enforce configured limit of {max_clone_size_kb} KB." + ) + return False, None - if estimated_size_kb is not None and estimated_size_kb > max_clone_size_kb: - return False, estimated_size_kb + if estimated_kb > max_clone_size_kb: + return False, estimated_kb - return True, estimated_size_kb + return True, estimated_kb def git_repo_initialize(facade_helper, session, repo_git): @@ -217,7 +260,16 @@ def git_repo_initialize(facade_helper, session, repo_git): if max_limit > 0: allowed, estimated_kb = check_repo_size_limit(git, max_limit, logger) if not allowed: - msg = f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) exceeds maximum clone size limit ({max_limit} KB)" + if estimated_kb is not None: + msg = ( + f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) " + f"exceeds maximum clone size limit ({max_limit} KB)" + ) + else: + msg = ( + f"Repo '{git}' could not be size-checked; " + f"blocking clone to enforce configured limit of {max_limit} KB" + ) update_repo_log(logger, facade_helper, row.repo_id, 'Failed (size limit)') facade_helper.log_activity('Error', msg) raise GitCloneError(msg) diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py index 3aa7c1686..9a9f29f42 100644 --- a/tests/test_tasks/test_git/test_repo_size_limit.py +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -1,34 +1,50 @@ import unittest -from unittest.mock import MagicMock, patch, call +from unittest.mock import MagicMock, patch import collectoss.tasks.github.util.github_data_access -from collectoss.tasks.git.util.facade_worker.facade_worker.repofetch import check_repo_size_limit, GitCloneError +from collectoss.tasks.git.util.facade_worker.facade_worker.repofetch import ( + check_repo_size_limit, GitCloneError, _get_github_repo_size_kb, _get_gitlab_repo_size_kb +) class TestRepoSizeLimit(unittest.TestCase): def setUp(self): - self.github_access_patcher = patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.__init__", return_value=None) + self.github_access_patcher = patch( + "collectoss.tasks.github.util.github_data_access.GithubDataAccess.__init__", + return_value=None + ) self.mock_github_init = self.github_access_patcher.start() def tearDown(self): self.github_access_patcher.stop() + # ------------------------------------------------------------------ + # check_repo_size_limit — limit disabled + # ------------------------------------------------------------------ + def test_size_limit_disabled(self): - allowed, estimated_kb = check_repo_size_limit("https://github.com/chaoss/CollectOSS", 0) + """When limit is 0 (disabled), always allow regardless of repo size.""" + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", 0 + ) self.assertTrue(allowed) self.assertIsNone(estimated_kb) + # ------------------------------------------------------------------ + # check_repo_size_limit — GitHub, normal (non-truncated) tree + # ------------------------------------------------------------------ + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") - def test_github_repo_under_limit_combined(self, mock_get_resource): - # bare size: 1000 KB, file tree: 500,000 bytes (~488 KB) - # estimated = 1000 + 488 = 1488 KB < 2000 KB limit + def test_github_repo_under_limit(self, mock_get_resource): + """Repo comfortably below limit — clone is allowed.""" + # bare: 1000 KB, blobs: 500,000 bytes = ~488 KB → ~1488 KB total mock_get_resource.side_effect = [ - {"size": 1000}, # repo info call + {"size": 1000}, {"truncated": False, "tree": [ {"type": "blob", "size": 300000}, {"type": "blob", "size": 200000}, - {"type": "tree", "size": None}, # directories are ignored - ]} # git tree call + {"type": "tree", "size": None}, # directories have no size + ]} ] allowed, estimated_kb = check_repo_size_limit( "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 @@ -37,9 +53,9 @@ def test_github_repo_under_limit_combined(self, mock_get_resource): self.assertAlmostEqual(estimated_kb, 1000 + (500000 / 1024), places=1) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") - def test_github_repo_exceeds_limit_combined(self, mock_get_resource): - # bare size: 1000 KB, file tree: 1,500,000 bytes (~1465 KB) - # estimated = 1000 + 1465 = 2465 KB > 2000 KB limit + def test_github_repo_exceeds_limit(self, mock_get_resource): + """Repo above limit — clone is blocked.""" + # bare: 1000 KB, blobs: 1,500,000 bytes = ~1465 KB → ~2465 KB total > 2000 KB mock_get_resource.side_effect = [ {"size": 1000}, {"truncated": False, "tree": [ @@ -53,23 +69,120 @@ def test_github_repo_exceeds_limit_combined(self, mock_get_resource): self.assertAlmostEqual(estimated_kb, 1000 + (1500000 / 1024), places=1) @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") - def test_github_truncated_tree_uses_bare_size_only(self, mock_get_resource): - # When tree is truncated (large repo), fall back to bare size only + def test_github_repo_exactly_at_limit(self, mock_get_resource): + """Repo exactly at the limit — clone is allowed (limit is exclusive upper bound).""" + mock_get_resource.side_effect = [ + {"size": 2000}, + {"truncated": False, "tree": []} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 + ) + self.assertTrue(allowed) + self.assertEqual(estimated_kb, 2000.0) + + # ------------------------------------------------------------------ + # check_repo_size_limit — GitHub, truncated tree + # ------------------------------------------------------------------ + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_truncated_tree_uses_partial_blob_data(self, mock_get_resource): + """When the tree is truncated, partial blob data is still counted. + + The GitHub git/trees API is NOT paginated for recursive requests. + truncated=True means entries were cut off, but the entries present + are still valid and should contribute to the size estimate. + """ + # bare: 1000 KB, partial blobs (truncated): 800,000 bytes = ~781 KB + # total: ~1781 KB > 1500 KB limit + mock_get_resource.side_effect = [ + {"size": 1000}, + {"truncated": True, "tree": [ + {"type": "blob", "size": 800000}, + ]} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1500 + ) + self.assertFalse(allowed) + self.assertAlmostEqual(estimated_kb, 1000 + (800000 / 1024), places=1) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_truncated_tree_under_limit_still_allowed(self, mock_get_resource): + """Truncated tree with partial data that falls under the limit — still allowed. + + The estimate is a lower bound when truncated, so we allow the clone. + """ + # bare: 500 KB, partial blobs: 100,000 bytes = ~98 KB → ~598 KB < 2000 KB mock_get_resource.side_effect = [ - {"size": 2500}, - {"truncated": True, "tree": []} # truncated — can't sum blobs + {"size": 500}, + {"truncated": True, "tree": [ + {"type": "blob", "size": 100000}, + ]} ] allowed, estimated_kb = check_repo_size_limit( "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 ) + self.assertTrue(allowed) + self.assertAlmostEqual(estimated_kb, 500 + (100000 / 1024), places=1) + + # ------------------------------------------------------------------ + # check_repo_size_limit — error handling (fail closed) + # ------------------------------------------------------------------ + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_api_error_blocks_clone(self, mock_get_resource): + """When the API call fails, cloning is BLOCKED (fail closed). + + A configured limit must not be silently bypassed because of a + network failure or API error. + """ + mock_get_resource.side_effect = Exception("API rate limit exceeded") + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 + ) + self.assertFalse(allowed) + self.assertIsNone(estimated_kb) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_malformed_api_response_blocks_clone(self, mock_get_resource): + """When the API returns a non-dict, _get_github_repo_size_kb raises, blocking the clone.""" + mock_get_resource.return_value = "not a dict" + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 + ) + self.assertFalse(allowed) + self.assertIsNone(estimated_kb) + + def test_unsupported_forge_blocks_clone(self): + """For unsupported forges, cloning is blocked when a limit is configured.""" + allowed, estimated_kb = check_repo_size_limit( + "https://bitbucket.org/some/repo", max_clone_size_kb=1000 + ) self.assertFalse(allowed) - self.assertEqual(estimated_kb, 2500.0) + self.assertIsNone(estimated_kb) + + def test_unsupported_forge_no_limit_allows_clone(self): + """For unsupported forges with no limit configured, cloning is allowed.""" + allowed, estimated_kb = check_repo_size_limit( + "https://bitbucket.org/some/repo", max_clone_size_kb=0 + ) + self.assertTrue(allowed) + self.assertIsNone(estimated_kb) + + # ------------------------------------------------------------------ + # check_repo_size_limit — GitLab + # ------------------------------------------------------------------ @patch("httpx.get") def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): + """GitLab repo above limit — clone is blocked.""" mock_response = MagicMock() mock_response.status_code = 200 - mock_response.json.return_value = {"statistics": {"repository_size": 5242880}} # 5120 KB + mock_response.raise_for_status = MagicMock() + mock_response.json.return_value = { + "statistics": {"repository_size": 5242880} # 5120 KB + } mock_httpx_get.return_value = mock_response allowed, estimated_kb = check_repo_size_limit( @@ -78,13 +191,15 @@ def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): self.assertFalse(allowed) self.assertAlmostEqual(estimated_kb, 5120.0, places=1) - @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") - def test_api_error_fallback(self, mock_get_resource): - mock_get_resource.side_effect = Exception("API rate limit exceeded") + @patch("httpx.get") + def test_gitlab_api_error_blocks_clone(self, mock_httpx_get): + """When GitLab API fails, cloning is blocked (fail closed).""" + mock_httpx_get.side_effect = Exception("Connection refused") + allowed, estimated_kb = check_repo_size_limit( - "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 + "https://gitlab.com/group/project", max_clone_size_kb=5000 ) - self.assertTrue(allowed) + self.assertFalse(allowed) self.assertIsNone(estimated_kb) From 22037aef2a9cd56c7baa75d0561dba65b14fb0b8 Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Sun, 27 Sep 2026 19:40:42 +0530 Subject: [PATCH 8/9] fix(repofetch): correct truncation docs, add missing tests Pre-review audit fixes: - Correct _get_github_repo_size_kb docstring: GitHub's non-recursive tree API does allow full traversal via sub-tree requests; the previous claim that 'partial is all we can get' was inaccurate. Document the actual trade-off: we intentionally use partial data as a lower-bound estimate to avoid N round-trips on large repos. - Update inline comment and truncation warning message to reflect the lower-bound framing consistently. - Add test_gitlab_repo_under_limit: GitLab repo below the configured limit should allow the clone (was previously untested). - Add test_negative_limit_treated_as_disabled: negative limit values are treated the same as 0 (disabled); make this explicit in tests. - Remove unused GitCloneError import from test file. Signed-off-by: Diptesh Roy --- .../facade_worker/facade_worker/repofetch.py | 30 ++++++++----- .../test_git/test_repo_size_limit.py | 43 ++++++++++++++++--- 2 files changed, 56 insertions(+), 17 deletions(-) diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index 799ce5ab9..ce9ccd0f1 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -53,12 +53,20 @@ def _get_github_repo_size_kb(repo_git: str, logger=None): 1. The bare repo size from the repo metadata API (GitHub 'size' field, in KB) 2. The working tree blob sizes from /git/trees/HEAD?recursive=1 (in bytes) - If the tree response is truncated (too many entries for one page), the partial - blob data from the response is still used — GitHub does not paginate this endpoint, - so partial is all we can get. This is still more accurate than ignoring it. - - Returns the estimated size in KB, or raises an exception if the API is unreachable - or returns unexpected data. + When the tree response is truncated (more than 100,000 entries or 7 MB), GitHub + returns whatever entries fit and sets truncated=True. GitHub does provide a way + to get the full tree by fetching sub-trees individually via the non-recursive + endpoint, but that traversal is not implemented here — it would require multiple + round-trips for large repos and adds significant complexity. + + Instead, the partial blob data that IS returned is counted as a lower-bound + estimate. Combined with the bare repo size from the metadata API, this gives a + reasonable estimate for typical cases. If a repo is very close to the configured + limit, operators should set a smaller limit to account for this potential + undercount. + + Returns the estimated size in KB, or raises an exception if the API returns + unexpected data or the request fails. """ from collectoss.tasks.github.util.util import get_owner_repo from collectoss.tasks.github.util.github_data_access import GithubDataAccess @@ -75,9 +83,10 @@ def _get_github_repo_size_kb(repo_git: str, logger=None): bare_size_kb = repo_info.get("size", 0) or 0 # Part 2: working tree blob sizes - # Note: this endpoint is NOT paginated. truncated=True means the response was - # cut off due to size, but the entries that ARE present are still valid and - # should be counted. We use whatever partial data we received. + # GitHub's recursive tree endpoint returns up to 100,000 entries (7 MB limit). + # When truncated=True, the entries present are still valid and are counted here + # as a lower-bound estimate. Full traversal via individual sub-tree requests is + # possible but not implemented — see function docstring for the trade-off. tree_url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/HEAD?recursive=1" tree_data = github_data_access.get_resource(tree_url) if not isinstance(tree_data, dict): @@ -91,7 +100,8 @@ def _get_github_repo_size_kb(repo_git: str, logger=None): if tree_data.get("truncated", False) and logger: logger.warning( f"Git tree response for {repo_git} was truncated — " - f"file size estimate uses partial data ({file_tree_bytes} bytes counted so far)" + f"size estimate is a lower bound ({file_tree_bytes} bytes counted from partial tree data). " + f"Consider setting a smaller limit if this repo is near your threshold." ) return bare_size_kb + (file_tree_bytes / 1024) diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py index 9a9f29f42..2459f7b47 100644 --- a/tests/test_tasks/test_git/test_repo_size_limit.py +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -2,7 +2,7 @@ from unittest.mock import MagicMock, patch import collectoss.tasks.github.util.github_data_access from collectoss.tasks.git.util.facade_worker.facade_worker.repofetch import ( - check_repo_size_limit, GitCloneError, _get_github_repo_size_kb, _get_gitlab_repo_size_kb + check_repo_size_limit, _get_github_repo_size_kb, _get_gitlab_repo_size_kb ) @@ -89,12 +89,14 @@ def test_github_repo_exactly_at_limit(self, mock_get_resource): def test_github_truncated_tree_uses_partial_blob_data(self, mock_get_resource): """When the tree is truncated, partial blob data is still counted. - The GitHub git/trees API is NOT paginated for recursive requests. - truncated=True means entries were cut off, but the entries present - are still valid and should contribute to the size estimate. + GitHub's recursive tree endpoint returns up to 100,000 entries. When + truncated=True, full coverage requires traversing sub-trees individually + (not implemented here). Instead, the partial blob sizes in the response + are used as a lower-bound estimate. If the lower bound already exceeds + the limit, the clone is blocked. """ # bare: 1000 KB, partial blobs (truncated): 800,000 bytes = ~781 KB - # total: ~1781 KB > 1500 KB limit + # total: ~1781 KB > 1500 KB limit → blocked mock_get_resource.side_effect = [ {"size": 1000}, {"truncated": True, "tree": [ @@ -109,9 +111,12 @@ def test_github_truncated_tree_uses_partial_blob_data(self, mock_get_resource): @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") def test_github_truncated_tree_under_limit_still_allowed(self, mock_get_resource): - """Truncated tree with partial data that falls under the limit — still allowed. + """Truncated tree whose partial sum falls under the limit — clone is allowed. - The estimate is a lower bound when truncated, so we allow the clone. + The partial blob sum is a lower bound on the true working-tree size. When + that lower bound is still below the limit, we allow the clone. Operators + who want a tighter safety margin for repos near the limit should configure + a smaller max_clone_size_kb. """ # bare: 500 KB, partial blobs: 100,000 bytes = ~98 KB → ~598 KB < 2000 KB mock_get_resource.side_effect = [ @@ -202,6 +207,30 @@ def test_gitlab_api_error_blocks_clone(self, mock_httpx_get): self.assertFalse(allowed) self.assertIsNone(estimated_kb) + @patch("httpx.get") + def test_gitlab_repo_under_limit(self, mock_httpx_get): + """GitLab repo below limit — clone is allowed.""" + mock_response = MagicMock() + mock_response.raise_for_status = MagicMock() + mock_response.json.return_value = { + "statistics": {"repository_size": 1024000} # 1000 KB + } + mock_httpx_get.return_value = mock_response + + allowed, estimated_kb = check_repo_size_limit( + "https://gitlab.com/group/project", max_clone_size_kb=5000 + ) + self.assertTrue(allowed) + self.assertAlmostEqual(estimated_kb, 1000.0, places=1) + + def test_negative_limit_treated_as_disabled(self): + """A negative limit value is treated as disabled — clone is always allowed.""" + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=-1 + ) + self.assertTrue(allowed) + self.assertIsNone(estimated_kb) + if __name__ == "__main__": unittest.main() From 12c921170140dbf408ed13548035df8f03ff0449 Mon Sep 17 00:00:00 2001 From: Diptesh Roy Date: Mon, 28 Sep 2026 23:27:33 +0530 Subject: [PATCH 9/9] docs: document max_clone_size_kb setting and link from error message Per MoralCode's request to document this setting so operators aren't surprised when repos stop cloning. - Add max_clone_size_kb section to configuration-file-reference.rst describing default (5 GB), how to disable (set to 0), and the truncated-tree lower-bound caveat - Include the docs URL in the GitCloneError message so the error log points operators directly to the configuration reference Signed-off-by: Diptesh Roy --- .../facade_worker/facade_worker/repofetch.py | 9 +++++-- .../configuration-file-reference.rst | 27 +++++++++++++++++++ 2 files changed, 34 insertions(+), 2 deletions(-) diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index ce9ccd0f1..4724940fd 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -270,15 +270,20 @@ def git_repo_initialize(facade_helper, session, repo_git): if max_limit > 0: allowed, estimated_kb = check_repo_size_limit(git, max_limit, logger) if not allowed: + docs_ref = ( + "See max_clone_size_kb in the configuration reference: " + "https://github.com/chaoss/CollectOSS/blob/main/docs/source" + "/development-guide/configuration-file-reference.rst" + ) if estimated_kb is not None: msg = ( f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) " - f"exceeds maximum clone size limit ({max_limit} KB)" + f"exceeds maximum clone size limit ({max_limit} KB). {docs_ref}" ) else: msg = ( f"Repo '{git}' could not be size-checked; " - f"blocking clone to enforce configured limit of {max_limit} KB" + f"blocking clone to enforce configured limit of {max_limit} KB. {docs_ref}" ) update_repo_log(logger, facade_helper, row.repo_id, 'Failed (size limit)') facade_helper.log_activity('Error', msg) diff --git a/docs/source/development-guide/configuration-file-reference.rst b/docs/source/development-guide/configuration-file-reference.rst index 27fe868c2..49d155601 100644 --- a/docs/source/development-guide/configuration-file-reference.rst +++ b/docs/source/development-guide/configuration-file-reference.rst @@ -3,6 +3,33 @@ Configuration file reference CollectOSS's configuration template file, which generates your locally deployed ``augur.config.json`` file, is found at ``collectoss/config.py``. You will notice a small collection of workers are turned on to start with, by examining the ``switch`` variable within the ``Workers`` block of the config file. You can also specify the number of processes to spawn for each worker using the ``workers`` command. The default is one, and we recommend you start here. If you are going to spawn multiple workers, be sure you have enough credentials cached in the ``operations.worker_oath`` table for the platforms you use. +Facade Worker Settings +----------------------- + +The following settings live under the ``Facade`` section of the configuration. + +``max_clone_size_kb`` +~~~~~~~~~~~~~~~~~~~~~ + +**Description:** +Maximum allowed estimated repository size (in kilobytes) before CollectOSS refuses to clone it. +This prevents unexpectedly large repositories from consuming disk space or stalling the facade worker. + +**Default:** ``5242880`` (5 GB) + +**Disable:** Set to ``0`` to disable the limit entirely and clone all repositories regardless of size. + +**Notes:** + +- The size estimate combines the repository's bare size (from the forge API) with the sum of blob + sizes in the working tree. For very large repositories (over 100,000 tree entries), the tree + response may be truncated, making the estimate a lower bound. If you have repositories close to + your configured limit, set the limit conservatively to account for this. +- When a clone is blocked, an error is logged with the estimated size and the configured limit. + Check the facade worker logs if repositories are unexpectedly skipped. +- GitHub and GitLab repositories are supported. For other forges, cloning is blocked when a limit + is configured, since size cannot be determined. + If you have questions or would like to help please open an issue on GitHub_. .. _GitHub: https://github.com/chaoss/collectoss/issues