diff --git a/collectoss/application/config.py b/collectoss/application/config.py index 43bef8b49..8ac04ca02 100644 --- a/collectoss/application/config.py +++ b/collectoss/application/config.py @@ -58,6 +58,7 @@ def redact_setting_value(section_name, setting_name, value): "run_analysis": 1, "run_facade_contributors": 1, "commit_messages": 1, + "max_clone_size_kb": 5242880, }, "Server": { "cache_expire": "3600", diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py index 2b536a3a4..b7a14d8a3 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/config.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/config.py @@ -134,6 +134,7 @@ def __init__(self,logger: Logger): self.multithreaded = worker_options["multithreaded"] self.create_xlsx_summary_files = worker_options["create_xlsx_summary_files"] self.commit_messages = worker_options["commit_messages"] + self.max_clone_size_kb = int(worker_options.get("max_clone_size_kb", 5242880)) self.tool_source = "Facade" self.data_source = "Git Log" diff --git a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py index 49f7fae21..4724940fd 100644 --- a/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py +++ b/collectoss/tasks/git/util/facade_worker/facade_worker/repofetch.py @@ -25,6 +25,7 @@ # and checks for any parents of HEAD that aren't already accounted for in the # repos. It also rebuilds analysis data, checks any changed affiliations and # aliases, and caches data for display. +import logging import html.parser import subprocess import os @@ -38,9 +39,149 @@ from collectoss.application.db.lib import execute_sql, get_repo_by_repo_git from typing_extensions import deprecated +logger = logging.getLogger(__name__) + class GitCloneError(Exception): pass + +def _get_github_repo_size_kb(repo_git: str, logger=None): + """ + Fetches the estimated size of a GitHub repository in KB. + + Combines two signals: + 1. The bare repo size from the repo metadata API (GitHub 'size' field, in KB) + 2. The working tree blob sizes from /git/trees/HEAD?recursive=1 (in bytes) + + When the tree response is truncated (more than 100,000 entries or 7 MB), GitHub + returns whatever entries fit and sets truncated=True. GitHub does provide a way + to get the full tree by fetching sub-trees individually via the non-recursive + endpoint, but that traversal is not implemented here — it would require multiple + round-trips for large repos and adds significant complexity. + + Instead, the partial blob data that IS returned is counted as a lower-bound + estimate. Combined with the bare repo size from the metadata API, this gives a + reasonable estimate for typical cases. If a repo is very close to the configured + limit, operators should set a smaller limit to account for this potential + undercount. + + Returns the estimated size in KB, or raises an exception if the API returns + unexpected data or the request fails. + """ + from collectoss.tasks.github.util.util import get_owner_repo + from collectoss.tasks.github.util.github_data_access import GithubDataAccess + owner, repo = get_owner_repo(repo_git) + github_data_access = GithubDataAccess(None, logger) + + # Part 1: bare repo size + repo_url = f"https://api.github.com/repos/{owner}/{repo}" + repo_info = github_data_access.get_resource(repo_url) + if not isinstance(repo_info, dict): + raise ValueError( + f"GitHub API returned unexpected response for repo metadata: {type(repo_info)}" + ) + bare_size_kb = repo_info.get("size", 0) or 0 + + # Part 2: working tree blob sizes + # GitHub's recursive tree endpoint returns up to 100,000 entries (7 MB limit). + # When truncated=True, the entries present are still valid and are counted here + # as a lower-bound estimate. Full traversal via individual sub-tree requests is + # possible but not implemented — see function docstring for the trade-off. + tree_url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/HEAD?recursive=1" + tree_data = github_data_access.get_resource(tree_url) + if not isinstance(tree_data, dict): + raise ValueError( + f"GitHub API returned unexpected response for tree data: {type(tree_data)}" + ) + file_tree_bytes = 0 + for item in tree_data.get("tree", []): + if item.get("type") == "blob" and item.get("size") is not None: + file_tree_bytes += item["size"] + if tree_data.get("truncated", False) and logger: + logger.warning( + f"Git tree response for {repo_git} was truncated — " + f"size estimate is a lower bound ({file_tree_bytes} bytes counted from partial tree data). " + f"Consider setting a smaller limit if this repo is near your threshold." + ) + + return bare_size_kb + (file_tree_bytes / 1024) + + +def _get_gitlab_repo_size_kb(repo_git: str, logger=None): + """ + Fetches the estimated size of a GitLab repository in KB using the statistics API. + + Returns the repository size in KB, or raises an exception if the API + is unreachable or returns unexpected data. + """ + import httpx + from urllib.parse import quote_plus + git_clean = repo_git.rstrip('/') + if git_clean.endswith('.git'): + git_clean = git_clean[:-4] + parts = git_clean.split("gitlab.com/") + if len(parts) < 2: + raise ValueError(f"Could not parse GitLab project path from: {repo_git}") + project_path = parts[1] + encoded_path = quote_plus(project_path) + url = f"https://gitlab.com/api/v4/projects/{encoded_path}?statistics=true" + response = httpx.get(url, timeout=10.0) + response.raise_for_status() + data = response.json() + stats = data.get("statistics", {}) + bytes_size = stats.get("repository_size") or data.get("repository_size") + if bytes_size is None: + raise ValueError(f"No repository_size field in GitLab response for {repo_git}") + return bytes_size / 1024 + + +def check_repo_size_limit(repo_git: str, max_clone_size_kb: int, logger=None): + """ + Checks whether cloning a repository is permitted under the configured size limit. + + Retrieves the estimated repository size using forge-specific logic, then + compares it against max_clone_size_kb. + + If the size CANNOT be determined (API error, network failure, unsupported forge), + cloning is BLOCKED. This is fail-closed behavior: a configured limit must not + be bypassed silently because of an error. + + Returns: + (True, estimated_kb) — clone is permitted + (False, estimated_kb) — clone is blocked because limit is exceeded + (False, None) — clone is blocked because size could not be determined + """ + if not max_clone_size_kb or max_clone_size_kb <= 0: + return True, None + + try: + if "github.com" in repo_git.lower(): + estimated_kb = _get_github_repo_size_kb(repo_git, logger) + elif "gitlab.com" in repo_git.lower(): + estimated_kb = _get_gitlab_repo_size_kb(repo_git, logger) + else: + # Unsupported forge — cannot determine size, fail closed + if logger: + logger.warning( + f"Cannot determine repo size for unsupported forge: {repo_git}. " + f"Blocking clone to enforce configured limit of {max_clone_size_kb} KB." + ) + return False, None + + except Exception as e: + if logger: + logger.warning( + f"Could not retrieve repo size for {repo_git}: {e}. " + f"Blocking clone to enforce configured limit of {max_clone_size_kb} KB." + ) + return False, None + + if estimated_kb > max_clone_size_kb: + return False, estimated_kb + + return True, estimated_kb + + def git_repo_initialize(facade_helper, session, repo_git): # Select any new git repos so we can set up their locations and git clone @@ -125,6 +266,29 @@ def git_repo_initialize(facade_helper, session, repo_git): execute_sql(query) return + max_limit = getattr(facade_helper, 'max_clone_size_kb', 0) + if max_limit > 0: + allowed, estimated_kb = check_repo_size_limit(git, max_limit, logger) + if not allowed: + docs_ref = ( + "See max_clone_size_kb in the configuration reference: " + "https://github.com/chaoss/CollectOSS/blob/main/docs/source" + "/development-guide/configuration-file-reference.rst" + ) + if estimated_kb is not None: + msg = ( + f"Repo '{git}' estimated clone size ({estimated_kb:.0f} KB) " + f"exceeds maximum clone size limit ({max_limit} KB). {docs_ref}" + ) + else: + msg = ( + f"Repo '{git}' could not be size-checked; " + f"blocking clone to enforce configured limit of {max_limit} KB. {docs_ref}" + ) + update_repo_log(logger, facade_helper, row.repo_id, 'Failed (size limit)') + facade_helper.log_activity('Error', msg) + raise GitCloneError(msg) + # Create the prerequisite directories try: pathlib.Path(repo_path).mkdir(parents=True, exist_ok=True) diff --git a/docs/source/development-guide/configuration-file-reference.rst b/docs/source/development-guide/configuration-file-reference.rst index 27fe868c2..49d155601 100644 --- a/docs/source/development-guide/configuration-file-reference.rst +++ b/docs/source/development-guide/configuration-file-reference.rst @@ -3,6 +3,33 @@ Configuration file reference CollectOSS's configuration template file, which generates your locally deployed ``augur.config.json`` file, is found at ``collectoss/config.py``. You will notice a small collection of workers are turned on to start with, by examining the ``switch`` variable within the ``Workers`` block of the config file. You can also specify the number of processes to spawn for each worker using the ``workers`` command. The default is one, and we recommend you start here. If you are going to spawn multiple workers, be sure you have enough credentials cached in the ``operations.worker_oath`` table for the platforms you use. +Facade Worker Settings +----------------------- + +The following settings live under the ``Facade`` section of the configuration. + +``max_clone_size_kb`` +~~~~~~~~~~~~~~~~~~~~~ + +**Description:** +Maximum allowed estimated repository size (in kilobytes) before CollectOSS refuses to clone it. +This prevents unexpectedly large repositories from consuming disk space or stalling the facade worker. + +**Default:** ``5242880`` (5 GB) + +**Disable:** Set to ``0`` to disable the limit entirely and clone all repositories regardless of size. + +**Notes:** + +- The size estimate combines the repository's bare size (from the forge API) with the sum of blob + sizes in the working tree. For very large repositories (over 100,000 tree entries), the tree + response may be truncated, making the estimate a lower bound. If you have repositories close to + your configured limit, set the limit conservatively to account for this. +- When a clone is blocked, an error is logged with the estimated size and the configured limit. + Check the facade worker logs if repositories are unexpectedly skipped. +- GitHub and GitLab repositories are supported. For other forges, cloning is blocked when a limit + is configured, since size cannot be determined. + If you have questions or would like to help please open an issue on GitHub_. .. _GitHub: https://github.com/chaoss/collectoss/issues diff --git a/tests/test_tasks/test_git/test_repo_size_limit.py b/tests/test_tasks/test_git/test_repo_size_limit.py new file mode 100644 index 000000000..2459f7b47 --- /dev/null +++ b/tests/test_tasks/test_git/test_repo_size_limit.py @@ -0,0 +1,236 @@ +import unittest +from unittest.mock import MagicMock, patch +import collectoss.tasks.github.util.github_data_access +from collectoss.tasks.git.util.facade_worker.facade_worker.repofetch import ( + check_repo_size_limit, _get_github_repo_size_kb, _get_gitlab_repo_size_kb +) + + +class TestRepoSizeLimit(unittest.TestCase): + + def setUp(self): + self.github_access_patcher = patch( + "collectoss.tasks.github.util.github_data_access.GithubDataAccess.__init__", + return_value=None + ) + self.mock_github_init = self.github_access_patcher.start() + + def tearDown(self): + self.github_access_patcher.stop() + + # ------------------------------------------------------------------ + # check_repo_size_limit — limit disabled + # ------------------------------------------------------------------ + + def test_size_limit_disabled(self): + """When limit is 0 (disabled), always allow regardless of repo size.""" + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", 0 + ) + self.assertTrue(allowed) + self.assertIsNone(estimated_kb) + + # ------------------------------------------------------------------ + # check_repo_size_limit — GitHub, normal (non-truncated) tree + # ------------------------------------------------------------------ + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_repo_under_limit(self, mock_get_resource): + """Repo comfortably below limit — clone is allowed.""" + # bare: 1000 KB, blobs: 500,000 bytes = ~488 KB → ~1488 KB total + mock_get_resource.side_effect = [ + {"size": 1000}, + {"truncated": False, "tree": [ + {"type": "blob", "size": 300000}, + {"type": "blob", "size": 200000}, + {"type": "tree", "size": None}, # directories have no size + ]} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 + ) + self.assertTrue(allowed) + self.assertAlmostEqual(estimated_kb, 1000 + (500000 / 1024), places=1) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_repo_exceeds_limit(self, mock_get_resource): + """Repo above limit — clone is blocked.""" + # bare: 1000 KB, blobs: 1,500,000 bytes = ~1465 KB → ~2465 KB total > 2000 KB + mock_get_resource.side_effect = [ + {"size": 1000}, + {"truncated": False, "tree": [ + {"type": "blob", "size": 1500000}, + ]} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 + ) + self.assertFalse(allowed) + self.assertAlmostEqual(estimated_kb, 1000 + (1500000 / 1024), places=1) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_repo_exactly_at_limit(self, mock_get_resource): + """Repo exactly at the limit — clone is allowed (limit is exclusive upper bound).""" + mock_get_resource.side_effect = [ + {"size": 2000}, + {"truncated": False, "tree": []} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 + ) + self.assertTrue(allowed) + self.assertEqual(estimated_kb, 2000.0) + + # ------------------------------------------------------------------ + # check_repo_size_limit — GitHub, truncated tree + # ------------------------------------------------------------------ + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_truncated_tree_uses_partial_blob_data(self, mock_get_resource): + """When the tree is truncated, partial blob data is still counted. + + GitHub's recursive tree endpoint returns up to 100,000 entries. When + truncated=True, full coverage requires traversing sub-trees individually + (not implemented here). Instead, the partial blob sizes in the response + are used as a lower-bound estimate. If the lower bound already exceeds + the limit, the clone is blocked. + """ + # bare: 1000 KB, partial blobs (truncated): 800,000 bytes = ~781 KB + # total: ~1781 KB > 1500 KB limit → blocked + mock_get_resource.side_effect = [ + {"size": 1000}, + {"truncated": True, "tree": [ + {"type": "blob", "size": 800000}, + ]} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1500 + ) + self.assertFalse(allowed) + self.assertAlmostEqual(estimated_kb, 1000 + (800000 / 1024), places=1) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_github_truncated_tree_under_limit_still_allowed(self, mock_get_resource): + """Truncated tree whose partial sum falls under the limit — clone is allowed. + + The partial blob sum is a lower bound on the true working-tree size. When + that lower bound is still below the limit, we allow the clone. Operators + who want a tighter safety margin for repos near the limit should configure + a smaller max_clone_size_kb. + """ + # bare: 500 KB, partial blobs: 100,000 bytes = ~98 KB → ~598 KB < 2000 KB + mock_get_resource.side_effect = [ + {"size": 500}, + {"truncated": True, "tree": [ + {"type": "blob", "size": 100000}, + ]} + ] + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=2000 + ) + self.assertTrue(allowed) + self.assertAlmostEqual(estimated_kb, 500 + (100000 / 1024), places=1) + + # ------------------------------------------------------------------ + # check_repo_size_limit — error handling (fail closed) + # ------------------------------------------------------------------ + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_api_error_blocks_clone(self, mock_get_resource): + """When the API call fails, cloning is BLOCKED (fail closed). + + A configured limit must not be silently bypassed because of a + network failure or API error. + """ + mock_get_resource.side_effect = Exception("API rate limit exceeded") + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 + ) + self.assertFalse(allowed) + self.assertIsNone(estimated_kb) + + @patch("collectoss.tasks.github.util.github_data_access.GithubDataAccess.get_resource") + def test_malformed_api_response_blocks_clone(self, mock_get_resource): + """When the API returns a non-dict, _get_github_repo_size_kb raises, blocking the clone.""" + mock_get_resource.return_value = "not a dict" + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=1000 + ) + self.assertFalse(allowed) + self.assertIsNone(estimated_kb) + + def test_unsupported_forge_blocks_clone(self): + """For unsupported forges, cloning is blocked when a limit is configured.""" + allowed, estimated_kb = check_repo_size_limit( + "https://bitbucket.org/some/repo", max_clone_size_kb=1000 + ) + self.assertFalse(allowed) + self.assertIsNone(estimated_kb) + + def test_unsupported_forge_no_limit_allows_clone(self): + """For unsupported forges with no limit configured, cloning is allowed.""" + allowed, estimated_kb = check_repo_size_limit( + "https://bitbucket.org/some/repo", max_clone_size_kb=0 + ) + self.assertTrue(allowed) + self.assertIsNone(estimated_kb) + + # ------------------------------------------------------------------ + # check_repo_size_limit — GitLab + # ------------------------------------------------------------------ + + @patch("httpx.get") + def test_gitlab_repo_exceeds_limit(self, mock_httpx_get): + """GitLab repo above limit — clone is blocked.""" + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.raise_for_status = MagicMock() + mock_response.json.return_value = { + "statistics": {"repository_size": 5242880} # 5120 KB + } + mock_httpx_get.return_value = mock_response + + allowed, estimated_kb = check_repo_size_limit( + "https://gitlab.com/group/project", max_clone_size_kb=5000 + ) + self.assertFalse(allowed) + self.assertAlmostEqual(estimated_kb, 5120.0, places=1) + + @patch("httpx.get") + def test_gitlab_api_error_blocks_clone(self, mock_httpx_get): + """When GitLab API fails, cloning is blocked (fail closed).""" + mock_httpx_get.side_effect = Exception("Connection refused") + + allowed, estimated_kb = check_repo_size_limit( + "https://gitlab.com/group/project", max_clone_size_kb=5000 + ) + self.assertFalse(allowed) + self.assertIsNone(estimated_kb) + + @patch("httpx.get") + def test_gitlab_repo_under_limit(self, mock_httpx_get): + """GitLab repo below limit — clone is allowed.""" + mock_response = MagicMock() + mock_response.raise_for_status = MagicMock() + mock_response.json.return_value = { + "statistics": {"repository_size": 1024000} # 1000 KB + } + mock_httpx_get.return_value = mock_response + + allowed, estimated_kb = check_repo_size_limit( + "https://gitlab.com/group/project", max_clone_size_kb=5000 + ) + self.assertTrue(allowed) + self.assertAlmostEqual(estimated_kb, 1000.0, places=1) + + def test_negative_limit_treated_as_disabled(self): + """A negative limit value is treated as disabled — clone is always allowed.""" + allowed, estimated_kb = check_repo_size_limit( + "https://github.com/chaoss/CollectOSS", max_clone_size_kb=-1 + ) + self.assertTrue(allowed) + self.assertIsNone(estimated_kb) + + +if __name__ == "__main__": + unittest.main()