#feat: Report progress of the crawling #64

This commit is contained in:
Hari Haran
2025-04-30 04:27:22 +00:00
parent 7478f1d990
commit cd947bbdc6
2 changed files with 71 additions and 42 deletions
+12 -1
View File
@@ -56,12 +56,23 @@ class FetchRepo(Node):
) )
else: else:
print(f"Crawling directory: {prep_res['local_dir']}...") print(f"Crawling directory: {prep_res['local_dir']}...")
def progress_callback(processed, total):
percentage = (processed / total) * 100 if total > 0 else 0
rounded_percentage = int(percentage)
if rounded_percentage > progress_callback.last_reported:
progress_callback.last_reported = rounded_percentage
print(f"\033[92mProgress: {processed}/{total} files ({rounded_percentage}%)\033[0m")
progress_callback.last_reported = -1
result = crawl_local_files( result = crawl_local_files(
directory=prep_res["local_dir"], directory=prep_res["local_dir"],
include_patterns=prep_res["include_patterns"], include_patterns=prep_res["include_patterns"],
exclude_patterns=prep_res["exclude_patterns"], exclude_patterns=prep_res["exclude_patterns"],
max_file_size=prep_res["max_file_size"], max_file_size=prep_res["max_file_size"],
use_relative_paths=prep_res["use_relative_paths"] use_relative_paths=prep_res["use_relative_paths"],
progress_callback=progress_callback
) )
# Convert dict to list of tuples: [(path, content), ...] # Convert dict to list of tuples: [(path, content), ...]
+51 -33
View File
@@ -1,7 +1,7 @@
import os import os
import fnmatch import fnmatch
def crawl_local_files(directory, include_patterns=None, exclude_patterns=None, max_file_size=None, use_relative_paths=True): def crawl_local_files(directory, include_patterns=None, exclude_patterns=None, max_file_size=None, use_relative_paths=True, progress_callback=None):
""" """
Crawl files in a local directory with similar interface as crawl_github_files. Crawl files in a local directory with similar interface as crawl_github_files.
@@ -11,6 +11,7 @@ def crawl_local_files(directory, include_patterns=None, exclude_patterns=None, m
exclude_patterns (set): File patterns to exclude (e.g. {"tests/*"}) exclude_patterns (set): File patterns to exclude (e.g. {"tests/*"})
max_file_size (int): Maximum file size in bytes max_file_size (int): Maximum file size in bytes
use_relative_paths (bool): Whether to use paths relative to directory use_relative_paths (bool): Whether to use paths relative to directory
progress_callback (callable): Function to report progress, takes (processed, total) as arguments
Returns: Returns:
dict: {"files": {filepath: content}} dict: {"files": {filepath: content}}
@@ -19,48 +20,65 @@ def crawl_local_files(directory, include_patterns=None, exclude_patterns=None, m
raise ValueError(f"Directory does not exist: {directory}") raise ValueError(f"Directory does not exist: {directory}")
files_dict = {} files_dict = {}
all_files = []
# Collect all files first to calculate total
for root, _, files in os.walk(directory): for root, _, files in os.walk(directory):
for filename in files: for filename in files:
filepath = os.path.join(root, filename) filepath = os.path.join(root, filename)
all_files.append(filepath)
# Get path relative to directory if requested total_files = len(all_files)
if use_relative_paths: processed_files = 0
relpath = os.path.relpath(filepath, directory)
else:
relpath = filepath
# Check if file matches any include pattern for filepath in all_files:
included = False # Get path relative to directory if requested
if include_patterns: if use_relative_paths:
for pattern in include_patterns: relpath = os.path.relpath(filepath, directory)
if fnmatch.fnmatch(relpath, pattern): else:
included = True relpath = filepath
break
else:
included = True
# Check if file matches any exclude pattern # Check if file matches any include pattern
excluded = False included = False
if exclude_patterns: if include_patterns:
for pattern in exclude_patterns: for pattern in include_patterns:
if fnmatch.fnmatch(relpath, pattern): if fnmatch.fnmatch(relpath, pattern):
excluded = True included = True
break break
else:
included = True
if not included or excluded: # Check if file matches any exclude pattern
continue excluded = False
if exclude_patterns:
for pattern in exclude_patterns:
if fnmatch.fnmatch(relpath, pattern):
excluded = True
break
# Check file size if not included or excluded:
if max_file_size and os.path.getsize(filepath) > max_file_size: processed_files += 1
continue if progress_callback:
progress_callback(processed_files, total_files)
continue
try: # Check file size
with open(filepath, 'r', encoding='utf-8') as f: if max_file_size and os.path.getsize(filepath) > max_file_size:
content = f.read() processed_files += 1
files_dict[relpath] = content if progress_callback:
except Exception as e: progress_callback(processed_files, total_files)
print(f"Warning: Could not read file {filepath}: {e}") continue
try:
with open(filepath, 'r', encoding='utf-8') as f:
content = f.read()
files_dict[relpath] = content
except Exception as e:
print(f"Warning: Could not read file {filepath}: {e}")
processed_files += 1
if progress_callback:
progress_callback(processed_files, total_files)
return {"files": files_dict} return {"files": files_dict}