mirror of
https://github.com/The-Pocket/PocketFlow-Tutorial-Codebase-Knowledge.git
synced 2026-08-29 16:40:32 +08:00
add timeout to requests and optimize directory traversal
- Added timeout=(30, 30) to requests.get for improved network robustness - Avoid unnecessary recursion into excluded directories to reduce API calls
This commit is contained in:
@@ -144,7 +144,7 @@ def crawl_github_files(
|
|||||||
"""Get brancshes of the repository"""
|
"""Get brancshes of the repository"""
|
||||||
|
|
||||||
url = f"https://api.github.com/repos/{owner}/{repo}/branches"
|
url = f"https://api.github.com/repos/{owner}/{repo}/branches"
|
||||||
response = requests.get(url, headers=headers)
|
response = requests.get(url, headers=headers, timeout=(30, 30))
|
||||||
|
|
||||||
if response.status_code == 404:
|
if response.status_code == 404:
|
||||||
if not token:
|
if not token:
|
||||||
@@ -165,7 +165,7 @@ def crawl_github_files(
|
|||||||
"""Check the repository has the given tree"""
|
"""Check the repository has the given tree"""
|
||||||
|
|
||||||
url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/{tree}"
|
url = f"https://api.github.com/repos/{owner}/{repo}/git/trees/{tree}"
|
||||||
response = requests.get(url, headers=headers)
|
response = requests.get(url, headers=headers, timeout=(30, 30))
|
||||||
|
|
||||||
return True if response.status_code == 200 else False
|
return True if response.status_code == 200 else False
|
||||||
|
|
||||||
@@ -216,7 +216,7 @@ def crawl_github_files(
|
|||||||
url = f"https://api.github.com/repos/{owner}/{repo}/contents/{path}"
|
url = f"https://api.github.com/repos/{owner}/{repo}/contents/{path}"
|
||||||
params = {"ref": ref} if ref != None else {}
|
params = {"ref": ref} if ref != None else {}
|
||||||
|
|
||||||
response = requests.get(url, headers=headers, params=params)
|
response = requests.get(url, headers=headers, params=params, timeout=(30, 30))
|
||||||
|
|
||||||
if response.status_code == 403 and 'rate limit exceeded' in response.text.lower():
|
if response.status_code == 403 and 'rate limit exceeded' in response.text.lower():
|
||||||
reset_time = int(response.headers.get('X-RateLimit-Reset', 0))
|
reset_time = int(response.headers.get('X-RateLimit-Reset', 0))
|
||||||
@@ -276,7 +276,7 @@ def crawl_github_files(
|
|||||||
# For files, get raw content
|
# For files, get raw content
|
||||||
if "download_url" in item and item["download_url"]:
|
if "download_url" in item and item["download_url"]:
|
||||||
file_url = item["download_url"]
|
file_url = item["download_url"]
|
||||||
file_response = requests.get(file_url, headers=headers)
|
file_response = requests.get(file_url, headers=headers, timeout=(30, 30))
|
||||||
|
|
||||||
# Final size check in case content-length header is available but differs from metadata
|
# Final size check in case content-length header is available but differs from metadata
|
||||||
content_length = int(file_response.headers.get('content-length', 0))
|
content_length = int(file_response.headers.get('content-length', 0))
|
||||||
@@ -292,7 +292,7 @@ def crawl_github_files(
|
|||||||
print(f"Failed to download {rel_path}: {file_response.status_code}")
|
print(f"Failed to download {rel_path}: {file_response.status_code}")
|
||||||
else:
|
else:
|
||||||
# Alternative method if download_url is not available
|
# Alternative method if download_url is not available
|
||||||
content_response = requests.get(item["url"], headers=headers)
|
content_response = requests.get(item["url"], headers=headers, timeout=(30, 30))
|
||||||
if content_response.status_code == 200:
|
if content_response.status_code == 200:
|
||||||
content_data = content_response.json()
|
content_data = content_response.json()
|
||||||
if content_data.get("encoding") == "base64" and "content" in content_data:
|
if content_data.get("encoding") == "base64" and "content" in content_data:
|
||||||
@@ -312,7 +312,19 @@ def crawl_github_files(
|
|||||||
print(f"Failed to get content for {rel_path}: {content_response.status_code}")
|
print(f"Failed to get content for {rel_path}: {content_response.status_code}")
|
||||||
|
|
||||||
elif item["type"] == "dir":
|
elif item["type"] == "dir":
|
||||||
# Recursively process subdirectories
|
# OLD IMPLEMENTATION (comment this block to test new implementation)
|
||||||
|
# Always recurse into directories without checking exclusions first
|
||||||
|
# fetch_contents(item_path)
|
||||||
|
|
||||||
|
# NEW IMPLEMENTATION (uncomment this block to test optimized version)
|
||||||
|
# # Check if directory should be excluded before recursing
|
||||||
|
if exclude_patterns:
|
||||||
|
dir_excluded = any(fnmatch.fnmatch(item_path, pattern) or
|
||||||
|
fnmatch.fnmatch(rel_path, pattern) for pattern in exclude_patterns)
|
||||||
|
if dir_excluded:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# # Only recurse if directory is not excluded
|
||||||
fetch_contents(item_path)
|
fetch_contents(item_path)
|
||||||
|
|
||||||
# Start crawling from the specified path
|
# Start crawling from the specified path
|
||||||
|
|||||||
Reference in New Issue
Block a user