From 0bea592b77e9f3f2ae6998b80f76588328e1ceb8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?SAL=C4=B0H=20=C3=96ZKARA?= Date: Tue, 14 Oct 2025 16:28:49 +0300 Subject: [PATCH] Refactor SEO description script to standalone Python file Moved the SEO description generation logic from an inline script in the GitHub Actions workflow to a dedicated Python script at .github/scripts/add_seo_descriptions.py. Updated the workflow to call this script directly, improving maintainability and readability. --- .github/scripts/add_seo_descriptions.py | 178 ++++++++++++++++++++++++ .github/workflows/auto-add-seo.yml | 177 +---------------------- 2 files changed, 180 insertions(+), 175 deletions(-) create mode 100644 .github/scripts/add_seo_descriptions.py diff --git a/.github/scripts/add_seo_descriptions.py b/.github/scripts/add_seo_descriptions.py new file mode 100644 index 0000000000..08ef581965 --- /dev/null +++ b/.github/scripts/add_seo_descriptions.py @@ -0,0 +1,178 @@ +import os +import sys +import re +from openai import OpenAI + +client = OpenAI(api_key=os.environ['OPENAI_API_KEY']) + +def has_seo_description(content): + """Check if content already has SEO description""" + # Match SEO description block with 3 or more backticks + pattern = r'```+json\s*//\[doc-seo\].*?```+' + return re.search(pattern, content, flags=re.DOTALL) is not None + +def is_content_too_short(content): + """Check if content is less than 200 characters""" + # Remove SEO tags if present for accurate count + # Match SEO description block with 3 or more backticks + clean_content = re.sub(r'```+json\s*//\[doc-seo\].*?```+\s*', '', content, flags=re.DOTALL) + + return len(clean_content.strip()) < 200 + +def get_content_preview(content, max_length=1000): + """Get preview of content for OpenAI""" + # Remove existing SEO tags if present + # Match SEO description block with 3 or more backticks + clean_content = re.sub(r'```+json\s*//\[doc-seo\].*?```+\s*', '', content, flags=re.DOTALL) + + return clean_content[:max_length].strip() + +def generate_description(content, filename): + """Generate SEO description using OpenAI with system prompt from OpenAIService.cs""" + try: + preview = get_content_preview(content) + + response = client.chat.completions.create( + model="gpt-4o-mini", + messages=[ + {"role": "system", "content": """Create a short and engaging summary (1–2 sentences) for sharing this documentation link on Discord, LinkedIn, Reddit, Twitter and Facebook. Clearly describe what the page explains or teaches. +Highlight the value for developers using ABP Framework. +Be written in a friendly and professional tone. +Stay under 150 characters. +--> https://abp.io/docs/latest <--"""}, + {"role": "user", "content": f"""Generate a concise, informative meta description for this documentation page. + +File: {filename} +Content Preview: +{preview} + +Requirements: +- Maximum 150 characters + +Generate only the description text, nothing else:"""} + ], + max_tokens=150, + temperature=0.7 + ) + + description = response.choices[0].message.content.strip() + + return description + except Exception as e: + print(f"āŒ Error generating description: {e}") + return f"Learn about {os.path.splitext(filename)[0]} in ABP Framework documentation." + +def add_seo_description(content, description): + """Add SEO description to content""" + # Escape special characters for JSON + escaped_desc = description.replace('\\', '\\\\').replace('"', '\\"').replace('\n', '\\n') + + seo_tag = f'''```json +//[doc-seo] +{{ + "Description": "{escaped_desc}" +}} +``` + +''' + return seo_tag + content + +def is_file_ignored(filepath, ignored_folders): + """Check if file is in an ignored folder""" + path_parts = filepath.split('/') + for ignored in ignored_folders: + if ignored in path_parts: + return True + return False + +def main(): + # Ignored folders from GitHub variable (or default values) + IGNORED_FOLDERS_STR = os.environ.get('IGNORED_FOLDERS', 'Blog-Posts,Community-Articles,_deleted,_resources') + IGNORED_FOLDERS = [folder.strip() for folder in IGNORED_FOLDERS_STR.split(',') if folder.strip()] + + # Get changed files from environment or command line + if len(sys.argv) > 1: + # Files passed as command line arguments + changed_files = sys.argv[1:] + else: + # Files from environment variable (for GitHub Actions) + changed_files_str = os.environ.get('CHANGED_FILES', '') + changed_files = [f.strip() for f in changed_files_str.strip().split('\n') if f.strip()] + + processed_count = 0 + skipped_count = 0 + skipped_too_short = 0 + skipped_ignored = 0 + updated_files = [] # Track actually updated files + + print("šŸ¤– Processing changed markdown files...\n") + print(f"🚫 Ignored folders: {', '.join(IGNORED_FOLDERS)}\n") + + for filepath in changed_files: + if not filepath.endswith('.md'): + continue + + # Check if file is in ignored folder + if is_file_ignored(filepath, IGNORED_FOLDERS): + print(f"šŸ“„ Processing: {filepath}") + print(f" 🚫 Skipped (ignored folder)\n") + skipped_ignored += 1 + skipped_count += 1 + continue + + print(f"šŸ“„ Processing: {filepath}") + + try: + # Read file + with open(filepath, 'r', encoding='utf-8') as f: + content = f.read() + + # Check if content is too short (less than 200 characters) + if is_content_too_short(content): + print(f" ā­ļø Skipped (content less than 200 characters)\n") + skipped_too_short += 1 + skipped_count += 1 + continue + + # Check if already has SEO description + if has_seo_description(content): + print(f" ā­ļø Skipped (already has SEO description)\n") + skipped_count += 1 + continue + + # Generate description + filename = os.path.basename(filepath) + print(f" šŸ¤– Generating description...") + description = generate_description(content, filename) + print(f" šŸ’” Generated: {description}") + + # Add SEO tag + updated_content = add_seo_description(content, description) + + # Write back + with open(filepath, 'w', encoding='utf-8') as f: + f.write(updated_content) + + print(f" āœ… Updated successfully\n") + processed_count += 1 + updated_files.append(filepath) # Track this file as updated + + except Exception as e: + print(f" āŒ Error: {e}\n") + + print(f"\nšŸ“Š Summary:") + print(f" āœ… Updated: {processed_count}") + print(f" ā­ļø Skipped (total): {skipped_count}") + print(f" ā­ļø Skipped (too short): {skipped_too_short}") + print(f" 🚫 Skipped (ignored folder): {skipped_ignored}") + + # Save counts and updated files list for next step + with open('/tmp/seo_stats.txt', 'w') as f: + f.write(f"{processed_count}\n{skipped_count}\n{skipped_too_short}\n{skipped_ignored}") + + # Save updated files list + with open('/tmp/seo_updated_files.txt', 'w') as f: + f.write('\n'.join(updated_files)) + +if __name__ == '__main__': + main() diff --git a/.github/workflows/auto-add-seo.yml b/.github/workflows/auto-add-seo.yml index 97edec5e76..8440585151 100644 --- a/.github/workflows/auto-add-seo.yml +++ b/.github/workflows/auto-add-seo.yml @@ -92,182 +92,9 @@ jobs: env: OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} IGNORED_FOLDERS: ${{ vars.DOCS_SEO_IGNORED_FOLDERS }} + CHANGED_FILES: ${{ steps.changed-files.outputs.changed_files }} run: | - python3 << 'PYTHON_SCRIPT' - import os - import sys - import re - from openai import OpenAI - - client = OpenAI(api_key=os.environ['OPENAI_API_KEY']) - - def has_seo_description(content): - """Check if content already has SEO description""" - return content.strip().startswith('```json') and '//[doc-seo]' in content - - def is_content_too_short(content): - """Check if content is less than 200 characters""" - # Remove SEO tags if present for accurate count - clean_content = content - if '//[doc-seo]' in clean_content: - parts = clean_content.split('```', 2) - if len(parts) > 2: - clean_content = parts[2].strip() - - return len(clean_content.strip()) < 200 - - def get_content_preview(content, max_length=1000): - """Get preview of content for OpenAI""" - # Remove existing SEO tags if present - clean_content = content - if '//[doc-seo]' in clean_content: - parts = clean_content.split('```', 2) - if len(parts) > 2: - clean_content = parts[2].strip() - - return clean_content[:max_length].strip() - - def generate_description(content, filename): - """Generate SEO description using OpenAI with system prompt from OpenAIService.cs""" - try: - preview = get_content_preview(content) - - response = client.chat.completions.create( - model="gpt-4o-mini", - messages=[ - {"role": "system", "content": """Create a short and engaging summary (1–2 sentences) for sharing this documentation link on Discord, LinkedIn, Reddit, Twitter and Facebook. Clearly describe what the page explains or teaches. - Highlight the value for developers using ABP Framework. - Be written in a friendly and professional tone. - Stay under 150 characters. - --> https://abp.io/docs/latest <--"""}, - {"role": "user", "content": f"""Generate a concise, informative meta description for this documentation page. - - File: {filename} - Content Preview: - {preview} - - Requirements: - - Maximum 150 characters - - Generate only the description text, nothing else:"""} - ], - max_tokens=150, - temperature=0.7 - ) - - description = response.choices[0].message.content.strip() - - return description - except Exception as e: - print(f"āŒ Error generating description: {e}") - return f"Learn about {os.path.splitext(filename)[0]} in ABP Framework documentation." - - def add_seo_description(content, description): - """Add SEO description to content""" - # Escape special characters for JSON - escaped_desc = description.replace('\\', '\\\\').replace('"', '\\"').replace('\n', '\\n') - - seo_tag = f'''```json - //[doc-seo] - {{ - "Description": "{escaped_desc}" - }} - ``` - - ''' - return seo_tag + content - - # Ignored folders from GitHub variable (or default values) - IGNORED_FOLDERS_STR = os.environ.get('IGNORED_FOLDERS', 'Blog-Posts,Community-Articles,_deleted,_resources') - IGNORED_FOLDERS = [folder.strip() for folder in IGNORED_FOLDERS_STR.split(',') if folder.strip()] - - def is_file_ignored(filepath): - """Check if file is in an ignored folder""" - path_parts = filepath.split('/') - for ignored in IGNORED_FOLDERS: - if ignored in path_parts: - return True - return False - - # Process changed files - changed_files_str = """${{ steps.changed-files.outputs.changed_files }}""" - changed_files = [f.strip() for f in changed_files_str.strip().split('\n') if f.strip()] - processed_count = 0 - skipped_count = 0 - skipped_too_short = 0 - skipped_ignored = 0 - updated_files = [] # Track actually updated files - - print("šŸ¤– Processing changed markdown files...\n") - print(f"🚫 Ignored folders: {', '.join(IGNORED_FOLDERS)}\n") - - for filepath in changed_files: - if not filepath.endswith('.md'): - continue - - # Check if file is in ignored folder - if is_file_ignored(filepath): - print(f"šŸ“„ Processing: {filepath}") - print(f" 🚫 Skipped (ignored folder)\n") - skipped_ignored += 1 - skipped_count += 1 - continue - - print(f"šŸ“„ Processing: {filepath}") - - try: - # Read file - with open(filepath, 'r', encoding='utf-8') as f: - content = f.read() - - # Check if content is too short (less than 200 characters) - if is_content_too_short(content): - print(f" ā­ļø Skipped (content less than 200 characters)\n") - skipped_too_short += 1 - skipped_count += 1 - continue - - # Check if already has SEO description - if has_seo_description(content): - print(f" ā­ļø Skipped (already has SEO description)\n") - skipped_count += 1 - continue - - # Generate description - filename = os.path.basename(filepath) - print(f" šŸ¤– Generating description...") - description = generate_description(content, filename) - print(f" šŸ’” Generated: {description}") - - # Add SEO tag - updated_content = add_seo_description(content, description) - - # Write back - with open(filepath, 'w', encoding='utf-8') as f: - f.write(updated_content) - - print(f" āœ… Updated successfully\n") - processed_count += 1 - updated_files.append(filepath) # Track this file as updated - - except Exception as e: - print(f" āŒ Error: {e}\n") - - print(f"\nšŸ“Š Summary:") - print(f" āœ… Updated: {processed_count}") - print(f" ā­ļø Skipped (total): {skipped_count}") - print(f" ā­ļø Skipped (too short): {skipped_too_short}") - print(f" 🚫 Skipped (ignored folder): {skipped_ignored}") - - # Save counts and updated files list for next step - with open('/tmp/seo_stats.txt', 'w') as f: - f.write(f"{processed_count}\n{skipped_count}\n{skipped_too_short}\n{skipped_ignored}") - - # Save updated files list - with open('/tmp/seo_updated_files.txt', 'w') as f: - f.write('\n'.join(updated_files)) - - PYTHON_SCRIPT + python3 .github/scripts/add_seo_descriptions.py - name: Commit and push changes