Python自动化脚本实战:提升工作效率的10个实用案例

Python是自动化脚本的首选语言。本文提供10个实用的自动化脚本案例,帮助你提升工作效率,减少重复劳动。

1. 批量文件重命名

#!/usr/bin/env python3
import os
import argparse

def batch_rename(directory, prefix, extension=None):
    """批量重命名文件"""
    files = sorted(os.listdir(directory))
    for i, filename in enumerate(files, 1):
        old_path = os.path.join(directory, filename)
        if os.path.isfile(old_path):
            ext = extension or os.path.splitext(filename)[1]
            new_name = f"{prefix}_{i:03d}{ext}"
            new_path = os.path.join(directory, new_name)
            os.rename(old_path, new_path)
            print(f"Renamed: {filename} -> {new_name}")

if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("dir", help="Directory containing files")
    parser.add_argument("prefix", help="Filename prefix")
    parser.add_argument("--ext", help="File extension", default=None)
    args = parser.parse_args()
    batch_rename(args.dir, args.prefix, args.ext)

2. 自动备份重要文件

#!/usr/bin/env python3
import shutil
import os
from datetime import datetime
import tarfile

def backup_files(source_dirs, backup_dir):
    """备份文件到指定目录"""
    timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
    backup_file = os.path.join(backup_dir, f"backup_{timestamp}.tar.gz")
    
    with tarfile.open(backup_file, "w:gz") as tar:
        for source in source_dirs:
            if os.path.exists(source):
                tar.add(source, arcname=os.path.basename(source))
                print(f"Added: {source}")
    
    print(f"Backup created: {backup_file}")

if __name__ == "__main__":
    sources = ["/home/user/documents", "/home/user/photos"]
    backup_dir = "/backup"
    backup_files(sources, backup_dir)

3. 网页内容抓取

#!/usr/bin/env python3
import requests
from bs4 import BeautifulSoup
import csv

def scrape_articles(url, output_file):
    """抓取文章标题和链接"""
    headers = {
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
    }
    
    response = requests.get(url, headers=headers)
    soup = BeautifulSoup(response.text, "html.parser")
    
    articles = []
    for item in soup.select(".article-item"):
        title = item.select_one("h2").get_text(strip=True)
        link = item.select_one("a")["href"]
        articles.append({"title": title, "link": link})
    
    # 保存到CSV
    with open(output_file, "w", newline="", encoding="utf-8") as f:
        writer = csv.DictWriter(f, fieldnames=["title", "link"])
        writer.writeheader()
        writer.writerows(articles)
    
    print(f"Scraped {len(articles)} articles")

if __name__ == "__main__":
    scrape_articles("https://example.com/news", "articles.csv")

4. 自动发送邮件

#!/usr/bin/env python3
import smtplib
from email.mime.text import MIMEText
from email.mime.multipart import MIMEMultipart
import os

def send_email(to_email, subject, body, html=False):
    """发送邮件"""
    smtp_server = os.getenv("SMTP_SERVER", "smtp.gmail.com")
    smtp_port = int(os.getenv("SMTP_PORT", 587))
    username = os.getenv("EMAIL_USER")
    password = os.getenv("EMAIL_PASS")
    
    msg = MIMEMultipart()
    msg["From"] = username
    msg["To"] = to_email
    msg["Subject"] = subject
    
    content_type = "html" if html else "plain"
    msg.attach(MIMEText(body, content_type))
    
    with smtplib.SMTP(smtp_server, smtp_port) as server:
        server.starttls()
        server.login(username, password)
        server.send_message(msg)
    
    print(f"Email sent to {to_email}")

if __name__ == "__main__":
    send_email(
        "[email protected]",
        "Daily Report",
        "

Report

Daily summary...

", html=True )

5. 监控网站可用性

#!/usr/bin/env python3
import requests
import time
import logging
from datetime import datetime

def monitor_websites(urls, check_interval=300):
    """监控网站可用性"""
    logging.basicConfig(
        filename="monitor.log",
        level=logging.INFO,
        format="%(asctime)s - %(levelname)s - %(message)s"
    )
    
    while True:
        for url in urls:
            try:
                start = time.time()
                response = requests.get(url, timeout=10)
                elapsed = time.time() - start
                
                if response.status_code == 200:
                    print(f"OK {url} - {elapsed:.2f}s")
                    logging.info(f"UP {url} - {elapsed:.2f}s")
                else:
                    print(f"FAIL {url} - Status {response.status_code}")
                    logging.warning(f"DOWN {url} - Status {response.status_code}")
            except Exception as e:
                print(f"ERROR {url} - Error: {e}")
                logging.error(f"ERROR {url} - {e}")
        
        time.sleep(check_interval)

if __name__ == "__main__":
    urls = ["https://example.com", "https://api.example.com"]
    monitor_websites(urls)

6. 自动整理下载文件夹

#!/usr/bin/env python3
import os
import shutil
from pathlib import Path

def organize_downloads(download_dir):
    """按扩展名整理下载文件"""
    categories = {
        "images": [".jpg", ".jpeg", ".png", ".gif", ".bmp", ".svg"],
        "documents": [".pdf", ".doc", ".docx", ".txt", ".xls", ".xlsx", ".ppt", ".pptx"],
        "archives": [".zip", ".rar", ".7z", ".tar", ".gz"],
        "videos": [".mp4", ".avi", ".mkv", ".mov"],
        "music": [".mp3", ".wav", ".flac", ".aac"],
        "scripts": [".py", ".js", ".sh", ".bat", ".ps1"],
    }
    
    download_path = Path(download_dir)
    
    for file_path in download_path.iterdir():
        if file_path.is_file():
            ext = file_path.suffix.lower()
            
            for category, extensions in categories.items():
                if ext in extensions:
                    dest_dir = download_path / category
                    dest_dir.mkdir(exist_ok=True)
                    dest_file = dest_dir / file_path.name
                    
                    # 处理重名
                    counter = 1
                    while dest_file.exists():
                        dest_file = dest_dir / f"{file_path.stem}_{counter}{ext}"
                        counter += 1
                    
                    shutil.move(str(file_path), str(dest_file))
                    print(f"Moved: {file_path.name} -> {category}/")
                    break

if __name__ == "__main__":
    organize_downloads(os.path.expanduser("~/Downloads"))

7. 日志分析工具

#!/usr/bin/env python3
import re
from collections import Counter
from datetime import datetime

def analyze_log(log_file):
    """分析日志文件,统计状态码和IP"""
    status_pattern = re.compile(r'" (\d{3}) ')
    ip_pattern = re.compile(r'^(\d+\.\d+\.\d+\.\d+)')
    
    status_counts = Counter()
    ip_counts = Counter()
    total_requests = 0
    
    with open(log_file, "r") as f:
        for line in f:
            total_requests += 1
            
            # 统计状态码
            status_match = status_pattern.search(line)
            if status_match:
                status = status_match.group(1)
                status_counts[status] += 1
            
            # 统计IP
            ip_match = ip_pattern.match(line)
            if ip_match:
                ip = ip_match.group(1)
                ip_counts[ip] += 1
    
    # 输出报告
    print(f"Total requests: {total_requests}")
    print("\nStatus codes:")
    for code, count in status_counts.most_common():
        percentage = (count / total_requests) * 100
        print(f"  {code}: {count} ({percentage:.1f}%)")
    
    print("\nTop 10 IPs:")
    for ip, count in ip_counts.most_common(10):
        print(f"  {ip}: {count}")

if __name__ == "__main__":
    analyze_log("/var/log/nginx/access.log")

8. 数据库自动备份

#!/usr/bin/env python3
import os
import subprocess
from datetime import datetime

def backup_mysql(host, user, password, database, backup_dir):
    """MySQL数据库备份"""
    timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
    backup_file = os.path.join(backup_dir, f"{database}_{timestamp}.sql")
    
    cmd = [
        "mysqldump",
        "-h", host,
        "-u", user,
        f"-p{password}",
        "--single-transaction",
        "--routines",
        "--triggers",
        database
    ]
    
    with open(backup_file, "w") as f:
        subprocess.run(cmd, stdout=f, check=True)
    
    print(f"Backup created: {backup_file}")

def backup_postgresql(db_name, backup_dir):
    """PostgreSQL数据库备份"""
    timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
    backup_file = os.path.join(backup_dir, f"{db_name}_{timestamp}.sql")
    
    env = os.environ.copy()
    env["PGPASSWORD"] = "your_password"
    
    cmd = [
        "pg_dump",
        "-U", "postgres",
        "-F", "c",
        "-f", backup_file,
        db_name
    ]
    
    subprocess.run(cmd, env=env, check=True)
    print(f"Backup created: {backup_file}")

if __name__ == "__main__":
    backup_mysql("localhost", "root", "password", "mydb", "/backup/mysql")

9. 自动部署脚本

#!/usr/bin/env python3
import subprocess
import sys

def deploy():
    """自动部署脚本"""
    commands = [
        ("Pulling latest code", "git pull origin main"),
        ("Installing dependencies", "pip install -r requirements.txt"),
        ("Running migrations", "python manage.py migrate"),
        ("Collecting static files", "python manage.py collectstatic --noinput"),
        ("Restarting service", "systemctl restart gunicorn"),
        ("Clearing cache", "redis-cli FLUSHDB"),
    ]
    
    for step, cmd in commands:
        print(f"\n{'='*50}")
        print(f"Step: {step}")
        print(f"Command: {cmd}")
        print('='*50)
        
        result = subprocess.run(cmd, shell=True, capture_output=True, text=True)
        
        if result.returncode != 0:
            print(f"ERROR: {result.stderr}")
            sys.exit(1)
        
        print(result.stdout)
    
    print("\nDeployment completed successfully!")

if __name__ == "__main__":
    deploy()

10. 系统资源监控

#!/usr/bin/env python3
import psutil
import time
import json
from datetime import datetime

def monitor_resources(interval=60):
    """监控系统资源使用情况"""
    while True:
        timestamp = datetime.now().isoformat()
        
        stats = {
            "timestamp": timestamp,
            "cpu": {
                "percent": psutil.cpu_percent(interval=1),
                "count": psutil.cpu_count(),
            },
            "memory": {
                "total": psutil.virtual_memory().total,
                "used": psutil.virtual_memory().used,
                "percent": psutil.virtual_memory().percent,
            },
            "disk": {
                "total": psutil.disk_usage("/").total,
                "used": psutil.disk_usage("/").used,
                "percent": psutil.disk_usage("/").percent,
            },
            "network": {
                "bytes_sent": psutil.net_io_counters().bytes_sent,
                "bytes_recv": psutil.net_io_counters().bytes_recv,
            }
        }
        
        # 保存到JSON
        with open("monitor.json", "a") as f:
            f.write(json.dumps(stats) + "\n")
        
        # 告警检查
        if stats["memory"]["percent"] > 90:
            print(f"WARNING: High memory usage {stats['memory']['percent']}%")
        
        if stats["disk"]["percent"] > 90:
            print(f"WARNING: High disk usage {stats['disk']['percent']}%")
        
        time.sleep(interval)

if __name__ == "__main__":
    monitor_resources()

实用技巧总结

  1. 添加日志记录:使用logging模块记录运行状态
  2. 异常处理:使用try-except捕获并处理错误
  3. 配置文件:将敏感信息(密码、API密钥)存储在环境变量或配置文件中
  4. 定时执行:使用cron(Linux)或Task Scheduler(Windows)定时运行
  5. 添加通知:关键操作完成后发送邮件或钉钉通知

定时任务配置

Linux Crontab

# 编辑crontab
crontab -e

# 每天凌晨2点执行备份
0 2 * * * /usr/bin/python3 /path/to/backup.py

# 每5分钟监控网站
*/5 * * * * /usr/bin/python3 /path/to/monitor.py
Next step: Learn Airflow, Prefect for workflow orchestration. Combine with RPA tools like PyAutoGUI for GUI automation.

发表评论