Python是自动化脚本的首选语言。本文提供10个实用的自动化脚本案例,帮助你提升工作效率,减少重复劳动。
1. 批量文件重命名
#!/usr/bin/env python3
import os
import argparse
def batch_rename(directory, prefix, extension=None):
"""批量重命名文件"""
files = sorted(os.listdir(directory))
for i, filename in enumerate(files, 1):
old_path = os.path.join(directory, filename)
if os.path.isfile(old_path):
ext = extension or os.path.splitext(filename)[1]
new_name = f"{prefix}_{i:03d}{ext}"
new_path = os.path.join(directory, new_name)
os.rename(old_path, new_path)
print(f"Renamed: {filename} -> {new_name}")
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("dir", help="Directory containing files")
parser.add_argument("prefix", help="Filename prefix")
parser.add_argument("--ext", help="File extension", default=None)
args = parser.parse_args()
batch_rename(args.dir, args.prefix, args.ext)
2. 自动备份重要文件
#!/usr/bin/env python3
import shutil
import os
from datetime import datetime
import tarfile
def backup_files(source_dirs, backup_dir):
"""备份文件到指定目录"""
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
backup_file = os.path.join(backup_dir, f"backup_{timestamp}.tar.gz")
with tarfile.open(backup_file, "w:gz") as tar:
for source in source_dirs:
if os.path.exists(source):
tar.add(source, arcname=os.path.basename(source))
print(f"Added: {source}")
print(f"Backup created: {backup_file}")
if __name__ == "__main__":
sources = ["/home/user/documents", "/home/user/photos"]
backup_dir = "/backup"
backup_files(sources, backup_dir)
3. 网页内容抓取
#!/usr/bin/env python3
import requests
from bs4 import BeautifulSoup
import csv
def scrape_articles(url, output_file):
"""抓取文章标题和链接"""
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}
response = requests.get(url, headers=headers)
soup = BeautifulSoup(response.text, "html.parser")
articles = []
for item in soup.select(".article-item"):
title = item.select_one("h2").get_text(strip=True)
link = item.select_one("a")["href"]
articles.append({"title": title, "link": link})
# 保存到CSV
with open(output_file, "w", newline="", encoding="utf-8") as f:
writer = csv.DictWriter(f, fieldnames=["title", "link"])
writer.writeheader()
writer.writerows(articles)
print(f"Scraped {len(articles)} articles")
if __name__ == "__main__":
scrape_articles("https://example.com/news", "articles.csv")
4. 自动发送邮件
#!/usr/bin/env python3
import smtplib
from email.mime.text import MIMEText
from email.mime.multipart import MIMEMultipart
import os
def send_email(to_email, subject, body, html=False):
"""发送邮件"""
smtp_server = os.getenv("SMTP_SERVER", "smtp.gmail.com")
smtp_port = int(os.getenv("SMTP_PORT", 587))
username = os.getenv("EMAIL_USER")
password = os.getenv("EMAIL_PASS")
msg = MIMEMultipart()
msg["From"] = username
msg["To"] = to_email
msg["Subject"] = subject
content_type = "html" if html else "plain"
msg.attach(MIMEText(body, content_type))
with smtplib.SMTP(smtp_server, smtp_port) as server:
server.starttls()
server.login(username, password)
server.send_message(msg)
print(f"Email sent to {to_email}")
if __name__ == "__main__":
send_email(
"[email protected]",
"Daily Report",
"Report
Daily summary...
",
html=True
)
5. 监控网站可用性
#!/usr/bin/env python3
import requests
import time
import logging
from datetime import datetime
def monitor_websites(urls, check_interval=300):
"""监控网站可用性"""
logging.basicConfig(
filename="monitor.log",
level=logging.INFO,
format="%(asctime)s - %(levelname)s - %(message)s"
)
while True:
for url in urls:
try:
start = time.time()
response = requests.get(url, timeout=10)
elapsed = time.time() - start
if response.status_code == 200:
print(f"OK {url} - {elapsed:.2f}s")
logging.info(f"UP {url} - {elapsed:.2f}s")
else:
print(f"FAIL {url} - Status {response.status_code}")
logging.warning(f"DOWN {url} - Status {response.status_code}")
except Exception as e:
print(f"ERROR {url} - Error: {e}")
logging.error(f"ERROR {url} - {e}")
time.sleep(check_interval)
if __name__ == "__main__":
urls = ["https://example.com", "https://api.example.com"]
monitor_websites(urls)
6. 自动整理下载文件夹
#!/usr/bin/env python3
import os
import shutil
from pathlib import Path
def organize_downloads(download_dir):
"""按扩展名整理下载文件"""
categories = {
"images": [".jpg", ".jpeg", ".png", ".gif", ".bmp", ".svg"],
"documents": [".pdf", ".doc", ".docx", ".txt", ".xls", ".xlsx", ".ppt", ".pptx"],
"archives": [".zip", ".rar", ".7z", ".tar", ".gz"],
"videos": [".mp4", ".avi", ".mkv", ".mov"],
"music": [".mp3", ".wav", ".flac", ".aac"],
"scripts": [".py", ".js", ".sh", ".bat", ".ps1"],
}
download_path = Path(download_dir)
for file_path in download_path.iterdir():
if file_path.is_file():
ext = file_path.suffix.lower()
for category, extensions in categories.items():
if ext in extensions:
dest_dir = download_path / category
dest_dir.mkdir(exist_ok=True)
dest_file = dest_dir / file_path.name
# 处理重名
counter = 1
while dest_file.exists():
dest_file = dest_dir / f"{file_path.stem}_{counter}{ext}"
counter += 1
shutil.move(str(file_path), str(dest_file))
print(f"Moved: {file_path.name} -> {category}/")
break
if __name__ == "__main__":
organize_downloads(os.path.expanduser("~/Downloads"))
7. 日志分析工具
#!/usr/bin/env python3
import re
from collections import Counter
from datetime import datetime
def analyze_log(log_file):
"""分析日志文件,统计状态码和IP"""
status_pattern = re.compile(r'" (\d{3}) ')
ip_pattern = re.compile(r'^(\d+\.\d+\.\d+\.\d+)')
status_counts = Counter()
ip_counts = Counter()
total_requests = 0
with open(log_file, "r") as f:
for line in f:
total_requests += 1
# 统计状态码
status_match = status_pattern.search(line)
if status_match:
status = status_match.group(1)
status_counts[status] += 1
# 统计IP
ip_match = ip_pattern.match(line)
if ip_match:
ip = ip_match.group(1)
ip_counts[ip] += 1
# 输出报告
print(f"Total requests: {total_requests}")
print("\nStatus codes:")
for code, count in status_counts.most_common():
percentage = (count / total_requests) * 100
print(f" {code}: {count} ({percentage:.1f}%)")
print("\nTop 10 IPs:")
for ip, count in ip_counts.most_common(10):
print(f" {ip}: {count}")
if __name__ == "__main__":
analyze_log("/var/log/nginx/access.log")
8. 数据库自动备份
#!/usr/bin/env python3
import os
import subprocess
from datetime import datetime
def backup_mysql(host, user, password, database, backup_dir):
"""MySQL数据库备份"""
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
backup_file = os.path.join(backup_dir, f"{database}_{timestamp}.sql")
cmd = [
"mysqldump",
"-h", host,
"-u", user,
f"-p{password}",
"--single-transaction",
"--routines",
"--triggers",
database
]
with open(backup_file, "w") as f:
subprocess.run(cmd, stdout=f, check=True)
print(f"Backup created: {backup_file}")
def backup_postgresql(db_name, backup_dir):
"""PostgreSQL数据库备份"""
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
backup_file = os.path.join(backup_dir, f"{db_name}_{timestamp}.sql")
env = os.environ.copy()
env["PGPASSWORD"] = "your_password"
cmd = [
"pg_dump",
"-U", "postgres",
"-F", "c",
"-f", backup_file,
db_name
]
subprocess.run(cmd, env=env, check=True)
print(f"Backup created: {backup_file}")
if __name__ == "__main__":
backup_mysql("localhost", "root", "password", "mydb", "/backup/mysql")
9. 自动部署脚本
#!/usr/bin/env python3
import subprocess
import sys
def deploy():
"""自动部署脚本"""
commands = [
("Pulling latest code", "git pull origin main"),
("Installing dependencies", "pip install -r requirements.txt"),
("Running migrations", "python manage.py migrate"),
("Collecting static files", "python manage.py collectstatic --noinput"),
("Restarting service", "systemctl restart gunicorn"),
("Clearing cache", "redis-cli FLUSHDB"),
]
for step, cmd in commands:
print(f"\n{'='*50}")
print(f"Step: {step}")
print(f"Command: {cmd}")
print('='*50)
result = subprocess.run(cmd, shell=True, capture_output=True, text=True)
if result.returncode != 0:
print(f"ERROR: {result.stderr}")
sys.exit(1)
print(result.stdout)
print("\nDeployment completed successfully!")
if __name__ == "__main__":
deploy()
10. 系统资源监控
#!/usr/bin/env python3
import psutil
import time
import json
from datetime import datetime
def monitor_resources(interval=60):
"""监控系统资源使用情况"""
while True:
timestamp = datetime.now().isoformat()
stats = {
"timestamp": timestamp,
"cpu": {
"percent": psutil.cpu_percent(interval=1),
"count": psutil.cpu_count(),
},
"memory": {
"total": psutil.virtual_memory().total,
"used": psutil.virtual_memory().used,
"percent": psutil.virtual_memory().percent,
},
"disk": {
"total": psutil.disk_usage("/").total,
"used": psutil.disk_usage("/").used,
"percent": psutil.disk_usage("/").percent,
},
"network": {
"bytes_sent": psutil.net_io_counters().bytes_sent,
"bytes_recv": psutil.net_io_counters().bytes_recv,
}
}
# 保存到JSON
with open("monitor.json", "a") as f:
f.write(json.dumps(stats) + "\n")
# 告警检查
if stats["memory"]["percent"] > 90:
print(f"WARNING: High memory usage {stats['memory']['percent']}%")
if stats["disk"]["percent"] > 90:
print(f"WARNING: High disk usage {stats['disk']['percent']}%")
time.sleep(interval)
if __name__ == "__main__":
monitor_resources()
实用技巧总结
- 添加日志记录:使用logging模块记录运行状态
- 异常处理:使用try-except捕获并处理错误
- 配置文件:将敏感信息(密码、API密钥)存储在环境变量或配置文件中
- 定时执行:使用cron(Linux)或Task Scheduler(Windows)定时运行
- 添加通知:关键操作完成后发送邮件或钉钉通知
定时任务配置
Linux Crontab
# 编辑crontab
crontab -e
# 每天凌晨2点执行备份
0 2 * * * /usr/bin/python3 /path/to/backup.py
# 每5分钟监控网站
*/5 * * * * /usr/bin/python3 /path/to/monitor.py
Next step: Learn Airflow, Prefect for workflow orchestration. Combine with RPA tools like PyAutoGUI for GUI automation.