增加web_api

This commit is contained in:
2026-02-05 15:13:54 +08:00
parent 443ec09c5c
commit d5edbc0723
43 changed files with 7036 additions and 2640 deletions

View File

@ -0,0 +1 @@
# GasFlux API Blueprints

View File

@ -0,0 +1,67 @@
"""
Configuration Blueprint
Provides configuration information and environment variables endpoints.
"""
import os
from flask import Blueprint
from ..app import Config
from ..shared import _format_response, log_performance, logger
# Create blueprint
config_bp = Blueprint('config', __name__, url_prefix='/config')
@config_bp.route('', methods=['GET'])
@log_performance
def get_config():
"""Get current configuration (without sensitive information)."""
logger.debug("Configuration requested")
try:
config_info = Config.to_dict()
# Remove potentially sensitive information
safe_config = config_info.copy()
data = {
"configuration": safe_config,
"environment_variables": {
"supported": [
"GASFLUX_HOST",
"GASFLUX_PORT",
"GASFLUX_DEBUG",
"GASFLUX_UPLOAD_FOLDER",
"GASFLUX_OUTPUT_FOLDER",
"GASFLUX_MAX_CONTENT_LENGTH",
"GASFLUX_LOG_LEVEL",
"GASFLUX_LOG_FILE",
"GASFLUX_CORS_ORIGINS",
"GASFLUX_TASK_CLEANUP_INTERVAL",
"GASFLUX_MAX_TASK_AGE",
"GASFLUX_THREADS",
"GASFLUX_CONNECTION_LIMIT",
"GASFLUX_CHANNEL_TIMEOUT"
],
"current_values": {
key: os.getenv(key, "not set") if key.startswith("GASFLUX_") else "internal"
for key in [
"GASFLUX_HOST", "GASFLUX_PORT", "GASFLUX_DEBUG",
"GASFLUX_UPLOAD_FOLDER", "GASFLUX_OUTPUT_FOLDER",
"GASFLUX_MAX_CONTENT_LENGTH", "GASFLUX_LOG_LEVEL",
"GASFLUX_LOG_FILE", "GASFLUX_CORS_ORIGINS",
"GASFLUX_TASK_CLEANUP_INTERVAL", "GASFLUX_MAX_TASK_AGE",
"GASFLUX_THREADS", "GASFLUX_CONNECTION_LIMIT",
"GASFLUX_CHANNEL_TIMEOUT"
]
}
}
}
return _format_response(200, "配置信息获取成功", data)
except Exception as e:
logger.error(f"Failed to retrieve configuration: {str(e)}", exc_info=True)
return _format_response(500, "获取配置信息失败", {
"error_details": str(e)
})

View File

@ -0,0 +1,68 @@
"""
Download Blueprint
Handles file download endpoints.
"""
from pathlib import Path
from flask import Blueprint, send_file, current_app
from ..shared import _format_response, log_performance, logger
# Create blueprint
download_bp = Blueprint('download', __name__, url_prefix='/download')
@download_bp.route('/<path:filename>')
@log_performance
def download_file(filename):
"""Download a processed file."""
from flask import request
logger.info(f"Download request for file: {filename} from IP {request.remote_addr}")
try:
# 支持两种路径格式:
# 1. 绝对路径(以 / 开头,如 /full/path/to/file)
# 2. 相对路径(task_id/filename)
if filename.startswith('/'):
# 绝对路径 - 直接使用
file_path = Path(filename)
else:
# 相对路径 - 相对于 OUTPUT_FOLDER
output_folder = Path(current_app.config.get('OUTPUT_FOLDER') or '')
if not output_folder:
logger.error("OUTPUT_FOLDER not configured")
return _format_response(500, "服务器配置错误")
# 解析 task_id/filename 格式
parts = filename.split('/', 1)
if len(parts) != 2:
logger.warning(f"Invalid relative path format: {filename}")
return _format_response(400, "无效的文件路径")
task_id, filename_part = parts
file_path = output_folder / task_id / filename_part
# Security check - ensure file is within output folder
file_path = file_path.resolve()
output_folder = Path(current_app.config.get('OUTPUT_FOLDER') or '').resolve()
if output_folder and not str(file_path).startswith(str(output_folder)):
logger.warning(f"Security violation: Attempted to access file outside output folder: {filename}")
return _format_response(403, "访问被拒绝")
if not file_path.exists():
logger.warning(f"File not found: {filename}")
return _format_response(404, "文件未找到")
if not file_path.is_file():
logger.warning(f"Path is not a file: {filename}")
return _format_response(400, "不是文件")
file_size = file_path.stat().st_size
logger.info(f"Serving file: {filename} ({file_size} bytes)")
return send_file(file_path)
except Exception as e:
logger.error(f"Error serving file {filename}: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")

View File

@ -0,0 +1,94 @@
"""
Health Check Blueprint
Provides API health monitoring and system status endpoints.
"""
import os
import time
import logging
from flask import Blueprint, request
from ..app import Config, stats_collector
from ..shared import task_status, TASK_STATUS_PENDING, TASK_STATUS_PROCESSING
from ..shared import _format_response, log_performance, logger
# Create blueprint
health_bp = Blueprint('health', __name__, url_prefix='/health')
@health_bp.route('', methods=['GET'])
@log_performance
def health_check():
"""API Health Check"""
logger.debug("Health check requested")
try:
# Check storage accessibility
uploads_writable = os.access(Config.UPLOAD_FOLDER, os.W_OK)
outputs_writable = os.access(Config.OUTPUT_FOLDER, os.W_OK)
# Check active tasks
active_tasks = len([t for t in task_status.values() if t.get("status") in [TASK_STATUS_PENDING, TASK_STATUS_PROCESSING]])
# Get basic stats for health check
stats_summary = stats_collector.get_summary()
health_data = {
"status": "healthy",
"version": "1.0.0",
"timestamp": time.time(),
"uptime": stats_summary['summary']['uptime_formatted'],
"storage": {
"uploads_writable": uploads_writable,
"outputs_writable": outputs_writable
},
"tasks": {
"active_count": active_tasks,
"total_tracked": len(task_status),
"total_processed": stats_summary['tasks']['total_completed'] + stats_summary['tasks']['total_failed'],
"success_rate_percent": stats_summary['tasks']['success_rate_percent']
},
"performance": {
"requests_per_second": stats_summary['summary']['requests_per_second'],
"avg_response_time_ms": stats_summary['performance']['avg_response_time_ms'],
"error_rate_percent": stats_summary['summary']['error_rate_percent']
}
}
# Determine health status based on metrics
is_healthy = True
issues = []
if not uploads_writable:
issues.append("上传文件夹不可写")
is_healthy = False
if not outputs_writable:
issues.append("输出文件夹不可写")
is_healthy = False
if active_tasks > 20: # High load threshold
issues.append(f"活跃任务数量过多 ({active_tasks})")
if stats_summary['summary']['error_rate_percent'] > 10: # High error rate
issues.append(f"错误率过高 ({stats_summary['summary']['error_rate_percent']:.1f}%)")
is_healthy = False
health_data["status"] = "healthy" if is_healthy else "degraded"
if issues:
health_data["issues"] = issues
# Log warnings for potential issues
for issue in issues:
logger.warning(f"Health check issue: {issue}")
status_level = logging.DEBUG if is_healthy else logging.WARNING
logger.log(status_level, f"Health check: {health_data['status']} (active tasks: {active_tasks})")
status_code = 200 if is_healthy else 503 # 503 Service Unavailable for degraded
return _format_response(status_code, "健康检查完成" if is_healthy else "服务不可用", health_data)
except Exception as e:
logger.error(f"Health check failed: {str(e)}", exc_info=True)
return _format_response(500, "健康检查失败", {
"status": "unhealthy",
"error": str(e),
"timestamp": time.time()
})

View File

@ -0,0 +1,192 @@
"""
Reports Blueprint
Provides report listing and management endpoints.
"""
import time
from pathlib import Path
from flask import Blueprint, request, current_app
from ..shared import _get_file_type, _format_response, log_performance, logger, task_status
from ..app import Config
# Create blueprint
reports_bp = Blueprint('reports', __name__, url_prefix='/reports')
@reports_bp.route('', methods=['GET'])
@log_performance
def list_reports():
"""List all generated reports with pagination and filtering."""
logger.debug("Reports list requested")
try:
# Parse query parameters
try:
page = int(request.args.get('page', 1))
if page < 1:
return _format_response(400, "Invalid parameter: page must be >= 1")
except (ValueError, TypeError):
return _format_response(400, "Invalid parameter: page must be a valid integer")
try:
per_page = int(request.args.get('per_page', 20))
if per_page < 1 or per_page > 100:
return _format_response(400, "Invalid parameter: per_page must be between 1 and 100")
except (ValueError, TypeError):
return _format_response(400, "Invalid parameter: per_page must be a valid integer")
sort_by = request.args.get('sort_by', 'created_at')
sort_order = request.args.get('sort_order', 'desc')
status_filter = request.args.get('status', None) # 'completed', 'failed', or None for all
# Validate sort parameters
valid_sort_fields = ['created_at', 'task_id', 'file_size', 'processing_time']
if sort_by not in valid_sort_fields:
return _format_response(400, f"Invalid parameter: sort_by must be one of {valid_sort_fields}")
if sort_order not in ['asc', 'desc']:
return _format_response(400, "Invalid parameter: sort_order must be 'asc' or 'desc'")
# Validate status filter
valid_statuses = ['completed', 'failed', None]
if status_filter is not None and status_filter not in ['completed', 'failed']:
return _format_response(400, "Invalid parameter: status must be 'completed', 'failed', or not specified")
# 兼容缺省:优先 app.config,其次 Config.OUTPUT_FOLDER
output_root = current_app.config.get('OUTPUT_FOLDER') or getattr(Config, 'OUTPUT_FOLDER', None)
if not output_root:
return _format_response(200, "报告列表获取成功", {
'reports': [],
'pagination': {'page': page, 'per_page': per_page, 'total_reports': 0, 'total_pages': 0, 'has_next': False, 'has_prev': False},
'filters': {'sort_by': sort_by, 'sort_order': sort_order, 'status': status_filter}
})
output_folder = Path(output_root)
reports = []
# Scan all task directories
if output_folder.exists():
for task_dir in output_folder.iterdir():
if not task_dir.is_dir():
continue
task_id = task_dir.name
# Get task information from global task_status
task_info = task_status.get(task_id, {})
task_status_value = task_info.get('status')
# Log task status for debugging
logger.debug(f"Task {task_id}: status from memory={task_status_value}, info={task_info}")
# 直接扫描平铺文件
files = [p for p in task_dir.iterdir() if p.is_file()]
if not files:
# 按需应用状态过滤
if status_filter:
continue
reports.append({
'task_id': task_id,
'report_name': "N/A",
'status': 'failed',
'created_at': task_dir.stat().st_mtime,
'file_count': 0,
'total_size': 0,
'processing_time_seconds': None,
'main_report': None,
'all_files': [],
'run_directory': f'{task_id}'
})
continue
# 识别主报告与统计
total_size = sum(f.stat().st_size for f in files)
created_at = max(f.stat().st_mtime for f in files) if files else task_dir.stat().st_mtime
def file_entry(p):
return {
'name': p.name,
'size': p.stat().st_size,
'type': _get_file_type(p.name),
# 使用相对路径下载,清晰且安全
'download_url': f"/download/{task_id}/{p.name}"
}
all_files = [file_entry(f) for f in files]
# 优先 CO2_report,其次任意 *_report_*.html
report_html = None
for f in files:
if f.name.endswith('_report_') and f.suffix == '.html':
report_html = file_entry(f)
break
if not report_html:
for f in files:
if f.name.endswith('.html'):
report_html = file_entry(f)
break
# 任务状态:若有报告或关键产物则视为 completed
has_outputs = any(f.name.startswith(('config_', 'output_vars_', 'processed_data_', 'CO2_report_')) for f in files)
task_status_value = 'completed' if has_outputs else 'unknown'
if status_filter and task_status_value != status_filter:
continue
# Create report entry
report_entry = {
'task_id': task_id,
'report_name': task_id,
'status': task_status_value,
'created_at': created_at,
'file_count': len(files),
'total_size': total_size,
'processing_time_seconds': None,
'main_report': report_html,
'all_files': all_files,
'run_directory': f'{task_id}'
}
reports.append(report_entry)
# Sort reports
reverse_order = sort_order == 'desc'
if sort_by == 'created_at':
reports.sort(key=lambda x: x['created_at'], reverse=reverse_order)
elif sort_by == 'task_id':
reports.sort(key=lambda x: x['task_id'], reverse=reverse_order)
elif sort_by == 'file_size':
reports.sort(key=lambda x: x['total_size'], reverse=reverse_order)
elif sort_by == 'processing_time':
reports.sort(key=lambda x: x['processing_time_seconds'] or 0, reverse=reverse_order)
# Paginate results
total_reports = len(reports)
start_idx = (page - 1) * per_page
end_idx = start_idx + per_page
paginated_reports = reports[start_idx:end_idx]
# Calculate pagination metadata
total_pages = (total_reports + per_page - 1) // per_page
response_data = {
'reports': paginated_reports,
'pagination': {
'page': page,
'per_page': per_page,
'total_reports': total_reports,
'total_pages': total_pages,
'has_next': page < total_pages,
'has_prev': page > 1
},
'filters': {
'sort_by': sort_by,
'sort_order': sort_order,
'status': status_filter
}
}
logger.info(f"Returning {len(paginated_reports)} reports (page {page}/{total_pages})")
return _format_response(200, "报告列表获取成功", response_data)
except Exception as e:
logger.error(f"Error listing reports: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")

View File

@ -0,0 +1,93 @@
"""
Statistics Blueprint
Provides API statistics and monitoring endpoints.
"""
import time
from flask import Blueprint, current_app
from ..shared import _format_response, log_performance, logger,stats_collector, task_status
# Create blueprint
stats_bp = Blueprint('stats', __name__, url_prefix='/stats')
@stats_bp.route('', methods=['GET'])
@log_performance
def get_stats():
"""Get detailed API statistics and monitoring data."""
logger.debug("Statistics requested")
try:
# Get detailed statistics
stats_data = stats_collector.get_summary()
# Add current system information
try:
import psutil
memory = psutil.virtual_memory()
disk = psutil.disk_usage(str(current_app.config['OUTPUT_FOLDER']))
stats_data['system'] = {
'memory_usage_percent': memory.percent,
'memory_used_gb': round(memory.used / (1024**3), 2),
'memory_total_gb': round(memory.total / (1024**3), 2),
'disk_usage_percent': disk.percent,
'disk_used_gb': round(disk.used / (1024**3), 2),
'disk_total_gb': round(disk.total / (1024**3), 2)
}
except ImportError:
# psutil not available
stats_data['system'] = {
'note': 'System metrics unavailable - install psutil for detailed monitoring'
}
except Exception as e:
logger.warning(f"Failed to collect system metrics: {e}")
stats_data['system'] = {'error': str(e)}
# Add recent task information
recent_tasks = []
current_time = time.time()
for task_id, task_info in list(task_status.items())[-20:]: # Last 20 tasks
age = current_time - task_info.get('updated_at', 0)
recent_tasks.append({
'task_id': task_id,
'status': task_info.get('status'),
'age_seconds': round(age, 1),
'message': task_info.get('message', '')[:100] # Truncate long messages
})
stats_data['recent_tasks'] = recent_tasks
return _format_response(200, "统计信息获取成功", stats_data)
except Exception as e:
logger.error(f"Failed to retrieve statistics: {str(e)}", exc_info=True)
return _format_response(500, "获取统计信息失败", {
"error_details": str(e)
})
@stats_bp.route('/reset', methods=['POST'])
@log_performance
def reset_stats():
"""Reset API statistics (admin function)."""
logger.warning("Statistics reset requested")
try:
# Reset statistics
stats_collector.reset_stats()
# Log the reset
logger.info("API statistics have been reset")
return _format_response(200, "统计信息重置成功", {
"timestamp": time.time()
})
except Exception as e:
logger.error(f"Failed to reset statistics: {str(e)}", exc_info=True)
return _format_response(500, "重置统计信息失败", {
"error_details": str(e)
})

View File

@ -0,0 +1,228 @@
"""
Task Pool Blueprint
Handles task pool management endpoints: listing tasks with pagination, pool statistics.
"""
from flask import Blueprint, request
from pathlib import Path
from ..shared import (
get_task_list,
get_task_pool_stats,
_format_response,
log_performance,
logger,
task_status,
TASK_STATUS_PENDING,
TASK_STATUS_PROCESSING,
TASK_STATUS_COMPLETED,
TASK_STATUS_FAILED
)
# Create blueprint
task_pool_bp = Blueprint('task_pool', __name__, url_prefix='/tasks')
def _build_simple_downloads_from_results(results: list[dict]) -> dict:
"""
Build direct download shortcuts for common files, based on task results.
This is intentionally minimal and frontend-friendly.
"""
downloads: dict = {}
def set_once(key: str, url: str):
if key not in downloads and url:
downloads[key] = url
for item in results or []:
if not isinstance(item, dict):
continue
rel_path = item.get('rel_path')
if not rel_path:
continue
name_l = (item.get('name') or '').lower()
url = f"/download/{rel_path}"
if name_l.endswith('.xlsx'):
set_once('data_xlsx', url)
elif name_l.endswith('.xls'):
set_once('data_xls', url)
elif name_l.endswith('ch4_report.html'):
set_once('report_ch4', url)
elif name_l.endswith('co2_report.html'):
set_once('report_co2', url)
elif name_l.endswith(('.yaml', '.yml')):
set_once('config', url)
elif name_l.endswith('.json') and 'output_vars' in name_l:
set_once('metadata', url)
elif name_l.endswith('.html'):
# fallback: any html report
set_once('report_html', url)
return downloads
def _lean_task_summary(task_summary: dict) -> dict:
"""Return a minimal task representation for frontend consumption."""
task_id = task_summary.get('task_id')
status = task_summary.get('status')
lean = {
'task_id': task_id,
'status': status,
'message': task_summary.get('message'),
'updated_at': task_summary.get('updated_at'),
}
if status == TASK_STATUS_COMPLETED and task_id:
full_task_info = task_status.get(task_id, {})
results = full_task_info.get('results', []) or []
downloads = _build_simple_downloads_from_results(results)
if downloads:
lean['downloads'] = downloads
return lean
@task_pool_bp.route('', methods=['GET'])
@log_performance
def list_tasks():
"""Get paginated list of tasks with optional filtering."""
logger.debug(f"Task list request from IP {request.remote_addr}")
try:
# Parse query parameters
status_filter = request.args.get('status')
if status_filter:
# Support comma-separated status values
status_filter = status_filter.split(',')
page = int(request.args.get('page', 1))
page_size = int(request.args.get('page_size', 20))
sort_by = request.args.get('sort_by', 'updated_at')
sort_order = request.args.get('sort_order', 'desc')
# Validate parameters
if page < 1:
return _format_response(400, "页码必须大于0")
if page_size < 1 or page_size > 100:
return _format_response(400, "每页数量必须在1-100之间")
valid_sort_fields = ['created_at', 'updated_at', 'status']
if sort_by not in valid_sort_fields:
return _format_response(400, f"排序字段必须是以下之一: {', '.join(valid_sort_fields)}")
if sort_order.lower() not in ['asc', 'desc']:
return _format_response(400, "排序顺序必须是 'asc' 或 'desc'")
# Get task list
result = get_task_list(
status_filter=status_filter,
page=page,
page_size=page_size,
sort_by=sort_by,
sort_order=sort_order,
cleanup=False
)
# Slim response: only task status + downloads (completed only)
result['tasks'] = [_lean_task_summary(t) for t in result.get('tasks', [])]
logger.debug(f"Returning {len(result['tasks'])} tasks (page {page} of {result['total_pages']})")
return _format_response(200, "任务列表查询成功", result)
except ValueError as e:
logger.warning(f"Invalid parameter in task list request: {str(e)}")
return _format_response(400, "参数格式错误")
except Exception as e:
logger.error(f"Error listing tasks: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")
@task_pool_bp.route('/stats', methods=['GET'])
@log_performance
def get_pool_stats():
"""Get task pool statistics."""
logger.debug(f"Task pool stats request from IP {request.remote_addr}")
try:
stats = get_task_pool_stats()
logger.debug(f"Pool stats: {stats['total_tasks']} total tasks, "
f"{stats['active_tasks']} active, {stats['queued_tasks']} queued")
return _format_response(200, "任务池统计信息查询成功", stats)
except Exception as e:
logger.error(f"Error getting pool stats: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")
@task_pool_bp.route('/active', methods=['GET'])
@log_performance
def get_active_tasks():
"""Get list of currently active (processing) tasks."""
logger.debug(f"Active tasks request from IP {request.remote_addr}")
try:
# Get all processing tasks, no pagination needed for active tasks
result = get_task_list(
status_filter=TASK_STATUS_PROCESSING,
page=1,
page_size=1000, # Large page size to get all active tasks
sort_by='updated_at',
sort_order='asc', # Oldest first
cleanup=False
)
active_tasks = result['tasks']
active_tasks = [_lean_task_summary(t) for t in active_tasks]
logger.debug(f"Returning {len(active_tasks)} active tasks")
return _format_response(200, "活跃任务查询成功", {
'active_tasks': active_tasks,
'count': len(active_tasks)
})
except Exception as e:
logger.error(f"Error getting active tasks: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")
@task_pool_bp.route('/queue', methods=['GET'])
@log_performance
def get_queued_tasks():
"""Get list of queued (pending) tasks."""
logger.debug(f"Queued tasks request from IP {request.remote_addr}")
try:
# Get all pending tasks, sorted by creation time
result = get_task_list(
status_filter=TASK_STATUS_PENDING,
page=1,
page_size=1000, # Large page size to get all queued tasks
sort_by='created_at',
sort_order='asc', # Oldest first (FIFO)
cleanup=False
)
queued_tasks = result['tasks']
queued_tasks = [_lean_task_summary(t) for t in queued_tasks]
logger.debug(f"Returning {len(queued_tasks)} queued tasks")
return _format_response(200, "队列任务查询成功", {
'queued_tasks': queued_tasks,
'count': len(queued_tasks),
'queue_position_info': "任务按创建时间排序,较早的任务优先处理"
})
except Exception as e:
logger.error(f"Error getting queued tasks: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")

View File

@ -0,0 +1,220 @@
"""
Tasks Blueprint
Handles task management endpoints: status query, update, and deletion.
"""
from flask import Blueprint, request
from ..shared import (
get_task_status,
update_task_status,
cleanup_old_tasks,
_format_response,
log_performance,
logger,
task_status,
_build_simple_downloads_from_results,
TASK_STATUS_COMPLETED,
TASK_STATUS_FAILED,
TASK_STATUS_PROCESSING,
TASK_STATUS_PENDING,
)
# Create blueprint
tasks_bp = Blueprint('tasks', __name__, url_prefix='/task')
@tasks_bp.route('/<task_id>', methods=['GET'])
@log_performance
def get_task_status_endpoint(task_id):
"""Get the status of a processing task."""
logger.debug(f"Status request for task {task_id}")
try:
# Note: cleanup_old_tasks() is disabled for individual task queries
# to preserve historical task data for task pool management
# cleanup_old_tasks()
task_info = get_task_status(task_id)
if task_info.get("status") == "not_found":
logger.warning(f"Status request for non-existent task {task_id} from IP {request.remote_addr}")
return _format_response(404, "任务未找到")
data = {
"task_id": task_id,
"status": task_info["status"],
"message": task_info.get("message", ""),
"updated_at": task_info.get("updated_at", 0)
}
if task_info["status"] == TASK_STATUS_COMPLETED:
results = task_info.get("results", [])
data["results"] = results
# Add direct download shortcuts for frontend (if available or can be derived)
downloads = task_info.get("downloads") or _build_simple_downloads_from_results(results)
if downloads:
data["downloads"] = downloads
logger.debug(f"Task {task_id}: Returning {len(results)} completed results")
return _format_response(200, "任务查询成功", data)
elif task_info["status"] == TASK_STATUS_FAILED:
error_msg = task_info.get("error", "未知错误")
data["error"] = error_msg
logger.warning(f"Task {task_id}: Returning failure status - {error_msg}")
# Return 200 for failed tasks since this is expected behavior, not an HTTP error
return _format_response(200, "任务处理失败", data)
else:
# Processing or pending status
return _format_response(200, "任务查询成功", data)
except Exception as e:
logger.error(f"Error retrieving status for task {task_id}: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")
@tasks_bp.route('/<task_id>', methods=['PUT'])
@log_performance
def update_task(task_id):
"""Update task status and information."""
logger.info(f"Task update request for {task_id} from IP {request.remote_addr}")
try:
# Validate task exists
task_info = get_task_status(task_id)
if task_info.get("status") == "not_found":
logger.warning(f"Update request for non-existent task {task_id}")
return _format_response(404, "任务未找到")
# Parse request data
data = request.get_json()
if not data:
return _format_response(400, "请求体必须是 JSON 格式")
# Validate allowed fields
allowed_fields = ['status', 'message', 'priority']
valid_statuses = [TASK_STATUS_PENDING, TASK_STATUS_PROCESSING,
TASK_STATUS_COMPLETED, TASK_STATUS_FAILED]
updates = {}
for field in allowed_fields:
if field in data:
if field == 'status' and data[field] not in valid_statuses:
return _format_response(400, f"无效状态。必须是以下之一: {', '.join(valid_statuses)}")
updates[field] = data[field]
if not updates:
return _format_response(400, "没有有效的字段可更新")
# Update task status
current_status = task_info.get('status')
new_status = updates.get('status', current_status)
message = updates.get('message', task_info.get('message'))
# Special handling for status changes
if 'status' in updates:
if new_status == TASK_STATUS_COMPLETED:
# For completed tasks, we might want to add fake results if none exist
if not task_info.get('results'):
logger.warning(f"Marking task {task_id} as completed but no results found")
elif new_status == TASK_STATUS_FAILED:
# For failed tasks, error message is required
error_msg = updates.get('message', 'Task manually marked as failed')
update_task_status(task_id, new_status, error_msg)
else:
update_task_status(task_id, new_status, message)
else:
# Only update message
update_task_status(task_id, current_status, message)
# Update priority if provided
if 'priority' in updates:
task_status[task_id]['priority'] = updates['priority']
# Get updated task info
updated_task = get_task_status(task_id)
data = {
"task_id": task_id,
"status": "updated",
"task_info": {
"status": updated_task.get("status"),
"message": updated_task.get("message"),
"updated_at": updated_task.get("updated_at", 0),
"priority": updated_task.get("priority", "normal")
}
}
logger.info(f"Task {task_id} updated: {updates}")
return _format_response(200, "任务更新成功", data)
except Exception as e:
logger.error(f"Error updating task {task_id}: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")
@tasks_bp.route('/<task_id>', methods=['DELETE'])
@log_performance
def delete_task(task_id):
"""Delete a task and its associated files."""
logger.info(f"Task deletion request for {task_id} from IP {request.remote_addr}")
try:
# Validate task exists
task_info = get_task_status(task_id)
if task_info.get("status") == "not_found":
logger.warning(f"Delete request for non-existent task {task_id}")
return _format_response(404, "任务未找到")
# Check if task is currently processing
if task_info.get("status") in [TASK_STATUS_PROCESSING, TASK_STATUS_PENDING]:
return _format_response(409, "无法删除当前正在处理或等待处理的任务", {
"task_status": task_info.get("status")
})
# Delete associated files
from pathlib import Path
from flask import current_app
import shutil
output_folder = Path(current_app.config['OUTPUT_FOLDER'])
task_folder = output_folder / task_id
files_deleted = 0
total_size_deleted = 0
if task_folder.exists():
try:
# Calculate total size before deletion
for file_path in task_folder.rglob('*'):
if file_path.is_file():
total_size_deleted += file_path.stat().st_size
# Delete the entire task folder
shutil.rmtree(task_folder)
files_deleted = 1 # Count as one folder deleted
logger.info(f"Deleted task folder: {task_folder}")
except Exception as e:
logger.error(f"Error deleting task folder {task_folder}: {str(e)}")
return _format_response(500, f"删除任务文件失败: {str(e)}")
# Remove from task status tracking
if task_id in task_status:
del task_status[task_id]
logger.info(f"Removed task {task_id} from status tracking")
data = {
"task_id": task_id,
"status": "deleted",
"details": {
"folders_deleted": files_deleted,
"total_size_deleted": total_size_deleted,
"task_status": task_info.get("status")
}
}
logger.info(f"Task {task_id} deleted successfully")
return _format_response(200, "任务及相关文件删除成功", data)
except Exception as e:
logger.error(f"Error deleting task {task_id}: {str(e)}", exc_info=True)
return _format_response(500, "内部服务器错误")

View File

@ -0,0 +1,134 @@
"""
Upload Blueprint
Handles file upload and processing initiation endpoints.
"""
import uuid
import threading
from pathlib import Path
from flask import Blueprint, request, current_app
from werkzeug.utils import secure_filename
import yaml
from io import BytesIO
from ..app import process_data_async
from ..shared import _format_response, log_performance, logger, ALLOWED_DATA_EXTENSIONS, ALLOWED_CONFIG_EXTENSIONS, allowed_file,update_task_status, TASK_STATUS_PENDING, TASK_STATUS_FAILED
# Create blueprint
upload_bp = Blueprint('upload', __name__, url_prefix='/upload')
@upload_bp.route('', methods=['POST'])
@log_performance
def upload_file():
logger.info("Received upload request")
logger.info(f"Request content length: {request.content_length} bytes")
# Check if data file is present
if 'file' not in request.files:
logger.warning("Upload failed: No data file part in request")
return _format_response(400, "未找到数据文件部分")
data_file = request.files['file']
config_file = request.files.get('config')
# Log file details
logger.info(f"Data file: {data_file.filename} (size: {getattr(data_file, 'content_length', 'unknown')} bytes)")
if config_file:
logger.info(f"Config file: {config_file.filename} (size: {getattr(config_file, 'content_length', 'unknown')} bytes)")
else:
logger.info("No custom config file provided, will use default")
if data_file.filename == '':
logger.warning("Upload failed: No data file selected (empty filename)")
return _format_response(400, "未选择数据文件")
if not allowed_file(data_file.filename, ALLOWED_DATA_EXTENSIONS):
logger.warning(f"Upload failed: Invalid data file type {data_file.filename} - allowed: {ALLOWED_DATA_EXTENSIONS}")
return _format_response(400, "无效的数据文件类型。只允许 .xlsx 和 .xls 格式。")
# Generate unique job ID
job_id = str(uuid.uuid4())
logger.info(f"Generated job ID: {job_id}")
# 1) Parse config content (parse in memory without saving first)
if config_file and config_file.filename != '':
if not allowed_file(config_file.filename, ALLOWED_CONFIG_EXTENSIONS):
return _format_response(400, "无效的配置文件类型。只允许 .yaml 和 .yml 格式。")
config_file.stream.seek(0)
config_text = config_file.read().decode('utf-8', errors='ignore')
try:
active_config = yaml.safe_load(config_text)
except Exception:
return _format_response(400, "配置文件解析失败")
# Reset stream for saving
config_file.stream = BytesIO(config_text.encode('utf-8'))
else:
default_config_path = Path(__file__).parent.parent / "gasflux_config.yaml"
with open(default_config_path, 'r', encoding='utf-8') as f:
active_config = yaml.safe_load(f)
# 2) Create job directories based on config['output_dir']
output_base = Path(active_config['output_dir']).expanduser()
job_upload_dir = output_base / "uploads" / job_id
job_output_dir = output_base / "outputs" / job_id
job_upload_dir.mkdir(parents=True, exist_ok=True)
job_output_dir.mkdir(parents=True, exist_ok=True)
logger.info(f"Job {job_id}: Created directories - Upload: {job_upload_dir}, Output: {job_output_dir}")
# 3) Save data file to job_upload_dir
data_filename = secure_filename(data_file.filename)
data_path = job_upload_dir / data_filename
try:
data_file.seek(0)
data_file.save(str(data_path))
logger.info(f"Job {job_id}: Data file saved successfully - Path: {data_path}")
except Exception as e:
logger.error(f"Job {job_id}: Failed to save data file {data_filename}: {str(e)}")
return _format_response(500, "保存数据文件失败")
# 4) Save config file to job_upload_dir
if config_file and config_file.filename != '':
config_filename = secure_filename(config_file.filename)
config_path = job_upload_dir / config_filename
try:
config_file.seek(0)
config_file.save(str(config_path))
active_config_path = config_path
logger.info(f"Job {job_id}: Custom config saved successfully - Path: {config_path}")
except Exception as e:
logger.error(f"Job {job_id}: Failed to save config file {config_filename}: {str(e)}")
return _format_response(500, "保存配置文件失败")
else:
# Copy default config for record keeping
config_path = job_upload_dir / "config.yaml"
with open(config_path, 'w', encoding='utf-8') as f:
yaml.safe_dump(active_config, f, allow_unicode=True)
active_config_path = config_path
logger.info(f"Job {job_id}: Default config saved for record - Path: {config_path}")
# Initialize task status
update_task_status(job_id, TASK_STATUS_PENDING, "Task queued for processing")
logger.info(f"Job {job_id}: Task status initialized as PENDING")
# Start background processing
try:
thread = threading.Thread(
target=process_data_async,
args=(job_id, data_path, active_config_path, job_output_dir)
)
thread.daemon = True
thread.start()
logger.info(f"Job {job_id}: Background processing thread started successfully")
except Exception as e:
logger.error(f"Job {job_id}: Failed to start background processing thread: {str(e)}")
update_task_status(job_id, TASK_STATUS_FAILED, error=str(e))
return _format_response(500, "启动处理失败")
logger.info(f"Job {job_id}: Upload process completed successfully, returning job ID to client")
return _format_response(202, "任务已接受并加入处理队列", {
"status": "accepted",
"job_id": job_id,
"task_status_url": f"/task/{job_id}"
})

View File

@ -0,0 +1,233 @@
"""
Web Blueprint
Provides web interface for the GasFlux API.
"""
import time
from pathlib import Path
from flask import Blueprint, render_template_string, current_app
from ..shared import log_performance, logger
from ..app import Config
# Create blueprint
web_bp = Blueprint('web', __name__)
@web_bp.route('/')
@log_performance
def index():
logger.debug("Index page requested")
# 递归查找所有生成的 HTML 报告
start_time = time.time()
all_reports = []
# 优先用 app.config 中的目录,其次回退到 Config.OUTPUT_FOLDER;都不存在则不列出文件
output_root = current_app.config.get('OUTPUT_FOLDER') or getattr(Config, 'OUTPUT_FOLDER', None)
if not output_root:
output_path = None
else:
output_path = Path(output_root)
if output_path and output_path.exists():
try:
for file in output_path.rglob("*.html"):
# 获取相对于 OUTPUT_FOLDER 的相对路径,用于下载链接
rel_path = file.relative_to(output_path).as_posix()
all_reports.append(rel_path)
scan_duration = time.time() - start_time
logger.debug(f"Report scan completed in {scan_duration:.3f}s - found {len(all_reports)} HTML reports")
except Exception as e:
logger.error(f"Error scanning for reports: {str(e)}")
all_reports = []
else:
logger.debug("No output directory configured yet, skipping report scan")
all_reports = []
return render_template_string('''
<!doctype html>
<html>
<head>
<title>GasFlux Web API</title>
<style>
body { font-family: sans-serif; margin: 40px; line-height: 1.6; background-color: #f4f7f6; }
.container { max-width: 900px; margin: auto; background: white; padding: 30px; border-radius: 12px; box-shadow: 0 4px 6px rgba(0,0,0,0.1); }
h1 { color: #2c3e50; border-bottom: 2px solid #3498db; padding-bottom: 10px; }
.upload-section { background: #f8f9fa; padding: 25px; border-radius: 8px; border-left: 5px solid #3498db; margin-bottom: 30px; }
.form-group { margin-bottom: 20px; }
label { display: block; font-weight: bold; margin-bottom: 8px; color: #34495e; }
input[type="file"] { display: block; width: 100%; padding: 10px; border: 1px solid #ddd; border-radius: 4px; }
input[type="submit"] { background: #3498db; color: white; border: none; padding: 12px 25px; border-radius: 4px; cursor: pointer; font-size: 16px; transition: background 0.3s; }
input[type="submit"]:hover { background: #2980b9; }
.results-section { margin-top: 40px; }
.report-item { margin-bottom: 15px; padding: 15px; border: 1px solid #eee; border-radius: 6px; display: flex; justify-content: space-between; align-items: center; }
.report-info { display: flex; flex-direction: column; }
.report-link { font-weight: bold; color: #2980b9; text-decoration: none; font-size: 1.1em; }
.report-link:hover { text-decoration: underline; }
.report-path { font-size: 0.85em; color: #7f8c8d; margin-top: 4px; }
.api-docs { margin-top: 50px; padding: 20px; background: #e8f4f8; border-radius: 8px; font-size: 0.9em; }
code { background: #eee; padding: 2px 5px; border-radius: 3px; font-family: monospace; }
</style>
</head>
<body>
<div class="container">
<h1>GasFlux Web API 控制台</h1>
<div class="upload-section">
<h2>新建处理任务</h2>
<form id="uploadForm" enctype=multipart/form-data>
<div class="form-group">
<label for="data_file">数据文件 (Excel):</label>
<input type="file" name="file" id="data_file" required>
</div>
<div class="form-group">
<label for="config_file">配置文件 (YAML) [可选]:</label>
<input type="file" name="config" id="config_file">
</div>
<button type="submit" id="submitBtn">开始上传并分析</button>
</form>
<div id="taskStatus" style="display: none; margin-top: 20px; padding: 15px; background: #e8f8e8; border-radius: 5px; border: 1px solid #28a745;">
<h3>任务状态</h3>
<p id="statusMessage">正在上传文件...</p>
<div id="progressBar" style="width: 100%; height: 20px; background: #f0f0f0; border-radius: 10px; margin: 10px 0; display: none;">
<div id="progressFill" style="height: 100%; background: #28a745; border-radius: 10px; width: 0%; transition: width 0.3s;"></div>
</div>
<p id="taskId" style="font-size: 0.9em; color: #666;"></p>
</div>
</div>
<div class="results-section">
<h2>已生成的报告</h2>
<div id="reports">
{% for report in reports %}
<div class="report-item">
<div class="report-info">
<a class="report-link" href="/download/{{ report }}" target="_blank">{{ report.split('/')[-1] }}</a>
<span class="report-path">任务 ID: {{ report.split('/')[0] }}</span>
</div>
<a href="/download/{{ report }}" download class="report-link" style="font-size: 0.9em;">下载</a>
</div>
{% else %}
<p style="color: #95a5a6;">暂无已生成的报告。</p>
{% endfor %}
</div>
</div>
<div class="api-docs">
<h3>API 调用指南 (开发者)</h3>
<p><strong>健康检查:</strong> <code>GET /health</code></p>
<p><strong>上传分析:</strong> <code>POST /upload</code></p>
<p><strong>查询任务状态:</strong> <code>GET /task/&lt;task_id&gt;</code></p>
<p>参数: <code>file</code> (Excel), <code>config</code> (YAML, 可选)</p>
<p>示例: <code>curl -X POST -F "file=@data.xlsx" http://localhost:5000/upload</code></p>
<p>状态查询: <code>curl http://localhost:5000/task/your-task-id</code></p>
</div>
<script>
document.getElementById('uploadForm').addEventListener('submit', async function(e) {
e.preventDefault();
const formData = new FormData(this);
const submitBtn = document.getElementById('submitBtn');
const taskStatus = document.getElementById('taskStatus');
const statusMessage = document.getElementById('statusMessage');
const taskIdElement = document.getElementById('taskId');
const progressBar = document.getElementById('progressBar');
const progressFill = document.getElementById('progressFill');
// Disable form and show status
submitBtn.disabled = true;
submitBtn.textContent = '上传中...';
taskStatus.style.display = 'block';
progressBar.style.display = 'block';
progressFill.style.width = '10%';
try {
// Upload file
statusMessage.textContent = '正在上传文件...';
const response = await fetch('/upload', {
method: 'POST',
body: formData
});
if (!response.ok) {
throw new Error(`HTTP ${response.status}: ${response.statusText}`);
}
const result = await response.json();
const taskId = result.data.job_id;
statusMessage.textContent = '文件上传成功,开始处理数据...';
taskIdElement.textContent = `任务ID: ${taskId}`;
progressFill.style.width = '30%';
// Poll for status
let pollCount = 0;
const maxPolls = 300; // 5 minutes max (every 1 second)
const pollStatus = async () => {
try {
const statusResponse = await fetch(`/task/${taskId}`);
const status = await statusResponse.json();
if (status.data.status === 'completed') {
statusMessage.textContent = '处理完成!正在准备下载链接...';
progressFill.style.width = '100%';
submitBtn.textContent = '处理完成!';
submitBtn.disabled = false;
// Reload page to show new reports
setTimeout(() => {
window.location.reload();
}, 2000);
} else if (status.data.status === 'failed') {
statusMessage.textContent = `处理失败: ${status.data.error || '未知错误'}`;
progressFill.style.backgroundColor = '#dc3545';
progressFill.style.width = '100%';
submitBtn.textContent = '处理失败';
submitBtn.disabled = false;
} else {
// Still processing
statusMessage.textContent = status.data.message || '正在处理中...';
const progressPercent = Math.min(30 + (pollCount * 70 / maxPolls), 90);
progressFill.style.width = `${progressPercent}%`;
pollCount++;
if (pollCount < maxPolls) {
setTimeout(pollStatus, 1000);
} else {
statusMessage.textContent = '处理超时,请稍后手动检查状态';
submitBtn.textContent = '处理超时';
submitBtn.disabled = false;
}
}
} catch (error) {
console.error('Status check failed:', error);
statusMessage.textContent = '状态检查失败,请手动刷新页面查看结果';
submitBtn.textContent = '状态检查失败';
submitBtn.disabled = false;
}
};
// Start polling
setTimeout(pollStatus, 1000);
} catch (error) {
console.error('Upload failed:', error);
statusMessage.textContent = `上传失败: ${error.message}`;
progressFill.style.backgroundColor = '#dc3545';
progressFill.style.width = '100%';
submitBtn.textContent = '上传失败';
submitBtn.disabled = false;
}
});
</script>
</div>
</body>
</html>
''', reports=all_reports)