Initial commit: GasFlux project with core processing pipelines

This commit is contained in:
2026-01-05 09:33:27 +08:00
commit f085d7c5fe
50 changed files with 10634 additions and 0 deletions

View File

@ -0,0 +1,934 @@
#!/usr/bin/env python3
"""
无人机数据处理脚本
处理Excel格式的无人机测量数据,转换为GasFlux标准输入格式。
支持两种使用方式:
1. 命令行调用:python data_processor.py input.xlsx
2. 直接调用:from data_processor import process_file; df = process_file('input.xlsx')
处理步骤:
1. 读取Excel文件
2. 删除不需要的列
3. 根据文件名修正时间格式
4. 坐标转换(经纬度除以10^7)
5. 计算气压(使用qiya.py)
6. 高度调整(减去最小高度)
7. 时间戳融合
8. 字段重命名
作者:GasFlux开发团队
"""
import pandas as pd
import numpy as np
from pathlib import Path
import re
from datetime import datetime
import sys
import os
from collections import Counter
try:
from tqdm import tqdm
HAS_TQDM = True
except ImportError:
HAS_TQDM = False
print("⚠️ 未安装tqdm库,将不显示进度条。如需进度条,请运行: pip install tqdm")
def create_height_bins(heights, bin_size=2.0):
"""
将高度数据按指定间隔分档
Args:
heights: 高度数据Series
bin_size: 每个档位的间隔(米)
Returns:
list: [(bin_min, bin_max, bin_center, count), ...] 每个档位的信息
"""
if len(heights) == 0:
return []
height_min = heights.min()
height_max = heights.max()
# 计算需要的档位数量
range_size = height_max - height_min
if range_size == 0:
# 所有高度相同
return [(height_min, height_max, height_min, len(heights))]
num_bins = max(1, int(np.ceil(range_size / bin_size)))
bins = []
for i in range(num_bins):
bin_min = height_min + i * bin_size
bin_max = min(height_min + (i + 1) * bin_size, height_max)
bin_center = (bin_min + bin_max) / 2
# 统计这个档位有多少数据点
count = ((heights >= bin_min) & (heights < bin_max)).sum()
if i == num_bins - 1: # 最后一个档位包含上限
count = ((heights >= bin_min) & (heights <= bin_max)).sum()
if count > 0: # 只保留有数据的档位
bins.append((bin_min, bin_max, bin_center, count))
return bins
# 导入qiya模块
try:
from .qiya import get_pressure_at_location
print("✅ 成功导入qiya模块")
except ImportError as e:
print(f"❌ 导入qiya模块失败: {e}")
print("请确保GasFlux包结构完整")
sys.exit(1)
def load_excel_data(file_path):
"""
读取Excel文件并进行初步处理
Args:
file_path: Excel文件路径
Returns:
pd.DataFrame: 读取的数据
"""
try:
print(f"正在读取文件: {file_path}")
# 读取Excel文件
df = pd.read_excel(file_path)
print(f"✅ 成功读取数据:{len(df)} 行,{len(df.columns)} 列")
print(f"列名:{list(df.columns)}")
return df
except Exception as e:
print(f"❌ 读取文件失败: {e}")
sys.exit(1)
def remove_columns(df, columns_to_remove):
"""
删除指定的列
Args:
df: 输入DataFrame
columns_to_remove: 要删除的列名列表
Returns:
pd.DataFrame: 删除列后的DataFrame
"""
print(f"\n删除列: {columns_to_remove}")
# 检查要删除的列是否存在
existing_columns = [col for col in columns_to_remove if col in df.columns]
missing_columns = [col for col in columns_to_remove if col not in df.columns]
if missing_columns:
print(f"⚠️ 以下列不存在(跳过): {missing_columns}")
if existing_columns:
df = df.drop(columns=existing_columns)
print(f"✅ 已删除 {len(existing_columns)} 列")
return df
def extract_hour_from_filename(filename):
"""
从文件名中提取小时信息
例如: "08_34_01_间隔高度5m.xlsx" -> "08"
Args:
filename: 文件名
Returns:
str: 小时字符串(两位数)
"""
# 使用正则表达式匹配小时部分
match = re.match(r'(\d{2})_', filename)
if match:
return match.group(1)
else:
print(f"⚠️ 无法从文件名 '{filename}' 中提取小时信息,使用默认值 '00'")
return "00"
def fix_time_column(df, filename):
"""
根据文件名修正时间列
Args:
df: 输入DataFrame
filename: 文件名
Returns:
pd.DataFrame: 修正后的DataFrame
"""
print("修正时间格式...")
# 提取小时信息
hour_prefix = extract_hour_from_filename(filename)
print(f"从文件名提取的小时: {hour_prefix}")
# 检查时间列是否存在
if '时间' not in df.columns:
print("❌ 未找到 '时间' 列")
return df
# 修正时间格式
def fix_single_time(time_str):
if pd.isna(time_str):
return time_str
time_str = str(time_str).strip()
# 如果时间格式类似 "0:34:01" 或 "00:34:01"
if re.match(r'^\d{1,2}:\d{2}:\d{2}$', time_str):
parts = time_str.split(':')
original_hour = int(parts[0])
filename_hour = int(hour_prefix)
# 小时相加
new_hour = (filename_hour + original_hour) % 24 # 防止超过24小时
# 保持分钟和秒不变
new_time = f"{new_hour:02d}:{parts[1]}:{parts[2]}"
return new_time
else:
# 其他格式,使用默认时间
return f"{hour_prefix}:00:00"
# 应用时间修正
original_times = df['时间'].head(3).tolist()
df['时间'] = df['时间'].apply(fix_single_time)
corrected_times = df['时间'].head(3).tolist()
print(f"时间修正示例:")
for orig, corr in zip(original_times, corrected_times):
print(f" {orig} → {corr}")
return df
def convert_coordinates(df):
"""
转换经纬度坐标(除以10^7)
Args:
df: 输入DataFrame
Returns:
pd.DataFrame: 转换后的DataFrame
"""
print("转换经纬度坐标...")
if '经度' in df.columns:
original_lon = df['经度'].head(3).tolist()
df['经度'] = df['经度'] / 1e7
converted_lon = df['经度'].head(3).tolist()
print("经度转换示例:")
for orig, conv in zip(original_lon, converted_lon):
print(".6f")
if '纬度' in df.columns:
original_lat = df['纬度'].head(3).tolist()
df['纬度'] = df['纬度'] / 1e7
converted_lat = df['纬度'].head(3).tolist()
print("纬度转换示例:")
for orig, conv in zip(original_lat, converted_lat):
print(".6f")
return df
def calculate_pressure(df, max_samples=None, height_tolerance=10.0, height_bin_size=2.0):
"""
计算气压数据(高度分档优化版)
Args:
df: 输入DataFrame
max_samples: 最大采样数量(None表示计算所有行)
height_tolerance: 高度变化容差(米),如果所有高度都在此范围内,只计算一次
height_bin_size: 高度分档间隔(米),每个档位使用中间高度计算气压
Returns:
pd.DataFrame: 添加气压列的DataFrame
"""
print("计算气压数据...")
# 检查必要列是否存在
required_cols = ['日期', '时间', '经度', '纬度', '融合高程']
missing_cols = [col for col in required_cols if col not in df.columns]
if missing_cols:
print(f"❌ 缺少必要列: {missing_cols}")
return df
# 检查高度变化范围
height_min = df['融合高程'].min()
height_max = df['融合高程'].max()
height_range = height_max - height_min
print(f"🏔️ 高度范围: {height_min:.1f} - {height_max:.1f} 米 (变化: {height_range:.1f} 米)")
# 创建高度分档
height_bins = create_height_bins(df['融合高程'], height_bin_size)
print(f"📏 高度分档: {len(height_bins)} 个档位 (间隔: {height_bin_size:.1f} 米)")
for i, (bin_min, bin_max, bin_center, count) in enumerate(height_bins):
print(f" 档位{i+1}: {bin_min:.1f}-{bin_max:.1f}m (中心: {bin_center:.1f}m, 数据: {count}行)")
# 决定计算策略
if height_range <= height_tolerance:
# 高度变化小,只计算一次气压
print("🎯 高度变化小,将使用平均高度计算一次气压")
use_single_calculation = True
mean_height = df['融合高程'].mean()
print(f"📍 使用平均高度: {mean_height:.1f} 米")
elif len(height_bins) == 1:
# 只有一个高度档位,使用档位中心高度
print("📦 只有一个高度档位,使用档位中心高度")
use_single_calculation = True
mean_height = height_bins[0][2] # bin_center
print(f"📍 使用档位中心高度: {mean_height:.1f} 米")
else:
# 高度变化大,使用分档计算
print("🏗️ 使用高度分档策略,减少API调用")
use_single_calculation = False
# 确定要处理的行数
if max_samples is None or max_samples >= len(df):
# 计算所有行
sample_df = df.copy()
actual_samples = len(df)
if not use_single_calculation:
print(f"📊 将计算所有 {len(df)} 行的气压数据")
else:
# 限制采样数量
print(f"⚠️ 数据量较大 ({len(df)} 行),只对前 {max_samples} 行计算气压")
sample_df = df.head(max_samples).copy()
actual_samples = max_samples
pressures = []
if use_single_calculation:
# 只计算一次气压
try:
# 使用第一行的日期和时间作为代表
first_row = sample_df.iloc[0]
# 转换日期格式 - 只提取日期部分,移除任何时间信息
date_str = str(first_row['日期'])
if ' ' in date_str:
date_str = date_str.split(' ')[0]
elif 'T' in date_str:
date_str = date_str.split('T')[0]
if '/' in date_str:
date_str = date_str.replace('/', '-')
date_parts = date_str.split('-')
if len(date_parts) == 3 and len(date_parts[0]) == 4:
year, month, day = date_parts
formatted_date = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
else:
raise ValueError(f"日期格式异常: {date_str}")
# 使用数据的代表性时间(整点小时,众数)
time_strings = []
for time_val in sample_df['时间']:
time_str = str(time_val).strip()
if ':' in time_str:
# 确保是有效的 HH:MM 格式,然后取整点小时
parts = time_str.split(':')
if len(parts) >= 2:
try:
hour = int(parts[0])
# 确保小时在有效范围内 (0-23)
if 0 <= hour <= 23:
time_strings.append(f"{hour:02d}:00")
except ValueError:
continue
if time_strings:
time_counts = Counter(time_strings)
formatted_time = time_counts.most_common(1)[0][0]
occurrence_count = time_counts.most_common(1)[0][1]
print(f"数据时间选择: {formatted_time} ({occurrence_count}/{len(time_strings)} 次,{occurrence_count/len(time_strings)*100:.1f}%)")
else:
formatted_time = "12:00"
print("无有效时间数据,使用默认中午12:00")
print("正在计算平均气压...")
pressure = get_pressure_at_location(
lat=sample_df['纬度'].mean(),
lon=sample_df['经度'].mean(),
altitude=mean_height,
date=formatted_date,
time=formatted_time
)
if pressure is not None:
pressures = [pressure] * len(sample_df)
print("✅ 平均气压计算成功,将应用到所有行")
else:
print("❌ 平均气压计算失败")
pressures = [None] * len(sample_df)
except Exception as e:
print(f"❌ 平均气压计算失败: {e}")
pressures = [None] * len(sample_df)
else:
# 使用高度分档策略
print("🏗️ 开始分档计算气压...")
# 为每个高度档位计算气压
bin_pressures = {} # bin_center -> pressure
# 设置进度条
iterator = height_bins
if HAS_TQDM:
iterator = tqdm(iterator, total=len(height_bins), desc="计算气压档位", unit="档")
for bin_min, bin_max, bin_center, count in iterator:
try:
# 使用第一行数据作为代表来获取日期和时间
# 找到这个档位中的一行数据
bin_rows = sample_df[(sample_df['融合高程'] >= bin_min) &
(sample_df['融合高程'] <= bin_max)]
if len(bin_rows) == 0:
continue
first_row = bin_rows.iloc[0]
# 转换日期格式 - 只提取日期部分,移除任何时间信息
date_str = str(first_row['日期'])
if ' ' in date_str:
date_str = date_str.split(' ')[0]
elif 'T' in date_str:
date_str = date_str.split('T')[0]
if '/' in date_str:
date_str = date_str.replace('/', '-')
date_parts = date_str.split('-')
if len(date_parts) == 3 and len(date_parts[0]) == 4:
year, month, day = date_parts
formatted_date = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
else:
print(f"⚠️ 档位高度 {bin_center:.1f}m 日期格式异常: {date_str}")
bin_pressures[bin_center] = None
continue
# 使用该档位数据的代表性时间(整点小时)
time_strings = []
for time_val in bin_rows['时间']:
time_str = str(time_val).strip()
if ':' in time_str:
# 确保是有效的 HH:MM 格式,然后取整点小时
parts = time_str.split(':')
if len(parts) >= 2:
try:
hour = int(parts[0])
# 确保小时在有效范围内 (0-23)
if 0 <= hour <= 23:
time_strings.append(f"{hour:02d}:00")
except ValueError:
continue
if time_strings:
# 使用最常见的时间(众数)
time_counts = Counter(time_strings)
formatted_time = time_counts.most_common(1)[0][0]
occurrence_count = time_counts.most_common(1)[0][1]
print(f" 档位时间选择: {formatted_time} ({occurrence_count}/{len(time_strings)} 次,{occurrence_count/len(time_strings)*100:.1f}%)")
else:
# 如果没有有效时间,使用中午12点
formatted_time = "12:00"
print(" 无有效时间数据,使用默认中午12:00")
# 计算这个档位的气压(使用平均位置和档位中心高度)
avg_lat = bin_rows['纬度'].mean()
avg_lon = bin_rows['经度'].mean()
pressure = get_pressure_at_location(
lat=avg_lat,
lon=avg_lon,
altitude=bin_center,
date=formatted_date,
time=formatted_time
)
bin_pressures[bin_center] = pressure
# 更新进度条
if HAS_TQDM:
success_count = sum(1 for p in bin_pressures.values() if p is not None)
iterator.set_description(f"计算档位 (成功: {success_count}/{len(bin_pressures)})")
except Exception as e:
print(f"❌ 计算高度档位 {bin_center:.1f}m 气压失败: {e}")
bin_pressures[bin_center] = None
# 为每一行分配对应档位的气压
pressures = []
for idx, row in sample_df.iterrows():
# 找到这个高度对应的档位
height = row['融合高程']
assigned_pressure = None
for bin_min, bin_max, bin_center, count in height_bins:
if bin_min <= height <= bin_max:
assigned_pressure = bin_pressures.get(bin_center)
break
pressures.append(assigned_pressure)
print(f"✅ 完成分档气压计算,共 {len(bin_pressures)} 个档位,{len([p for p in bin_pressures.values() if p is not None])} 个成功")
# 添加气压列
df['pressure'] = None # 初始化
df.loc[sample_df.index, 'pressure'] = pressures
# 对于未计算的行,使用插值或平均值填充
if max_samples is not None and len(df) > max_samples:
# 只计算了部分行,用平均值填充其余行
valid_pressures_for_mean = [p for p in pressures if p is not None]
if valid_pressures_for_mean:
mean_pressure = sum(valid_pressures_for_mean) / len(valid_pressures_for_mean)
df['pressure'] = df['pressure'].fillna(mean_pressure)
print(f"使用平均气压填充其余 {len(df) - max_samples} 行: {mean_pressure:.1f} hPa")
# 统计信息
valid_pressures = [p for p in pressures if p is not None]
if valid_pressures:
avg_pressure = sum(valid_pressures) / len(valid_pressures)
print(f"成功计算 {len(valid_pressures)}/{actual_samples} 个气压值,平均值: {avg_pressure:.1f} hPa")
else:
print("⚠️ 未能计算出任何气压值")
return df
def adjust_altitude(df):
"""
调整融合高程(减去最小值)
Args:
df: 输入DataFrame
Returns:
pd.DataFrame: 调整后的DataFrame
"""
print("调整融合高程...")
if '融合高程' in df.columns:
min_altitude = df['融合高程'].min()
print(".2f")
original_alt = df['融合高程'].head(3).tolist()
df['融合高程'] = df['融合高程'] - min_altitude
adjusted_alt = df['融合高程'].head(3).tolist()
print("高度调整示例:")
for orig, adj in zip(original_alt, adjusted_alt):
print(".2f")
return df
def merge_timestamp(df):
"""
融合日期和时间列为时间戳
Args:
df: 输入DataFrame
Returns:
pd.DataFrame: 融合后的DataFrame
"""
print("融合日期和时间...")
if '日期' in df.columns and '时间' in df.columns:
timestamps = []
for idx, row in df.iterrows():
try:
date_str = str(row['日期'])
time_str = str(row['时间'])
# 清理日期字符串 - 移除任何时间部分
date_str = date_str.strip()
if ' ' in date_str:
date_str = date_str.split(' ')[0] # 只取日期部分
if 'T' in date_str:
date_str = date_str.split('T')[0] # 处理ISO格式
# 标准化日期格式
if '/' in date_str:
date_str = date_str.replace('/', '-')
# 确保日期格式正确
date_parts = date_str.split('-')
if len(date_parts) == 3:
year, month, day = date_parts
date_formatted = f"{year.zfill(4)}-{month.zfill(2)}-{day.zfill(2)}"
else:
print(f"⚠️ 日期格式异常: '{date_str}',使用当前日期")
date_formatted = datetime.now().strftime("%Y-%m-%d")
# 时间字符串已经是修正后的格式(如 "08:34:01"),直接使用
time_str = time_str.strip()
if ':' in time_str and len(time_str.split(':')) >= 2:
time_formatted = time_str
else:
print(f"⚠️ 时间格式异常: '{time_str}',使用默认时间")
time_formatted = "12:00:00"
# 组合时间戳 - 直接连接日期和时间
timestamp = f"{date_formatted} {time_formatted}"
timestamps.append(timestamp)
except Exception as e:
print(f"❌ 处理第 {idx+1} 行时间戳失败: {e}")
timestamps.append(f"{datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
df['timestamp'] = timestamps
print("时间戳融合示例:")
for i in range(min(3, len(timestamps))):
print(f" {df.loc[i, '日期']} + {df.loc[i, '时间']} → {timestamps[i]}")
return df
def rename_columns(df):
"""
重命名字段为GasFlux标准格式
Args:
df: 输入DataFrame
Returns:
pd.DataFrame: 重命名后的DataFrame
"""
print("重命名字段...")
# 定义字段映射
column_mapping = {
'timestamp': 'timestamp', # 时间戳(已创建)
'经度': 'longitude', # 经度 → latitude
'纬度': 'latitude', # 纬度 → longitude
'融合高程': 'height_ato', # 融合高程 → height_ato
'修正风向': 'winddir', # 修正风向 → winddir
'修正风速': 'windspeed', # 修正风速 → windspeed
'风温': 'temperature', # 风温 → temperature
'pressure': 'pressure', # 气压(已计算)
'CH4': 'ch4', # CH4保持不变
'pitch': 'course_elevation', # pitch → course_elevation
'yaw': 'course_azimuth' # yaw → course_azimuth
}
# 重命名存在的列
columns_to_rename = {}
for old_name, new_name in column_mapping.items():
if old_name in df.columns:
columns_to_rename[old_name] = new_name
if columns_to_rename:
df = df.rename(columns=columns_to_rename)
print("字段重命名:")
for old, new in columns_to_rename.items():
print(f" {old} → {new}")
# 只保留GasFlux需要的列
required_columns = ['timestamp', 'latitude', 'longitude', 'height_ato', 'windspeed', 'winddir', 'temperature', 'pressure', 'ch4', 'course_elevation', 'course_azimuth']
existing_required_columns = [col for col in required_columns if col in df.columns]
if len(existing_required_columns) != len(required_columns):
missing = [col for col in required_columns if col not in df.columns]
print(f"⚠️ 缺少必需列: {missing}")
# 移除不需要的列,只保留必需的列
df = df[existing_required_columns]
print(f"最终保留列: {existing_required_columns}")
return df
def process_excel_file(file_path):
"""
处理单个Excel文件的主函数
Args:
file_path: Excel文件路径
"""
print(f"=== 开始处理文件: {file_path} ===\n")
# 获取文件名(用于时间修正)
filename = Path(file_path).name
# 1. 读取数据
df = load_excel_data(file_path)
# 2. 删除不需要的列
columns_to_remove = [
'高程', '速度x', '速度y', '速度z',
'四元数_q0', '四元数_q1', '四元数_q2', '四元数_q3',
'roll', 'H2O', # 保留pitch和yaw,将重命名为course_elevation和course_azimuth
'原始风向', '原始风速'
]
df = remove_columns(df, columns_to_remove)
# 3. 修正时间格式
df = fix_time_column(df, filename)
# 4. 坐标转换
df = convert_coordinates(df)
# 5. 计算气压
df = calculate_pressure(df, max_samples=None, height_tolerance=10.0, height_bin_size=2.0) # 计算所有行,高度容差10米,分档2米
# 6. 高度调整
df = adjust_altitude(df)
# 7. 时间戳融合
df = merge_timestamp(df)
# 调试:检查当前列
print(f"时间戳融合后列名: {list(df.columns)}")
if 'timestamp' in df.columns:
print(f"timestamp列示例: {df['timestamp'].head(3).tolist()}")
# 8. 字段重命名
df = rename_columns(df)
# 保存处理结果
output_path = Path(file_path).with_suffix('.processed.csv')
df.to_csv(output_path, index=False)
print(f"\n✅ 处理完成!")
print(f"📁 输出文件: {output_path}")
print(f"📊 最终数据形状: {df.shape[0]} 行 × {df.shape[1]} 列")
print(f"📋 最终列名: {list(df.columns)}")
return df
def process_file(input_file, output_file=None):
"""
直接处理Excel文件的函数(不使用命令行参数)
Args:
input_file: 输入Excel文件路径(字符串或Path对象)
output_file: 输出CSV文件路径(可选,字符串或Path对象)
Returns:
pd.DataFrame: 处理后的DataFrame
"""
# 转换为Path对象
input_path = Path(input_file)
# 检查输入文件
if not input_path.exists():
raise FileNotFoundError(f"输入文件不存在: {input_path}")
if input_path.suffix.lower() not in ['.xlsx', '.xls']:
raise ValueError(f"输入文件必须是Excel格式 (.xlsx 或 .xls),当前文件: {input_path}")
# 处理文件
df = process_excel_file(str(input_path))
# 如果指定了输出路径,额外保存一份
if output_file:
output_path = Path(output_file)
df.to_csv(output_path, index=False)
print(f"📁 额外保存到: {output_path}")
return df
def interactive_input():
"""
交互式输入模式 - 手动输入参数
Returns:
tuple: (input_file, output_file) 文件路径元组
"""
print("=== 手动输入模式 ===")
print("请按照提示输入参数...\n")
# 输入文件路径
while True:
input_file = input("请输入Excel文件路径 (例如: data.xlsx): ").strip()
if not input_file:
print("❌ 文件路径不能为空,请重新输入")
continue
input_path = Path(input_file)
if not input_path.exists():
print(f"❌ 文件不存在: {input_path}")
print("提示: 请确保文件路径正确,或者将文件放在当前目录下")
continue
if input_path.suffix.lower() not in ['.xlsx', '.xls']:
print(f"❌ 文件格式错误: {input_path.suffix}")
print("只支持 .xlsx 和 .xls 格式的Excel文件")
continue
break
# 输出文件路径(可选)
output_file = input("请输入输出CSV文件路径 (可选,直接回车使用默认): ").strip()
if not output_file:
output_file = None
print("使用默认输出文件名")
print(f"\n✅ 输入确认:")
print(f" 输入文件: {input_file}")
print(f" 输出文件: {output_file or '自动生成'}")
confirm = input("\n确认开始处理? (y/N): ").strip().lower()
if confirm not in ['y', 'yes', '是', '确认']:
print("❌ 用户取消操作")
return None, None
return input_file, output_file
def main(input_file=None, output_file=None, interactive=False):
"""
主函数 - 支持多种输入方式
Args:
input_file: 输入文件路径(直接调用时使用)
output_file: 输出文件路径(直接调用时使用)
interactive: 是否启用交互式输入模式
"""
# 如果启用交互式输入
if interactive:
input_file, output_file = interactive_input()
if input_file is None:
return None
# 如果提供了直接参数,使用直接参数
if input_file is not None:
try:
return process_file(input_file, output_file)
except Exception as e:
print(f"❌ 处理失败: {e}")
raise
# 否则使用命令行参数
import argparse
parser = argparse.ArgumentParser(
description="无人机数据处理工具 - 将Excel数据转换为GasFlux格式",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""
使用方式:
1. 命令行模式:
python data_processor.py data.xlsx
python data_processor.py data.xlsx -o output.csv
2. 交互式模式:
python data_processor.py --interactive
3. 直接调用:
from data_processor import process_file
df = process_file('data.xlsx', 'output.csv')
4. Python脚本调用:
from data_processor import main
df = main(input_file='data.xlsx', output_file='output.csv')
处理步骤:
1. 读取Excel文件
2. 删除不需要的列
3. 根据文件名修正时间格式
4. 经纬度坐标转换 (除以10^7)
5. 计算气压数据
6. 高度调整 (减去最小值)
7. 时间戳融合
8. 字段重命名为GasFlux格式
"""
)
parser.add_argument('input_file', nargs='?', help='输入的Excel文件路径')
parser.add_argument('-o', '--output', help='输出CSV文件路径(可选,默认自动生成)')
parser.add_argument('-i', '--interactive', action='store_true', help='启用交互式输入模式')
args = parser.parse_args()
# 如果启用交互式模式
if args.interactive:
input_file, output_file = interactive_input()
if input_file is None:
return None
else:
input_file = args.input_file
output_file = args.output
# 如果没有提供输入文件,显示帮助和使用示例
if not input_file:
parser.print_help()
print("\n" + "="*60)
print("📖 使用示例:")
print("="*60)
print("1. 命令行模式:")
print(" python data_processor.py your_file.xlsx")
print(" python data_processor.py data.xlsx -o output.csv")
print("")
print("2. Python脚本中直接调用:")
print(" from data_processor import process_file")
print(" df = process_file('input.xlsx')")
print(" df = process_file('input.xlsx', 'output.csv')")
print("")
print("3. 交互式模式:")
print(" python data_processor.py --interactive")
print("="*60)
return None
# 检查输入文件
input_path = Path(input_file)
if not input_path.exists():
print(f"❌ 错误:输入文件不存在: {input_path}")
sys.exit(1)
if input_path.suffix.lower() not in ['.xlsx', '.xls']:
print(f"❌ 错误:输入文件必须是Excel格式 (.xlsx 或 .xls)")
sys.exit(1)
# 处理文件
try:
df = process_excel_file(str(input_path))
# 如果指定了输出路径,额外保存一份
if output_file:
output_path = Path(output_file)
df.to_csv(output_path, index=False)
print(f"📁 额外保存到: {output_path}")
return df
except KeyboardInterrupt:
print("\n⚠️ 用户中断处理")
sys.exit(1)
except Exception as e:
print(f"\n❌ 处理失败: {e}")
import traceback
traceback.print_exc()
sys.exit(1)
if __name__ == "__main__":
# 当直接运行脚本时,使用命令行参数模式
main()