Initial commit: GasFlux project with core processing pipelines
This commit is contained in:
934
src/gasflux/data_processor.py
Normal file
934
src/gasflux/data_processor.py
Normal file
@ -0,0 +1,934 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
无人机数据处理脚本
|
||||
|
||||
处理Excel格式的无人机测量数据,转换为GasFlux标准输入格式。
|
||||
|
||||
支持两种使用方式:
|
||||
1. 命令行调用:python data_processor.py input.xlsx
|
||||
2. 直接调用:from data_processor import process_file; df = process_file('input.xlsx')
|
||||
|
||||
处理步骤:
|
||||
1. 读取Excel文件
|
||||
2. 删除不需要的列
|
||||
3. 根据文件名修正时间格式
|
||||
4. 坐标转换(经纬度除以10^7)
|
||||
5. 计算气压(使用qiya.py)
|
||||
6. 高度调整(减去最小高度)
|
||||
7. 时间戳融合
|
||||
8. 字段重命名
|
||||
|
||||
作者:GasFlux开发团队
|
||||
"""
|
||||
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
from pathlib import Path
|
||||
import re
|
||||
from datetime import datetime
|
||||
import sys
|
||||
import os
|
||||
from collections import Counter
|
||||
|
||||
try:
|
||||
from tqdm import tqdm
|
||||
HAS_TQDM = True
|
||||
except ImportError:
|
||||
HAS_TQDM = False
|
||||
print("⚠️ 未安装tqdm库,将不显示进度条。如需进度条,请运行: pip install tqdm")
|
||||
|
||||
|
||||
def create_height_bins(heights, bin_size=2.0):
|
||||
"""
|
||||
将高度数据按指定间隔分档
|
||||
|
||||
Args:
|
||||
heights: 高度数据Series
|
||||
bin_size: 每个档位的间隔(米)
|
||||
|
||||
Returns:
|
||||
list: [(bin_min, bin_max, bin_center, count), ...] 每个档位的信息
|
||||
"""
|
||||
if len(heights) == 0:
|
||||
return []
|
||||
|
||||
height_min = heights.min()
|
||||
height_max = heights.max()
|
||||
|
||||
# 计算需要的档位数量
|
||||
range_size = height_max - height_min
|
||||
if range_size == 0:
|
||||
# 所有高度相同
|
||||
return [(height_min, height_max, height_min, len(heights))]
|
||||
|
||||
num_bins = max(1, int(np.ceil(range_size / bin_size)))
|
||||
|
||||
bins = []
|
||||
for i in range(num_bins):
|
||||
bin_min = height_min + i * bin_size
|
||||
bin_max = min(height_min + (i + 1) * bin_size, height_max)
|
||||
bin_center = (bin_min + bin_max) / 2
|
||||
|
||||
# 统计这个档位有多少数据点
|
||||
count = ((heights >= bin_min) & (heights < bin_max)).sum()
|
||||
if i == num_bins - 1: # 最后一个档位包含上限
|
||||
count = ((heights >= bin_min) & (heights <= bin_max)).sum()
|
||||
|
||||
if count > 0: # 只保留有数据的档位
|
||||
bins.append((bin_min, bin_max, bin_center, count))
|
||||
|
||||
return bins
|
||||
|
||||
# 导入qiya模块
|
||||
try:
|
||||
from .qiya import get_pressure_at_location
|
||||
print("✅ 成功导入qiya模块")
|
||||
except ImportError as e:
|
||||
print(f"❌ 导入qiya模块失败: {e}")
|
||||
print("请确保GasFlux包结构完整")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def load_excel_data(file_path):
|
||||
"""
|
||||
读取Excel文件并进行初步处理
|
||||
|
||||
Args:
|
||||
file_path: Excel文件路径
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 读取的数据
|
||||
"""
|
||||
try:
|
||||
print(f"正在读取文件: {file_path}")
|
||||
|
||||
# 读取Excel文件
|
||||
df = pd.read_excel(file_path)
|
||||
print(f"✅ 成功读取数据:{len(df)} 行,{len(df.columns)} 列")
|
||||
print(f"列名:{list(df.columns)}")
|
||||
|
||||
return df
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ 读取文件失败: {e}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def remove_columns(df, columns_to_remove):
|
||||
"""
|
||||
删除指定的列
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
columns_to_remove: 要删除的列名列表
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 删除列后的DataFrame
|
||||
"""
|
||||
print(f"\n删除列: {columns_to_remove}")
|
||||
|
||||
# 检查要删除的列是否存在
|
||||
existing_columns = [col for col in columns_to_remove if col in df.columns]
|
||||
missing_columns = [col for col in columns_to_remove if col not in df.columns]
|
||||
|
||||
if missing_columns:
|
||||
print(f"⚠️ 以下列不存在(跳过): {missing_columns}")
|
||||
|
||||
if existing_columns:
|
||||
df = df.drop(columns=existing_columns)
|
||||
print(f"✅ 已删除 {len(existing_columns)} 列")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def extract_hour_from_filename(filename):
|
||||
"""
|
||||
从文件名中提取小时信息
|
||||
|
||||
例如: "08_34_01_间隔高度5m.xlsx" -> "08"
|
||||
|
||||
Args:
|
||||
filename: 文件名
|
||||
|
||||
Returns:
|
||||
str: 小时字符串(两位数)
|
||||
"""
|
||||
# 使用正则表达式匹配小时部分
|
||||
match = re.match(r'(\d{2})_', filename)
|
||||
if match:
|
||||
return match.group(1)
|
||||
else:
|
||||
print(f"⚠️ 无法从文件名 '{filename}' 中提取小时信息,使用默认值 '00'")
|
||||
return "00"
|
||||
|
||||
|
||||
def fix_time_column(df, filename):
|
||||
"""
|
||||
根据文件名修正时间列
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
filename: 文件名
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 修正后的DataFrame
|
||||
"""
|
||||
print("修正时间格式...")
|
||||
|
||||
# 提取小时信息
|
||||
hour_prefix = extract_hour_from_filename(filename)
|
||||
print(f"从文件名提取的小时: {hour_prefix}")
|
||||
|
||||
# 检查时间列是否存在
|
||||
if '时间' not in df.columns:
|
||||
print("❌ 未找到 '时间' 列")
|
||||
return df
|
||||
|
||||
# 修正时间格式
|
||||
def fix_single_time(time_str):
|
||||
if pd.isna(time_str):
|
||||
return time_str
|
||||
|
||||
time_str = str(time_str).strip()
|
||||
|
||||
# 如果时间格式类似 "0:34:01" 或 "00:34:01"
|
||||
if re.match(r'^\d{1,2}:\d{2}:\d{2}$', time_str):
|
||||
parts = time_str.split(':')
|
||||
original_hour = int(parts[0])
|
||||
filename_hour = int(hour_prefix)
|
||||
|
||||
# 小时相加
|
||||
new_hour = (filename_hour + original_hour) % 24 # 防止超过24小时
|
||||
|
||||
# 保持分钟和秒不变
|
||||
new_time = f"{new_hour:02d}:{parts[1]}:{parts[2]}"
|
||||
return new_time
|
||||
else:
|
||||
# 其他格式,使用默认时间
|
||||
return f"{hour_prefix}:00:00"
|
||||
|
||||
# 应用时间修正
|
||||
original_times = df['时间'].head(3).tolist()
|
||||
df['时间'] = df['时间'].apply(fix_single_time)
|
||||
corrected_times = df['时间'].head(3).tolist()
|
||||
|
||||
print(f"时间修正示例:")
|
||||
for orig, corr in zip(original_times, corrected_times):
|
||||
print(f" {orig} → {corr}")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def convert_coordinates(df):
|
||||
"""
|
||||
转换经纬度坐标(除以10^7)
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 转换后的DataFrame
|
||||
"""
|
||||
print("转换经纬度坐标...")
|
||||
|
||||
if '经度' in df.columns:
|
||||
original_lon = df['经度'].head(3).tolist()
|
||||
df['经度'] = df['经度'] / 1e7
|
||||
converted_lon = df['经度'].head(3).tolist()
|
||||
print("经度转换示例:")
|
||||
for orig, conv in zip(original_lon, converted_lon):
|
||||
print(".6f")
|
||||
|
||||
if '纬度' in df.columns:
|
||||
original_lat = df['纬度'].head(3).tolist()
|
||||
df['纬度'] = df['纬度'] / 1e7
|
||||
converted_lat = df['纬度'].head(3).tolist()
|
||||
print("纬度转换示例:")
|
||||
for orig, conv in zip(original_lat, converted_lat):
|
||||
print(".6f")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def calculate_pressure(df, max_samples=None, height_tolerance=10.0, height_bin_size=2.0):
|
||||
"""
|
||||
计算气压数据(高度分档优化版)
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
max_samples: 最大采样数量(None表示计算所有行)
|
||||
height_tolerance: 高度变化容差(米),如果所有高度都在此范围内,只计算一次
|
||||
height_bin_size: 高度分档间隔(米),每个档位使用中间高度计算气压
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 添加气压列的DataFrame
|
||||
"""
|
||||
print("计算气压数据...")
|
||||
|
||||
# 检查必要列是否存在
|
||||
required_cols = ['日期', '时间', '经度', '纬度', '融合高程']
|
||||
missing_cols = [col for col in required_cols if col not in df.columns]
|
||||
|
||||
if missing_cols:
|
||||
print(f"❌ 缺少必要列: {missing_cols}")
|
||||
return df
|
||||
|
||||
# 检查高度变化范围
|
||||
height_min = df['融合高程'].min()
|
||||
height_max = df['融合高程'].max()
|
||||
height_range = height_max - height_min
|
||||
|
||||
print(f"🏔️ 高度范围: {height_min:.1f} - {height_max:.1f} 米 (变化: {height_range:.1f} 米)")
|
||||
# 创建高度分档
|
||||
height_bins = create_height_bins(df['融合高程'], height_bin_size)
|
||||
print(f"📏 高度分档: {len(height_bins)} 个档位 (间隔: {height_bin_size:.1f} 米)")
|
||||
|
||||
for i, (bin_min, bin_max, bin_center, count) in enumerate(height_bins):
|
||||
print(f" 档位{i+1}: {bin_min:.1f}-{bin_max:.1f}m (中心: {bin_center:.1f}m, 数据: {count}行)")
|
||||
# 决定计算策略
|
||||
if height_range <= height_tolerance:
|
||||
# 高度变化小,只计算一次气压
|
||||
print("🎯 高度变化小,将使用平均高度计算一次气压")
|
||||
use_single_calculation = True
|
||||
mean_height = df['融合高程'].mean()
|
||||
print(f"📍 使用平均高度: {mean_height:.1f} 米")
|
||||
elif len(height_bins) == 1:
|
||||
# 只有一个高度档位,使用档位中心高度
|
||||
print("📦 只有一个高度档位,使用档位中心高度")
|
||||
use_single_calculation = True
|
||||
mean_height = height_bins[0][2] # bin_center
|
||||
print(f"📍 使用档位中心高度: {mean_height:.1f} 米")
|
||||
else:
|
||||
# 高度变化大,使用分档计算
|
||||
print("🏗️ 使用高度分档策略,减少API调用")
|
||||
use_single_calculation = False
|
||||
|
||||
# 确定要处理的行数
|
||||
if max_samples is None or max_samples >= len(df):
|
||||
# 计算所有行
|
||||
sample_df = df.copy()
|
||||
actual_samples = len(df)
|
||||
if not use_single_calculation:
|
||||
print(f"📊 将计算所有 {len(df)} 行的气压数据")
|
||||
else:
|
||||
# 限制采样数量
|
||||
print(f"⚠️ 数据量较大 ({len(df)} 行),只对前 {max_samples} 行计算气压")
|
||||
sample_df = df.head(max_samples).copy()
|
||||
actual_samples = max_samples
|
||||
|
||||
pressures = []
|
||||
|
||||
if use_single_calculation:
|
||||
# 只计算一次气压
|
||||
try:
|
||||
# 使用第一行的日期和时间作为代表
|
||||
first_row = sample_df.iloc[0]
|
||||
|
||||
# 转换日期格式 - 只提取日期部分,移除任何时间信息
|
||||
date_str = str(first_row['日期'])
|
||||
if ' ' in date_str:
|
||||
date_str = date_str.split(' ')[0]
|
||||
elif 'T' in date_str:
|
||||
date_str = date_str.split('T')[0]
|
||||
|
||||
if '/' in date_str:
|
||||
date_str = date_str.replace('/', '-')
|
||||
|
||||
date_parts = date_str.split('-')
|
||||
if len(date_parts) == 3 and len(date_parts[0]) == 4:
|
||||
year, month, day = date_parts
|
||||
formatted_date = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
else:
|
||||
raise ValueError(f"日期格式异常: {date_str}")
|
||||
|
||||
# 使用数据的代表性时间(整点小时,众数)
|
||||
time_strings = []
|
||||
for time_val in sample_df['时间']:
|
||||
time_str = str(time_val).strip()
|
||||
if ':' in time_str:
|
||||
# 确保是有效的 HH:MM 格式,然后取整点小时
|
||||
parts = time_str.split(':')
|
||||
if len(parts) >= 2:
|
||||
try:
|
||||
hour = int(parts[0])
|
||||
# 确保小时在有效范围内 (0-23)
|
||||
if 0 <= hour <= 23:
|
||||
time_strings.append(f"{hour:02d}:00")
|
||||
except ValueError:
|
||||
continue
|
||||
|
||||
if time_strings:
|
||||
time_counts = Counter(time_strings)
|
||||
formatted_time = time_counts.most_common(1)[0][0]
|
||||
occurrence_count = time_counts.most_common(1)[0][1]
|
||||
print(f"数据时间选择: {formatted_time} ({occurrence_count}/{len(time_strings)} 次,{occurrence_count/len(time_strings)*100:.1f}%)")
|
||||
else:
|
||||
formatted_time = "12:00"
|
||||
print("无有效时间数据,使用默认中午12:00")
|
||||
|
||||
print("正在计算平均气压...")
|
||||
pressure = get_pressure_at_location(
|
||||
lat=sample_df['纬度'].mean(),
|
||||
lon=sample_df['经度'].mean(),
|
||||
altitude=mean_height,
|
||||
date=formatted_date,
|
||||
time=formatted_time
|
||||
)
|
||||
|
||||
if pressure is not None:
|
||||
pressures = [pressure] * len(sample_df)
|
||||
print("✅ 平均气压计算成功,将应用到所有行")
|
||||
else:
|
||||
print("❌ 平均气压计算失败")
|
||||
pressures = [None] * len(sample_df)
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ 平均气压计算失败: {e}")
|
||||
pressures = [None] * len(sample_df)
|
||||
|
||||
else:
|
||||
# 使用高度分档策略
|
||||
print("🏗️ 开始分档计算气压...")
|
||||
|
||||
# 为每个高度档位计算气压
|
||||
bin_pressures = {} # bin_center -> pressure
|
||||
|
||||
# 设置进度条
|
||||
iterator = height_bins
|
||||
if HAS_TQDM:
|
||||
iterator = tqdm(iterator, total=len(height_bins), desc="计算气压档位", unit="档")
|
||||
|
||||
for bin_min, bin_max, bin_center, count in iterator:
|
||||
try:
|
||||
# 使用第一行数据作为代表来获取日期和时间
|
||||
# 找到这个档位中的一行数据
|
||||
bin_rows = sample_df[(sample_df['融合高程'] >= bin_min) &
|
||||
(sample_df['融合高程'] <= bin_max)]
|
||||
if len(bin_rows) == 0:
|
||||
continue
|
||||
|
||||
first_row = bin_rows.iloc[0]
|
||||
|
||||
# 转换日期格式 - 只提取日期部分,移除任何时间信息
|
||||
date_str = str(first_row['日期'])
|
||||
if ' ' in date_str:
|
||||
date_str = date_str.split(' ')[0]
|
||||
elif 'T' in date_str:
|
||||
date_str = date_str.split('T')[0]
|
||||
|
||||
if '/' in date_str:
|
||||
date_str = date_str.replace('/', '-')
|
||||
|
||||
date_parts = date_str.split('-')
|
||||
if len(date_parts) == 3 and len(date_parts[0]) == 4:
|
||||
year, month, day = date_parts
|
||||
formatted_date = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
else:
|
||||
print(f"⚠️ 档位高度 {bin_center:.1f}m 日期格式异常: {date_str}")
|
||||
bin_pressures[bin_center] = None
|
||||
continue
|
||||
|
||||
# 使用该档位数据的代表性时间(整点小时)
|
||||
time_strings = []
|
||||
for time_val in bin_rows['时间']:
|
||||
time_str = str(time_val).strip()
|
||||
if ':' in time_str:
|
||||
# 确保是有效的 HH:MM 格式,然后取整点小时
|
||||
parts = time_str.split(':')
|
||||
if len(parts) >= 2:
|
||||
try:
|
||||
hour = int(parts[0])
|
||||
# 确保小时在有效范围内 (0-23)
|
||||
if 0 <= hour <= 23:
|
||||
time_strings.append(f"{hour:02d}:00")
|
||||
except ValueError:
|
||||
continue
|
||||
|
||||
if time_strings:
|
||||
# 使用最常见的时间(众数)
|
||||
time_counts = Counter(time_strings)
|
||||
formatted_time = time_counts.most_common(1)[0][0]
|
||||
occurrence_count = time_counts.most_common(1)[0][1]
|
||||
print(f" 档位时间选择: {formatted_time} ({occurrence_count}/{len(time_strings)} 次,{occurrence_count/len(time_strings)*100:.1f}%)")
|
||||
else:
|
||||
# 如果没有有效时间,使用中午12点
|
||||
formatted_time = "12:00"
|
||||
print(" 无有效时间数据,使用默认中午12:00")
|
||||
|
||||
# 计算这个档位的气压(使用平均位置和档位中心高度)
|
||||
avg_lat = bin_rows['纬度'].mean()
|
||||
avg_lon = bin_rows['经度'].mean()
|
||||
|
||||
pressure = get_pressure_at_location(
|
||||
lat=avg_lat,
|
||||
lon=avg_lon,
|
||||
altitude=bin_center,
|
||||
date=formatted_date,
|
||||
time=formatted_time
|
||||
)
|
||||
|
||||
bin_pressures[bin_center] = pressure
|
||||
|
||||
# 更新进度条
|
||||
if HAS_TQDM:
|
||||
success_count = sum(1 for p in bin_pressures.values() if p is not None)
|
||||
iterator.set_description(f"计算档位 (成功: {success_count}/{len(bin_pressures)})")
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ 计算高度档位 {bin_center:.1f}m 气压失败: {e}")
|
||||
bin_pressures[bin_center] = None
|
||||
|
||||
# 为每一行分配对应档位的气压
|
||||
pressures = []
|
||||
for idx, row in sample_df.iterrows():
|
||||
# 找到这个高度对应的档位
|
||||
height = row['融合高程']
|
||||
assigned_pressure = None
|
||||
|
||||
for bin_min, bin_max, bin_center, count in height_bins:
|
||||
if bin_min <= height <= bin_max:
|
||||
assigned_pressure = bin_pressures.get(bin_center)
|
||||
break
|
||||
|
||||
pressures.append(assigned_pressure)
|
||||
|
||||
print(f"✅ 完成分档气压计算,共 {len(bin_pressures)} 个档位,{len([p for p in bin_pressures.values() if p is not None])} 个成功")
|
||||
|
||||
# 添加气压列
|
||||
df['pressure'] = None # 初始化
|
||||
df.loc[sample_df.index, 'pressure'] = pressures
|
||||
|
||||
# 对于未计算的行,使用插值或平均值填充
|
||||
if max_samples is not None and len(df) > max_samples:
|
||||
# 只计算了部分行,用平均值填充其余行
|
||||
valid_pressures_for_mean = [p for p in pressures if p is not None]
|
||||
if valid_pressures_for_mean:
|
||||
mean_pressure = sum(valid_pressures_for_mean) / len(valid_pressures_for_mean)
|
||||
df['pressure'] = df['pressure'].fillna(mean_pressure)
|
||||
print(f"使用平均气压填充其余 {len(df) - max_samples} 行: {mean_pressure:.1f} hPa")
|
||||
|
||||
# 统计信息
|
||||
valid_pressures = [p for p in pressures if p is not None]
|
||||
if valid_pressures:
|
||||
avg_pressure = sum(valid_pressures) / len(valid_pressures)
|
||||
print(f"成功计算 {len(valid_pressures)}/{actual_samples} 个气压值,平均值: {avg_pressure:.1f} hPa")
|
||||
else:
|
||||
print("⚠️ 未能计算出任何气压值")
|
||||
return df
|
||||
|
||||
|
||||
def adjust_altitude(df):
|
||||
"""
|
||||
调整融合高程(减去最小值)
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 调整后的DataFrame
|
||||
"""
|
||||
print("调整融合高程...")
|
||||
|
||||
if '融合高程' in df.columns:
|
||||
min_altitude = df['融合高程'].min()
|
||||
print(".2f")
|
||||
|
||||
original_alt = df['融合高程'].head(3).tolist()
|
||||
df['融合高程'] = df['融合高程'] - min_altitude
|
||||
adjusted_alt = df['融合高程'].head(3).tolist()
|
||||
|
||||
print("高度调整示例:")
|
||||
for orig, adj in zip(original_alt, adjusted_alt):
|
||||
print(".2f")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def merge_timestamp(df):
|
||||
"""
|
||||
融合日期和时间列为时间戳
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 融合后的DataFrame
|
||||
"""
|
||||
print("融合日期和时间...")
|
||||
|
||||
if '日期' in df.columns and '时间' in df.columns:
|
||||
timestamps = []
|
||||
|
||||
for idx, row in df.iterrows():
|
||||
try:
|
||||
date_str = str(row['日期'])
|
||||
time_str = str(row['时间'])
|
||||
|
||||
# 清理日期字符串 - 移除任何时间部分
|
||||
date_str = date_str.strip()
|
||||
if ' ' in date_str:
|
||||
date_str = date_str.split(' ')[0] # 只取日期部分
|
||||
if 'T' in date_str:
|
||||
date_str = date_str.split('T')[0] # 处理ISO格式
|
||||
|
||||
# 标准化日期格式
|
||||
if '/' in date_str:
|
||||
date_str = date_str.replace('/', '-')
|
||||
|
||||
# 确保日期格式正确
|
||||
date_parts = date_str.split('-')
|
||||
if len(date_parts) == 3:
|
||||
year, month, day = date_parts
|
||||
date_formatted = f"{year.zfill(4)}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
else:
|
||||
print(f"⚠️ 日期格式异常: '{date_str}',使用当前日期")
|
||||
date_formatted = datetime.now().strftime("%Y-%m-%d")
|
||||
|
||||
# 时间字符串已经是修正后的格式(如 "08:34:01"),直接使用
|
||||
time_str = time_str.strip()
|
||||
if ':' in time_str and len(time_str.split(':')) >= 2:
|
||||
time_formatted = time_str
|
||||
else:
|
||||
print(f"⚠️ 时间格式异常: '{time_str}',使用默认时间")
|
||||
time_formatted = "12:00:00"
|
||||
|
||||
# 组合时间戳 - 直接连接日期和时间
|
||||
timestamp = f"{date_formatted} {time_formatted}"
|
||||
timestamps.append(timestamp)
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ 处理第 {idx+1} 行时间戳失败: {e}")
|
||||
timestamps.append(f"{datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
|
||||
|
||||
df['timestamp'] = timestamps
|
||||
|
||||
print("时间戳融合示例:")
|
||||
for i in range(min(3, len(timestamps))):
|
||||
print(f" {df.loc[i, '日期']} + {df.loc[i, '时间']} → {timestamps[i]}")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def rename_columns(df):
|
||||
"""
|
||||
重命名字段为GasFlux标准格式
|
||||
|
||||
Args:
|
||||
df: 输入DataFrame
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 重命名后的DataFrame
|
||||
"""
|
||||
print("重命名字段...")
|
||||
|
||||
# 定义字段映射
|
||||
column_mapping = {
|
||||
'timestamp': 'timestamp', # 时间戳(已创建)
|
||||
'经度': 'longitude', # 经度 → latitude
|
||||
'纬度': 'latitude', # 纬度 → longitude
|
||||
'融合高程': 'height_ato', # 融合高程 → height_ato
|
||||
'修正风向': 'winddir', # 修正风向 → winddir
|
||||
'修正风速': 'windspeed', # 修正风速 → windspeed
|
||||
'风温': 'temperature', # 风温 → temperature
|
||||
'pressure': 'pressure', # 气压(已计算)
|
||||
'CH4': 'ch4', # CH4保持不变
|
||||
'pitch': 'course_elevation', # pitch → course_elevation
|
||||
'yaw': 'course_azimuth' # yaw → course_azimuth
|
||||
}
|
||||
|
||||
# 重命名存在的列
|
||||
columns_to_rename = {}
|
||||
for old_name, new_name in column_mapping.items():
|
||||
if old_name in df.columns:
|
||||
columns_to_rename[old_name] = new_name
|
||||
|
||||
if columns_to_rename:
|
||||
df = df.rename(columns=columns_to_rename)
|
||||
print("字段重命名:")
|
||||
for old, new in columns_to_rename.items():
|
||||
print(f" {old} → {new}")
|
||||
|
||||
# 只保留GasFlux需要的列
|
||||
required_columns = ['timestamp', 'latitude', 'longitude', 'height_ato', 'windspeed', 'winddir', 'temperature', 'pressure', 'ch4', 'course_elevation', 'course_azimuth']
|
||||
existing_required_columns = [col for col in required_columns if col in df.columns]
|
||||
|
||||
if len(existing_required_columns) != len(required_columns):
|
||||
missing = [col for col in required_columns if col not in df.columns]
|
||||
print(f"⚠️ 缺少必需列: {missing}")
|
||||
|
||||
# 移除不需要的列,只保留必需的列
|
||||
df = df[existing_required_columns]
|
||||
print(f"最终保留列: {existing_required_columns}")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def process_excel_file(file_path):
|
||||
"""
|
||||
处理单个Excel文件的主函数
|
||||
|
||||
Args:
|
||||
file_path: Excel文件路径
|
||||
"""
|
||||
print(f"=== 开始处理文件: {file_path} ===\n")
|
||||
|
||||
# 获取文件名(用于时间修正)
|
||||
filename = Path(file_path).name
|
||||
|
||||
# 1. 读取数据
|
||||
df = load_excel_data(file_path)
|
||||
|
||||
# 2. 删除不需要的列
|
||||
columns_to_remove = [
|
||||
'高程', '速度x', '速度y', '速度z',
|
||||
'四元数_q0', '四元数_q1', '四元数_q2', '四元数_q3',
|
||||
'roll', 'H2O', # 保留pitch和yaw,将重命名为course_elevation和course_azimuth
|
||||
'原始风向', '原始风速'
|
||||
]
|
||||
df = remove_columns(df, columns_to_remove)
|
||||
|
||||
# 3. 修正时间格式
|
||||
df = fix_time_column(df, filename)
|
||||
|
||||
# 4. 坐标转换
|
||||
df = convert_coordinates(df)
|
||||
|
||||
# 5. 计算气压
|
||||
df = calculate_pressure(df, max_samples=None, height_tolerance=10.0, height_bin_size=2.0) # 计算所有行,高度容差10米,分档2米
|
||||
|
||||
# 6. 高度调整
|
||||
df = adjust_altitude(df)
|
||||
|
||||
# 7. 时间戳融合
|
||||
df = merge_timestamp(df)
|
||||
|
||||
# 调试:检查当前列
|
||||
print(f"时间戳融合后列名: {list(df.columns)}")
|
||||
if 'timestamp' in df.columns:
|
||||
print(f"timestamp列示例: {df['timestamp'].head(3).tolist()}")
|
||||
|
||||
# 8. 字段重命名
|
||||
df = rename_columns(df)
|
||||
|
||||
# 保存处理结果
|
||||
output_path = Path(file_path).with_suffix('.processed.csv')
|
||||
df.to_csv(output_path, index=False)
|
||||
|
||||
print(f"\n✅ 处理完成!")
|
||||
print(f"📁 输出文件: {output_path}")
|
||||
print(f"📊 最终数据形状: {df.shape[0]} 行 × {df.shape[1]} 列")
|
||||
print(f"📋 最终列名: {list(df.columns)}")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def process_file(input_file, output_file=None):
|
||||
"""
|
||||
直接处理Excel文件的函数(不使用命令行参数)
|
||||
|
||||
Args:
|
||||
input_file: 输入Excel文件路径(字符串或Path对象)
|
||||
output_file: 输出CSV文件路径(可选,字符串或Path对象)
|
||||
|
||||
Returns:
|
||||
pd.DataFrame: 处理后的DataFrame
|
||||
"""
|
||||
# 转换为Path对象
|
||||
input_path = Path(input_file)
|
||||
|
||||
# 检查输入文件
|
||||
if not input_path.exists():
|
||||
raise FileNotFoundError(f"输入文件不存在: {input_path}")
|
||||
|
||||
if input_path.suffix.lower() not in ['.xlsx', '.xls']:
|
||||
raise ValueError(f"输入文件必须是Excel格式 (.xlsx 或 .xls),当前文件: {input_path}")
|
||||
|
||||
# 处理文件
|
||||
df = process_excel_file(str(input_path))
|
||||
|
||||
# 如果指定了输出路径,额外保存一份
|
||||
if output_file:
|
||||
output_path = Path(output_file)
|
||||
df.to_csv(output_path, index=False)
|
||||
print(f"📁 额外保存到: {output_path}")
|
||||
|
||||
return df
|
||||
|
||||
|
||||
def interactive_input():
|
||||
"""
|
||||
交互式输入模式 - 手动输入参数
|
||||
|
||||
Returns:
|
||||
tuple: (input_file, output_file) 文件路径元组
|
||||
"""
|
||||
print("=== 手动输入模式 ===")
|
||||
print("请按照提示输入参数...\n")
|
||||
|
||||
# 输入文件路径
|
||||
while True:
|
||||
input_file = input("请输入Excel文件路径 (例如: data.xlsx): ").strip()
|
||||
if not input_file:
|
||||
print("❌ 文件路径不能为空,请重新输入")
|
||||
continue
|
||||
|
||||
input_path = Path(input_file)
|
||||
if not input_path.exists():
|
||||
print(f"❌ 文件不存在: {input_path}")
|
||||
print("提示: 请确保文件路径正确,或者将文件放在当前目录下")
|
||||
continue
|
||||
|
||||
if input_path.suffix.lower() not in ['.xlsx', '.xls']:
|
||||
print(f"❌ 文件格式错误: {input_path.suffix}")
|
||||
print("只支持 .xlsx 和 .xls 格式的Excel文件")
|
||||
continue
|
||||
|
||||
break
|
||||
|
||||
# 输出文件路径(可选)
|
||||
output_file = input("请输入输出CSV文件路径 (可选,直接回车使用默认): ").strip()
|
||||
if not output_file:
|
||||
output_file = None
|
||||
print("使用默认输出文件名")
|
||||
|
||||
print(f"\n✅ 输入确认:")
|
||||
print(f" 输入文件: {input_file}")
|
||||
print(f" 输出文件: {output_file or '自动生成'}")
|
||||
|
||||
confirm = input("\n确认开始处理? (y/N): ").strip().lower()
|
||||
if confirm not in ['y', 'yes', '是', '确认']:
|
||||
print("❌ 用户取消操作")
|
||||
return None, None
|
||||
|
||||
return input_file, output_file
|
||||
|
||||
|
||||
def main(input_file=None, output_file=None, interactive=False):
|
||||
"""
|
||||
主函数 - 支持多种输入方式
|
||||
|
||||
Args:
|
||||
input_file: 输入文件路径(直接调用时使用)
|
||||
output_file: 输出文件路径(直接调用时使用)
|
||||
interactive: 是否启用交互式输入模式
|
||||
"""
|
||||
# 如果启用交互式输入
|
||||
if interactive:
|
||||
input_file, output_file = interactive_input()
|
||||
if input_file is None:
|
||||
return None
|
||||
|
||||
# 如果提供了直接参数,使用直接参数
|
||||
if input_file is not None:
|
||||
try:
|
||||
return process_file(input_file, output_file)
|
||||
except Exception as e:
|
||||
print(f"❌ 处理失败: {e}")
|
||||
raise
|
||||
|
||||
# 否则使用命令行参数
|
||||
import argparse
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
description="无人机数据处理工具 - 将Excel数据转换为GasFlux格式",
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""
|
||||
使用方式:
|
||||
|
||||
1. 命令行模式:
|
||||
python data_processor.py data.xlsx
|
||||
python data_processor.py data.xlsx -o output.csv
|
||||
|
||||
2. 交互式模式:
|
||||
python data_processor.py --interactive
|
||||
|
||||
3. 直接调用:
|
||||
from data_processor import process_file
|
||||
df = process_file('data.xlsx', 'output.csv')
|
||||
|
||||
4. Python脚本调用:
|
||||
from data_processor import main
|
||||
df = main(input_file='data.xlsx', output_file='output.csv')
|
||||
|
||||
处理步骤:
|
||||
1. 读取Excel文件
|
||||
2. 删除不需要的列
|
||||
3. 根据文件名修正时间格式
|
||||
4. 经纬度坐标转换 (除以10^7)
|
||||
5. 计算气压数据
|
||||
6. 高度调整 (减去最小值)
|
||||
7. 时间戳融合
|
||||
8. 字段重命名为GasFlux格式
|
||||
"""
|
||||
)
|
||||
|
||||
parser.add_argument('input_file', nargs='?', help='输入的Excel文件路径')
|
||||
parser.add_argument('-o', '--output', help='输出CSV文件路径(可选,默认自动生成)')
|
||||
parser.add_argument('-i', '--interactive', action='store_true', help='启用交互式输入模式')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# 如果启用交互式模式
|
||||
if args.interactive:
|
||||
input_file, output_file = interactive_input()
|
||||
if input_file is None:
|
||||
return None
|
||||
else:
|
||||
input_file = args.input_file
|
||||
output_file = args.output
|
||||
|
||||
# 如果没有提供输入文件,显示帮助和使用示例
|
||||
if not input_file:
|
||||
parser.print_help()
|
||||
print("\n" + "="*60)
|
||||
print("📖 使用示例:")
|
||||
print("="*60)
|
||||
print("1. 命令行模式:")
|
||||
print(" python data_processor.py your_file.xlsx")
|
||||
print(" python data_processor.py data.xlsx -o output.csv")
|
||||
print("")
|
||||
print("2. Python脚本中直接调用:")
|
||||
print(" from data_processor import process_file")
|
||||
print(" df = process_file('input.xlsx')")
|
||||
print(" df = process_file('input.xlsx', 'output.csv')")
|
||||
print("")
|
||||
print("3. 交互式模式:")
|
||||
print(" python data_processor.py --interactive")
|
||||
print("="*60)
|
||||
return None
|
||||
|
||||
# 检查输入文件
|
||||
input_path = Path(input_file)
|
||||
if not input_path.exists():
|
||||
print(f"❌ 错误:输入文件不存在: {input_path}")
|
||||
sys.exit(1)
|
||||
|
||||
if input_path.suffix.lower() not in ['.xlsx', '.xls']:
|
||||
print(f"❌ 错误:输入文件必须是Excel格式 (.xlsx 或 .xls)")
|
||||
sys.exit(1)
|
||||
|
||||
# 处理文件
|
||||
try:
|
||||
df = process_excel_file(str(input_path))
|
||||
|
||||
# 如果指定了输出路径,额外保存一份
|
||||
if output_file:
|
||||
output_path = Path(output_file)
|
||||
df.to_csv(output_path, index=False)
|
||||
print(f"📁 额外保存到: {output_path}")
|
||||
|
||||
return df
|
||||
|
||||
except KeyboardInterrupt:
|
||||
print("\n⚠️ 用户中断处理")
|
||||
sys.exit(1)
|
||||
except Exception as e:
|
||||
print(f"\n❌ 处理失败: {e}")
|
||||
import traceback
|
||||
traceback.print_exc()
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# 当直接运行脚本时,使用命令行参数模式
|
||||
main()
|
||||
Reference in New Issue
Block a user