autool/utils_ios/ios_count.py
2026-06-17 19:44:18 +08:00

421 lines
15 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# coding: utf-8
"""
iOS 批量测试域名统计工具
用法:
# 统计 output/ 下所有批次结果(原有逻辑)
python ios_count.py
# 只统计指定批次结果文件
python ios_count.py --batch_file output/ios_batch_result_20260225_151849.csv
"""
import os
import csv
import glob
import re
import argparse
from collections import defaultdict
from pathlib import Path
def build_bundle2info(mapping_csv: str) -> dict:
"""
从 appstore_mapping.csv 建立 bundle_id -> {app_id, app_name} 的映射。
同一 bundle_id 多条记录时保留最新last_updated 最大)的那条。
"""
result = {}
if not os.path.isfile(mapping_csv):
return result
try:
with open(mapping_csv, 'r', encoding='utf-8-sig') as f:
reader = csv.DictReader(f)
for row in reader:
bid = row.get('bundle_id', '').strip()
if not bid:
continue
lu = row.get('last_updated', '').strip()
if bid not in result or lu > result[bid]['last_updated']:
result[bid] = {
'app_id': row.get('app_id', '').strip(),
'app_name': row.get('app_name', '').strip(),
'last_updated': lu,
}
except Exception as e:
print(f"[WARN] 读取 {mapping_csv} 失败: {e}")
return result
def build_appid2rawname(apps_csv: str) -> dict:
"""
从 ios_only_apps.csv 建立 纯数字App_ID -> Name 的映射。
"""
result = {}
if not os.path.isfile(apps_csv):
return result
try:
with open(apps_csv, 'r', encoding='utf-8-sig') as f:
reader = csv.DictReader(f)
for row in reader:
numeric_id = row.get('App_ID', '').strip()
name = row.get('Name', '').strip()
if numeric_id and name:
result[numeric_id] = name
except Exception as e:
print(f"[WARN] 读取 {apps_csv} 失败: {e}")
return result
def extract_second_level_domain(domain):
"""提取二级域名,格式:*.example.com"""
if not domain:
return ""
domain = str(domain).lower().replace('http://', '').replace('https://', '').strip().strip('"\'')
parts = domain.split('.')
if len(parts) <= 2:
return domain
return "*." + ".".join(parts[-2:])
def load_success_bundle_ids(output_dir):
"""
从 output/ 下所有 ios_batch_result_*.csv 中读取 SUCCESS 记录。
若同一 bundle_id 多次出现,取步数/时长最大值(保留最佳测试结果)。
返回: dict {
bundle_id: {
'app_name': str,
'raw_app_name': str,
'app_id': str,
'timestamp': str, # 最新记录的 test_timestamp%Y%m%d_%H%M%S
'droidbot_steps': int,
'guiagent_steps': int,
'duration_seconds': int,
}
}
"""
bundle_ids = {}
pattern = os.path.join(output_dir, 'ios_batch_result_*.csv')
csv_files = sorted(glob.glob(pattern))
if not csv_files:
print(f"未找到 ios_batch_result_*.csv 文件(搜索路径: {output_dir}")
return bundle_ids
print(f"共找到 {len(csv_files)} 个批次结果文件:")
for f in csv_files:
print(f" {os.path.basename(f)}")
for csv_path in csv_files:
_load_csv_into(csv_path, bundle_ids)
print(f"\n共找到 {len(bundle_ids)} 个 SUCCESS 状态的应用bundle_id")
return bundle_ids
def load_success_bundle_ids_from_file(csv_path):
"""
从单个 ios_batch_result_*.csv 文件读取所有 SUCCESS 记录。
同一 bundle_id 多次出现时取步数/时长最大值。
返回值格式同 load_success_bundle_ids。
"""
bundle_ids = {}
if not os.path.isfile(csv_path):
print(f"[ERROR] 文件不存在: {csv_path}")
return bundle_ids
print(f"使用批次结果文件: {csv_path}")
_load_csv_into(csv_path, bundle_ids)
print(f"共找到 {len(bundle_ids)} 个 SUCCESS 状态的应用bundle_id")
return bundle_ids
def _load_csv_into(csv_path, bundle_ids: dict):
"""将单个批次 CSV 中的 SUCCESS 记录合并进 bundle_ids 字典。"""
def safe_int(val):
try:
return int(float(val or 0))
except (ValueError, TypeError):
return 0
try:
with open(csv_path, 'r', encoding='utf-8-sig') as f:
reader = csv.DictReader(f)
for row in reader:
status = row.get('status', '').strip()
bundle_id = row.get('bundle_id', '').strip()
if status != 'SUCCESS' or not bundle_id:
continue
app_name = row.get('app_name', '').strip()
raw_app_name = row.get('raw_app_name', '').strip()
app_id = row.get('app_id', '').strip()
timestamp = row.get('timestamp', '').strip()
droidbot_steps = safe_int(row.get('droidbot_steps', 0))
guiagent_steps = safe_int(row.get('guiagent_steps', 0))
duration_seconds = safe_int(row.get('duration_seconds', 0))
if bundle_id not in bundle_ids:
bundle_ids[bundle_id] = {
'app_name': app_name,
'raw_app_name': raw_app_name,
'app_id': app_id,
'timestamp': timestamp,
'droidbot_steps': droidbot_steps,
'guiagent_steps': guiagent_steps,
'duration_seconds': duration_seconds,
}
else:
# 同一应用多次出现,取步数/时长最大值;时间戳取最新(字符串比较即可)
cur = bundle_ids[bundle_id]
cur['droidbot_steps'] = max(cur['droidbot_steps'], droidbot_steps)
cur['guiagent_steps'] = max(cur['guiagent_steps'], guiagent_steps)
cur['duration_seconds'] = max(cur['duration_seconds'], duration_seconds)
if timestamp > cur['timestamp']:
cur['timestamp'] = timestamp
cur['app_name'] = app_name
cur['raw_app_name'] = raw_app_name
cur['app_id'] = app_id
except Exception as e:
print(f"读取 {csv_path} 时出错: {e}")
def bundle_id_to_dir_prefix(bundle_id):
"""
将 bundle_id 中的点号替换为下划线,作为测试结果目录的前缀。
com.anthropic.claude -> com_anthropic_claude
"""
return bundle_id.replace('.', '_')
def find_test_dir_by_timestamp(test_base_dir, bundle_id, timestamp):
"""
单批次模式:通过 bundle_id + timestamp 直接组装测试目录路径。
目录格式:{test_base_dir}/{bundle_prefix}_iOS_{timestamp}
timestamp 格式:%Y%m%d_%H%M%S与 ios_test.py 输出目录名一致)
若精确路径不存在,回退到模糊查找该 timestamp 最接近的目录。
"""
prefix = bundle_id_to_dir_prefix(bundle_id)
exact = os.path.join(test_base_dir, f"{prefix}_iOS_{timestamp}")
if os.path.isdir(exact):
return exact
# 回退:模糊查找前缀匹配的目录中时间戳最接近的
return find_latest_test_dir(test_base_dir, bundle_id)
def find_latest_test_dir(output_dir, bundle_id):
"""
多批次模式:查找时间戳最大(最新)的测试结果目录。
目录命名格式:{bundle_id前缀}_iOS_{时间戳}
"""
prefix = bundle_id_to_dir_prefix(bundle_id)
pattern = os.path.join(output_dir, f"{prefix}_iOS_*")
candidates = [d for d in glob.glob(pattern) if os.path.isdir(d)]
if not candidates:
return None
def extract_timestamp(path):
m = re.search(r'_iOS_(\d{8}_\d{6})$', os.path.basename(path))
return m.group(1) if m else ''
return sorted(candidates, key=extract_timestamp)[-1]
def parse_flows_csv(test_dir, bundle_id):
"""
解析测试目录下 traffic/*.pcap.flows.csv提取非空 TargetDomain 域名集合。
返回: (domains_set, second_level_domains_set)
"""
domains = set()
second_level_domains = set()
traffic_dir = os.path.join(test_dir, 'traffic')
if not os.path.isdir(traffic_dir):
return domains, second_level_domains
flows_files = glob.glob(os.path.join(traffic_dir, '*.flows.csv'))
for flows_path in flows_files:
try:
with open(flows_path, 'r', encoding='utf-8-sig') as f:
reader = csv.DictReader(f)
for row in reader:
domain = row.get('TargetDomain', '').strip()
if domain:
domains.add(domain)
sld = extract_second_level_domain(domain)
if sld:
second_level_domains.add(sld)
except Exception as e:
print(f"解析 {flows_path} 时出错: {e}")
return domains, second_level_domains
def process_all_data(batch_file: str = None):
"""
主处理函数。
Args:
batch_file: 若指定,只处理该批次 CSV否则处理 output/ 下所有批次。
"""
script_dir = os.path.dirname(os.path.abspath(__file__))
output_dir = os.path.join(os.path.dirname(script_dir), 'output')
if not os.path.isdir(output_dir):
output_dir = os.path.join(os.getcwd(), 'output')
# 测试结果子目录
test_base_dir = os.path.join(output_dir, 'ios_test')
if not os.path.isdir(test_base_dir):
test_base_dir = output_dir
print(f"[WARN] output/ios_test 目录不存在,回退到 output/ 查找测试目录")
print(f"批次结果目录: {output_dir}")
print(f"测试结果目录: {test_base_dir}\n")
# ── 加载回退用映射表 ─────────────────────────────────────────────────────
mapping_csv = os.path.join(output_dir, 'ios', 'appstore_mapping.csv')
apps_csv = os.path.join(script_dir, 'ios_only_apps.csv')
bundle2info = build_bundle2info(mapping_csv)
appid2rawname = build_appid2rawname(apps_csv)
# ── 1. 读取 SUCCESS 记录 ──────────────────────────────────────────────
single_mode = batch_file is not None
if single_mode:
bundle_ids = load_success_bundle_ids_from_file(batch_file)
else:
bundle_ids = load_success_bundle_ids(output_dir)
if not bundle_ids:
print("没有找到 SUCCESS 状态的应用,退出。")
return
# ── 2. 找测试目录,解析域名数据 ──────────────────────────────────────
app_data = {}
missing_dirs = []
print("\n开始查找并解析测试结果目录...")
for bundle_id, info in sorted(bundle_ids.items()):
if single_mode and info.get('timestamp'):
# 单批次模式精确定位bundle_id + timestamp
test_dir = find_test_dir_by_timestamp(test_base_dir, bundle_id, info['timestamp'])
else:
# 多批次模式:取最新目录
test_dir = find_latest_test_dir(test_base_dir, bundle_id)
if not test_dir:
missing_dirs.append(bundle_id)
continue
domains, second_level_domains = parse_flows_csv(test_dir, bundle_id)
# ── 回退填充 app_id / raw_app_name ──────────────────────────────────
app_id = info.get('app_id', '')
app_name = info['app_name']
raw_app_name = info.get('raw_app_name', '')
if not app_id or not raw_app_name:
# 1. 用 bundle_id 查 appstore_mapping 得 app_id
binfo = bundle2info.get(bundle_id, {})
if not app_id:
app_id = binfo.get('app_id', '')
if not app_name:
app_name = binfo.get('app_name', '')
if not raw_app_name and app_id:
# 2. 去掉字母前缀(如 "id" -> 纯数字),查 ios_only_apps
numeric_id = re.sub(r'^[^0-9]+', '', app_id)
raw_app_name = appid2rawname.get(numeric_id, '')
app_data[bundle_id] = {
'app_name': app_name,
'raw_app_name': raw_app_name,
'app_id': app_id,
'test_dir': os.path.basename(test_dir),
'domains': domains,
'second_level_domains': second_level_domains,
'droidbot_steps': info['droidbot_steps'],
'guiagent_steps': info['guiagent_steps'],
'duration_seconds': info['duration_seconds'],
}
print(f"\n成功处理 {len(app_data)} 个应用。")
if missing_dirs:
print(f"以下 {len(missing_dirs)} 个 bundle_id 未找到测试结果目录:")
for bid in missing_dirs:
print(f" {bid}")
# ── 3. 输出汇总 CSV ────────────────────────────────────────────────────
from datetime import datetime
output_file = os.path.join(output_dir, f'ios_app_domain_summary_{datetime.now().strftime("%Y%m%d_%H%M%S")}.csv')
headers = [
'app_name', 'raw_app_name', 'app_id', 'bundle_id',
'unique_domain_count', 'unique_second_level_domain_count',
'droidbot_steps', 'guiagent_steps', 'duration_seconds',
'test_dir', 'unique_domain_names', 'unique_second_level_domains',
]
try:
with open(output_file, 'w', newline='', encoding='utf-8-sig') as csvfile:
writer = csv.writer(csvfile)
writer.writerow(headers)
count = 0
for bundle_id in sorted(app_data.keys()):
info = app_data[bundle_id]
domains = sorted(info['domains'])
slds = sorted(info['second_level_domains'])
writer.writerow([
info['app_name'],
info['raw_app_name'],
info['app_id'],
bundle_id,
len(domains),
len(slds),
info['droidbot_steps'],
info['guiagent_steps'],
info['duration_seconds'],
info['test_dir'],
', '.join(domains),
', '.join(slds),
])
count += 1
print(f"\n处理完成!统计了 {count} 个 App。")
print(f"结果已保存至: {output_file}")
except Exception as e:
print(f"写入 CSV 时出错: {e}")
def main():
parser = argparse.ArgumentParser(
description="iOS 批量测试域名统计工具",
formatter_class=argparse.RawTextHelpFormatter,
epilog="""
示例:
# 统计所有批次
python ios_count.py
# 只统计指定批次
python ios_count.py --batch_file output/ios_batch_result_20260225_151849.csv
"""
)
parser.add_argument(
'--batch_file', '-f',
type=str, default=None,
help='指定单个 ios_batch_result_*.csv 文件路径;不指定则处理 output/ 下所有批次'
)
args = parser.parse_args()
process_all_data(batch_file=args.batch_file)
if __name__ == "__main__":
main()