python
import pymongo
import pandas as pd
# 这些字段在 MongoDB 中必然存在(或默认存在),无需检查
ALWAYS_EXIST_FIELDS = {"_id"}
def check_index_fields(collection):
"""
检查指定集合中索引字段是否在文档中存在。
返回不存在的字段名列表。
"""
indexes = collection.index_information()
index_fields = set()
for info in indexes.values():
for field, _ in info['key']:
index_fields.add(field)
# 去除必然存在的字段,只检查可能缺失的字段
to_check = index_fields - ALWAYS_EXIST_FIELDS
print(f" [LOG] 待检查索引字段(已去除必然存在字段): {sorted(to_check) if to_check else '无'}")
missing_fields = []
for field in sorted(to_check):
exists = collection.find_one({field: {"$exists": True}}) is not None
print(f" [LOG] 检查字段 '{field}': {'存在' if exists else '缺失'}")
if not exists:
missing_fields.append(field)
return missing_fields
if __name__ == "__main__":
# 连接 MongoDB(请根据实际情况修改连接串)
client = pymongo.MongoClient("mongodb://user:password!@127.0.0.1:27017/ota?authSource=ota&directConnection=true")
db = client["ota"]
collection_names = db.list_collection_names()
print(f"[LOG] 当前数据库包含 {len(collection_names)} 个集合,开始遍历...")
results = []
for idx, coll_name in enumerate(collection_names, start=1):
print(f"[LOG] ({idx}/{len(collection_names)}) 正在检查集合: {coll_name}")
coll = db[coll_name]
missing = check_index_fields(coll)
if missing: # 只保留存在缺失字段的集合
results.append({
"collection": coll_name,
"missing_fields": ", ".join(missing)
})
print(f" [LOG] 集合 {coll_name} 存在缺失字段,已记录")
else:
print(f" [LOG] 集合 {coll_name} 没有缺失字段,不保存到 Excel")
if results:
df = pd.DataFrame(results)
df.to_excel("D:/index_missing_fields.xlsx", index=False)
print(f"[LOG] 检查完成,共 {len(results)} 个集合存在缺失字段,结果已保存到 index_missing_fields.xlsx")
else:
print("[LOG] 所有集合均无缺失字段,无需生成 Excel 文件")