Files
KG_ICH/dofile/kg_project/scripts/extract_kg_csv.py
T
pengxiao 834cad729f init: KG_ICH 项目初始化
- data/: 非遗地理编码数据(GIS shapefile + CSV)
- dofile/kg_project/: 知识图谱构建代码(纳入主仓库)
- dofile/visulization/: 可视化数据与路线图
- officefile/: 文献、草稿、bib 文档
- officefile/latex/: Overleaf 同步目录(独立管理,不纳入)
- output/: 输出目录
- logs/: 日志目录
2026-05-30 00:52:36 +08:00

439 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""
从Excel提取数据生成知识图谱CSV文件
保持原始数据表述不变
"""
import pandas as pd
import re
import json
from pathlib import Path
def extract_batches(df):
"""提取所有唯一的批次(保持原始表述)"""
batches = {}
batch_counter = {}
for batch_str in df['项目批次'].dropna().unique():
if not batch_str or str(batch_str).strip() == '':
continue
# 解析批次字符串(可能包含多个批次)
batch_items = re.split(r'[,、,]', str(batch_str))
for item in batch_items:
item = item.strip()
if not item:
continue
# 生成批次ID(使用原始表述的哈希)
if item not in batch_counter:
batch_counter[item] = 1
else:
batch_counter[item] += 1
batch_id = f"BATCH-{abs(hash(item)) % 100000:05d}"
if batch_id not in batches:
batches[batch_id] = {
'name': item, # 保持原始表述
'original_string': item
}
return batches
def parse_inheritor_field(inheritor_str):
"""解析传承人字段,保持原始表述(如"吴明新(国)""""
if not inheritor_str or str(inheritor_str).strip() in ['', '']:
return []
# 按顿号、逗号分割,保持原始表述
inheritors = re.split(r'[、,,\n]', str(inheritor_str))
inheritors = [inh.strip() for inh in inheritors if inh.strip() and inh.strip() != '']
return inheritors
def extract_inheritors(df):
"""提取所有唯一传承人(保持原始表述)"""
inheritors = {}
inheritor_counter = {}
for idx, row in df.iterrows():
inheritor_str = row.get('代表性传承人', '')
if not inheritor_str or str(inheritor_str).strip() in ['', '']:
continue
# 解析传承人列表
inheritor_names = parse_inheritor_field(inheritor_str)
for name in inheritor_names:
# 使用原始名称(包括括号)作为key
if name not in inheritor_counter:
inheritor_counter[name] = 1
else:
inheritor_counter[name] += 1
# 生成传承人ID
inheritor_id = f"INH-{abs(hash(name)) % 100000:05d}-{inheritor_counter[name]}"
if inheritor_id not in inheritors:
inheritors[inheritor_id] = {
'name': name # 保持原始表述,如"吴明新(国)"
}
return inheritors
def extract_institutions(df):
"""提取所有唯一保护机构(保持原始表述)"""
institutions = {}
inst_counter = {}
for idx, row in df.iterrows():
inst_str = row.get('项目保护单位', '')
if not inst_str or str(inst_str).strip() in ['', '']:
continue
inst_name = str(inst_str).strip()
# 使用原始机构名
if inst_name not in inst_counter:
inst_counter[inst_name] = 1
else:
inst_counter[inst_name] += 1
# 生成机构ID
inst_id = f"INST-{abs(hash(inst_name)) % 100000:05d}-{inst_counter[inst_name]}"
if inst_id not in institutions:
institutions[inst_id] = {
'name': inst_name # 保持原始表述
}
return institutions
def parse_batch_field(batch_str, batches_dict):
"""解析批次字段,返回批次ID列表"""
if not batch_str or str(batch_str).strip() == '':
return []
batches = re.split(r'[,、,]', str(batch_str))
batch_ids = []
for batch in batches:
batch = batch.strip()
if batch:
# 查找对应的批次ID
for batch_id, batch_info in batches_dict.items():
if batch_info['name'] == batch:
batch_ids.append(batch_id)
break
return batch_ids
def get_institution_id(institution_name, institutions_dict):
"""获取机构ID"""
if not institution_name or str(institution_name).strip() in ['', '']:
return None
# 从已提取的机构中查找
for inst_id, inst_info in institutions_dict.items():
if inst_info['name'] == str(institution_name).strip():
return inst_id
return None
def get_inheritor_id(inheritor_name, inheritors_dict):
"""获取传承人ID"""
if not inheritor_name or not inheritor_name.strip():
return None
# 从已提取的传承人中查找
for inheritor_id, inheritor_info in inheritors_dict.items():
if inheritor_info['name'] == inheritor_name:
return inheritor_id
return None
def clean_data(df):
"""数据清洗"""
# 去除完全空白的行
df = df.dropna(how='all')
# 填充缺失值为空字符串
for col in df.columns:
if df[col].dtype == 'object':
df[col] = df[col].fillna('')
return df
def build_nodes(df):
"""构建节点数据(保持原始表述)"""
nodes_list = []
print("正在构建节点...")
# 1. ICH_Project节点
print(" - 构建ICH_Project节点...")
for idx, row in df.iterrows():
seq_num = row.get('总序号', idx + 1)
project_name = row.get('项目名称', '')
# 确定级别(从批次字段推断)
batch_str = str(row.get('项目批次', ''))
level = '国家级' if '国家级' in batch_str else ('省级' if '省级' in batch_str else '')
nodes_list.append({
'id': f"ICH-{int(seq_num)}",
'label': str(project_name),
'type': 'ICH_Project',
'properties': json.dumps({
'level': level,
'category': row.get('类别', ''), # 保持原始表述,如"曲艺类"
'batch': batch_str,
'protection_unit': row.get('项目保护单位', '')
}, ensure_ascii=False)
})
# 2. Category节点(从数据中动态提取,保持原始表述)
print(" - 构建Category节点...")
unique_categories = df['类别'].dropna().unique()
for cat_name in unique_categories:
# 使用类别名称作为ID(使用哈希避免特殊字符)
cat_id = f"CAT-{abs(hash(cat_name)) % 100000:05d}"
nodes_list.append({
'id': cat_id,
'label': cat_name, # 保持原始表述,如"曲艺类"
'type': 'Category',
'properties': '{}'
})
# 3. Batch节点(动态识别,保持原始表述)
print(" - 构建Batch节点...")
batches = extract_batches(df)
for batch_id, batch_info in batches.items():
nodes_list.append({
'id': batch_id,
'label': batch_info['name'], # 保持原始表述,如"国家级第1批"
'type': 'Batch',
'properties': json.dumps({
'original_string': batch_info['original_string']
}, ensure_ascii=False)
})
# 4. Inheritor节点(动态识别,保持原始表述)
print(" - 构建Inheritor节点...")
inheritors = extract_inheritors(df)
for inheritor_id, inheritor_info in inheritors.items():
nodes_list.append({
'id': inheritor_id,
'label': inheritor_info['name'], # 保持原始表述,如"吴明新(国)"
'type': 'Inheritor',
'properties': '{}'
})
# 5. Institution节点(动态识别,保持原始表述)
print(" - 构建Institution节点...")
institutions = extract_institutions(df)
for inst_id, inst_info in institutions.items():
nodes_list.append({
'id': inst_id,
'label': inst_info['name'], # 保持原始表述
'type': 'Institution',
'properties': '{}'
})
return nodes_list, batches, inheritors, institutions
def build_relations(df, batches, inheritors, institutions):
"""构建关系数据"""
relations_list = []
print("正在构建关系...")
# 建立名称到ID的快速查找映射
category_map = {}
for cat_name in df['类别'].dropna().unique():
cat_id = f"CAT-{abs(hash(cat_name)) % 100000:05d}"
category_map[cat_name] = cat_id
for idx, row in df.iterrows():
seq_num = row.get('总序号', idx + 1)
project_id = f"ICH-{int(seq_num)}"
# 1. 项目 → 类别
category_name = row.get('类别', '')
if category_name and category_name in category_map:
category_id = category_map[category_name]
relations_list.append({
'source': project_id,
'target': category_id,
'type': 'BELONGS_TO',
'properties': '{}'
})
# 2. 项目 → 批次(支持多个批次)
batch_str = row.get('项目批次', '')
if batch_str:
batch_ids = parse_batch_field(batch_str, batches)
for batch_id in batch_ids:
relations_list.append({
'source': project_id,
'target': batch_id,
'type': 'SELECTED_IN_BATCH',
'properties': '{}'
})
# 3. 项目 → 保护机构
institution_name = row.get('项目保护单位', '')
if institution_name and str(institution_name).strip() not in ['', '']:
institution_id = get_institution_id(institution_name, institutions)
if institution_id:
relations_list.append({
'source': project_id,
'target': institution_id,
'type': 'PROTECTED_BY',
'properties': '{}'
})
# 4. 项目 → 传承人(支持多个传承人)
inheritor_str = row.get('代表性传承人', '')
if inheritor_str:
inheritor_names = parse_inheritor_field(inheritor_str)
for name in inheritor_names:
inheritor_id = get_inheritor_id(name, inheritors)
if inheritor_id:
relations_list.append({
'source': project_id,
'target': inheritor_id,
'type': 'HAS_INHERITOR',
'properties': '{}'
})
return relations_list
def generate_report(nodes_df, relations_df, output_dir):
"""生成统计报告"""
report_lines = []
report_lines.append("# 知识图谱CSV提取报告\n")
report_lines.append(f"生成时间: {pd.Timestamp.now().strftime('%Y-%m-%d %H:%M:%S')}\n")
report_lines.append("---\n\n")
# 节点统计
report_lines.append("## 节点统计\n\n")
report_lines.append(f"**节点总数**: {len(nodes_df)}\n\n")
node_type_counts = nodes_df['type'].value_counts().sort_index()
report_lines.append("| 节点类型 | 数量 | 占比 |\n")
report_lines.append("|---------|------|------|\n")
for node_type, count in node_type_counts.items():
percentage = (count / len(nodes_df) * 100)
report_lines.append(f"| {node_type} | {count} | {percentage:.1f}% |\n")
# 关系统计
report_lines.append("\n## 关系统计\n\n")
report_lines.append(f"**关系总数**: {len(relations_df)}\n\n")
rel_type_counts = relations_df['type'].value_counts().sort_index()
report_lines.append("| 关系类型 | 数量 | 占比 |\n")
report_lines.append("|---------|------|------|\n")
for rel_type, count in rel_type_counts.items():
percentage = (count / len(relations_df) * 100)
report_lines.append(f"| {rel_type} | {count} | {percentage:.1f}% |\n")
# 保存报告
report_path = output_dir / 'extraction_report.md'
with open(report_path, 'w', encoding='utf-8') as f:
f.writelines(report_lines)
print(f"\n报告已生成: {report_path}")
# 打印统计信息
print("\n" + "="*50)
print("数据提取完成")
print("="*50)
print(f"\n节点总数: {len(nodes_df)}")
for node_type, count in node_type_counts.items():
print(f" - {node_type}: {count}")
print(f"\n关系总数: {len(relations_df)}")
for rel_type, count in rel_type_counts.items():
print(f" - {rel_type}: {count}")
print("="*50)
def extract_data_from_excel():
"""从Excel提取数据并生成知识图谱CSV文件"""
print("开始从Excel提取数据...")
# 1. 读取Excel
excel_path = r'E:\Project\2026_KG_ICH\data\黑龙江国家级和省级非遗名单.xlsx'
print(f"读取文件: {excel_path}")
df = pd.read_excel(excel_path)
print(f"原始数据: {len(df)} 行 x {len(df.columns)}")
# 2. 提取所需列
columns = ['总序号', '类别', '项目名称', '项目批次', '项目保护单位', '代表性传承人']
print(f"\n提取列: {', '.join(columns)}")
# 检查列是否存在
available_cols = [col for col in columns if col in df.columns]
if len(available_cols) < len(columns):
missing = set(columns) - set(available_cols)
print(f"警告: 以下列不存在: {missing}")
df = df[available_cols]
print(f"提取后数据: {len(df)} 行 x {len(df.columns)}")
# 3. 数据清洗
print("\n数据清洗...")
df = clean_data(df)
print(f"清洗后数据: {len(df)}")
# 4. 构建节点
nodes_list, batches, inheritors, institutions = build_nodes(df)
nodes_df = pd.DataFrame(nodes_list)
print(f"节点数据: {len(nodes_df)}")
# 5. 构建关系
relations_list = build_relations(df, batches, inheritors, institutions)
relations_df = pd.DataFrame(relations_list)
print(f"关系数据: {len(relations_df)}")
# 6. 保存CSV
output_dir = Path(r'E:\Project\2026_KG_ICH\dofile\kg_project\output')
output_dir.mkdir(parents=True, exist_ok=True)
print(f"\n保存文件到: {output_dir}")
# 保存为UTF-8-BOM编码(Excel友好)
nodes_path = output_dir / 'nodes.csv'
relations_path = output_dir / 'rels.csv'
nodes_df.to_csv(nodes_path, index=False, encoding='utf-8-sig')
relations_df.to_csv(relations_path, index=False, encoding='utf-8-sig')
print(f" - nodes.csv: {nodes_path}")
print(f" - rels.csv: {relations_path}")
# 7. 生成统计报告
generate_report(nodes_df, relations_df, output_dir)
return nodes_df, relations_df
if __name__ == '__main__':
extract_data_from_excel()