Initial: integrated 2025 LawGraph (graphrag_pipeline) + 2026 kg_project

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
2026-06-17 10:13:39 +08:00
commit 6c1a69af0d
83 changed files with 14295 additions and 0 deletions
+221
View File
@@ -0,0 +1,221 @@
"""
Word文档读取器 - 读取城市更新法规数据库中所有.docx文件
"""
import json
import re
import logging
from pathlib import Path
from datetime import datetime
from typing import List, Dict, Any
try:
from docx import Document
except ImportError:
print("错误: 请先安装 python-docx")
print("运行: pip install python-docx")
raise
class DocxReader:
"""Word文档读取器"""
# 目录类别映射
CATEGORY_MAP = {
'0知识图谱目录': {'id': 0, 'type': 'Index', 'name': '知识图谱目录'},
'1法律': {'id': 1, 'type': 'Law', 'name': '法律'},
'2行政法规': {'id': 2, 'type': 'AdministrativeRegulation', 'name': '行政法规'},
'3部门规章': {'id': 3, 'type': 'DepartmentalRule', 'name': '部门规章'},
'4党中央国务院文件': {'id': 4, 'type': 'PolicyDocument', 'name': '党中央国务院文件'},
'5国家主管部门文件': {'id': 5, 'type': 'PolicyDocument', 'name': '国家主管部门文件'},
'6主要技术标准': {'id': 6, 'type': 'PolicyDocument', 'name': '主要技术标准'},
}
def __init__(self, source_dir: str):
self.source_dir = Path(source_dir)
self.logger = self._setup_logger()
self.documents = []
def _setup_logger(self):
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
)
return logging.getLogger(__name__)
def _extract_text_from_docx(self, filepath: Path) -> str:
"""从Word文档提取纯文本"""
doc = Document(str(filepath))
paragraphs = []
for para in doc.paragraphs:
text = para.text.strip()
if text:
paragraphs.append(text)
# 也提取表格中的文本
for table in doc.tables:
for row in table.rows:
row_text = []
for cell in row.cells:
cell_text = cell.text.strip()
if cell_text:
row_text.append(cell_text)
if row_text:
paragraphs.append(' | '.join(row_text))
return '\n'.join(paragraphs)
def _parse_filename_info(self, filename: str) -> Dict[str, Any]:
"""从文件名解析基本信息"""
info = {
'original_filename': filename,
'document_number': None,
'document_name': None,
'is_draft': False,
}
# 去掉扩展名
name_without_ext = Path(filename).stem
# 检测是否为草案/征求意见稿
if '草案' in name_without_ext or '征求意见稿' in name_without_ext:
info['is_draft'] = True
# 尝试提取编号前缀 (如 "3-中华人民共和国城乡规划法")
match = re.match(r'^(\d+)[-—]\s*(.+)$', name_without_ext)
if match:
info['file_number'] = int(match.group(1))
name_part = match.group(2)
else:
name_part = name_without_ext
# 提取书名号中的名称
title_match = re.search(r'《(.+?)》', name_part)
if title_match:
info['document_name'] = title_match.group(1)
else:
# 去掉常见前缀
cleaned = re.sub(r'^(中华人民共和国|国务院|国土资源部|建设部|住房城乡建设部|自然资源部)\s*', '', name_part)
info['document_name'] = cleaned if cleaned else name_part
return info
def read_all_documents(self) -> List[Dict[str, Any]]:
"""读取所有Word文档"""
self.logger.info(f"开始读取文档目录: {self.source_dir}")
self.documents = []
errors = []
for subdir_name, category_info in self.CATEGORY_MAP.items():
subdir = self.source_dir / subdir_name
if not subdir.exists():
self.logger.warning(f"子目录不存在: {subdir}")
continue
docx_files = sorted(subdir.glob('*.docx'))
self.logger.info(f" {subdir_name}: 发现 {len(docx_files)} 个文件")
for docx_file in docx_files:
try:
text = self._extract_text_from_docx(docx_file)
filename_info = self._parse_filename_info(docx_file.name)
doc_record = {
'doc_id': f"{category_info['type'][:3].upper()}-{len(self.documents):04d}",
'filename': docx_file.name,
'category_id': category_info['id'],
'category_name': category_info['name'],
'document_type': category_info['type'],
'subdirectory': subdir_name,
'raw_text': text,
'text_length': len(text),
'conversion_timestamp': datetime.now().isoformat(),
**filename_info,
}
self.documents.append(doc_record)
except Exception as e:
error_msg = f"读取失败: {docx_file.name} - {str(e)}"
self.logger.error(error_msg)
errors.append({
'filename': docx_file.name,
'subdirectory': subdir_name,
'error': str(e)
})
self.logger.info(f"读取完成: {len(self.documents)} 个文档, {len(errors)} 个错误")
return self.documents
def save_to_json(self, output_path: str):
"""保存到JSON文件"""
output = Path(output_path)
output.parent.mkdir(parents=True, exist_ok=True)
result = {
'metadata': {
'total_documents': len(self.documents),
'source_dir': str(self.source_dir),
'created_at': datetime.now().isoformat(),
'category_distribution': {},
},
'documents': self.documents,
}
# 统计分类分布
for doc in self.documents:
cat = doc['category_name']
result['metadata']['category_distribution'][cat] = \
result['metadata']['category_distribution'].get(cat, 0) + 1
with open(output, 'w', encoding='utf-8') as f:
json.dump(result, f, indent=2, ensure_ascii=False)
self.logger.info(f"已保存到: {output}")
return result
def print_summary(self):
"""打印摘要"""
print("\n" + "=" * 60)
print("文档读取摘要")
print("=" * 60)
print(f"总文档数: {len(self.documents)}")
# 按类别统计
category_counts = {}
total_chars = 0
for doc in self.documents:
cat = doc['category_name']
category_counts[cat] = category_counts.get(cat, 0) + 1
total_chars += doc['text_length']
print("\n类别分布:")
for cat, count in sorted(category_counts.items()):
print(f" {cat}: {count}")
print(f"\n总字符数: {total_chars:,}")
print(f"平均字符数: {total_chars // max(len(self.documents), 1):,}")
# 文本长度分布
lengths = [doc['text_length'] for doc in self.documents]
if lengths:
print(f"\n文本长度:")
print(f" 最短: {min(lengths):,} 字符")
print(f" 最长: {max(lengths):,} 字符")
print(f" 中位数: {sorted(lengths)[len(lengths)//2]:,} 字符")
print("=" * 60)
def main():
source_dir = r"E:\Project\SI\2026_KG_PlanningLaw\data\城市规划法律法规\城市更新法规数据库"
output_path = r"E:\Project\SI\2026_KG_PlanningLaw\dofile\kg_project\output\all_documents.json"
reader = DocxReader(source_dir)
reader.read_all_documents()
reader.save_to_json(output_path)
reader.print_summary()
if __name__ == '__main__':
main()