Initial commit: 国土空间规划课程智能体 v1.0

单容器 Docker 架构的国土空间规划课程智能问答系统,集成 FastAPI 后端与 Next.js 前端。

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
2026-05-22 09:40:18 +08:00
commit ddbb79b9f6
167 changed files with 44147 additions and 0 deletions
+162
View File
@@ -0,0 +1,162 @@
"""
导入书籍结构到数据库
解析LaTeX文件并导入书籍、章节、节、小节(知识点)
层级结构:书籍 -> Chapter -> Section -> Subsection(知识点)
"""
import sys
import re
from pathlib import Path
# 添加项目根目录到路径
sys.path.insert(0, str(Path(__file__).parent.parent))
from sqlalchemy.orm import Session
from src.core.database import SessionLocal, create_tables
from src.core.config import get_settings
from src.models.book_structure import Book, Chapter, Section, Subsection
from src.services.latex_parser import LaTeXParser
def import_book_structure():
"""导入书籍结构"""
settings = get_settings()
book_dir = Path(settings.book_dir)
# 如果相对路径不存在,尝试从项目根目录查找
if not book_dir.exists():
# 脚本位置: dofile/backend/scripts/import_book_structure.py
# 项目根目录: 向上3级
script_dir = Path(__file__).parent.parent.parent.parent # 项目根目录
book_dir = script_dir / "data" / "book"
if not book_dir.exists():
print(f"错误: 书籍目录不存在")
print(f"尝试的路径1: {settings.book_dir}")
print(f"尝试的路径2: {book_dir}")
return
print(f"使用书籍目录: {book_dir}")
# 创建数据库表
print("创建数据库表...")
create_tables()
# 创建数据库会话
db: Session = SessionLocal()
try:
# 清空旧数据(如果存在)
db.query(Subsection).delete()
db.query(Section).delete()
db.query(Chapter).delete()
db.query(Book).delete()
db.commit()
print("已清空旧的书籍结构数据。")
# 创建书籍
book_title = "国土空间规划:概念、原理、方法"
book_description = "基于LaTeX书籍内容构建的知识图谱系统"
book = Book(title=book_title, description=book_description)
db.add(book)
db.commit()
db.refresh(book)
print(f"已创建书籍: {book.title}")
parser = LaTeXParser(book_dir)
# 解析主文件获取章节顺序
main_tex_path = book_dir / "main.tex"
if not main_tex_path.exists():
print(f"错误: main.tex 文件不存在于 {book_dir}")
return
main_content = main_tex_path.read_text(encoding='utf-8')
chapter_input_re = re.compile(r'^\\input{(chapter\d+)}', re.MULTILINE)
chapter_files_in_order = []
for match in chapter_input_re.finditer(main_content):
chapter_files_in_order.append(f"{match.group(1)}.tex")
if not chapter_files_in_order:
print("未在 main.tex 中找到章节文件引用。")
return
total_chapters = 0
total_sections = 0
total_subsections = 0
# 遍历章节文件并导入
for chapter_idx, chapter_filename in enumerate(chapter_files_in_order):
chapter_file_path = book_dir / chapter_filename
if not chapter_file_path.exists():
print(f"警告: 章节文件 {chapter_file_path} 不存在,跳过。")
continue
print(f"解析LaTeX文件: {chapter_file_path.name}...")
chapter_structure = parser.parse_chapter_file(chapter_file_path)
if not chapter_structure:
print(f"警告: 文件 {chapter_file_path.name} 未解析出任何结构。")
continue
# 每个文件只包含一个 \chapter
parsed_chapter_data = chapter_structure[0]
chapter_obj = Chapter(
book_id=book.id,
chapter_number=parsed_chapter_data["chapter_number"],
title=parsed_chapter_data["title"],
file_path=chapter_file_path.name,
start_line=parsed_chapter_data["start_line"],
end_line=parsed_chapter_data["end_line"],
display_order=parsed_chapter_data["chapter_number"]
)
db.add(chapter_obj)
db.flush() # Flush to get chapter_obj.id
total_chapters += 1
# 导入节(Section
for section_data in parsed_chapter_data.get("sections", []):
section_obj = Section(
chapter_id=chapter_obj.id,
section_number=section_data["section_number"],
title=section_data["title"],
file_path=chapter_file_path.name,
start_line=section_data["start_line"],
end_line=section_data["end_line"],
display_order=section_data["section_number"]
)
db.add(section_obj)
db.flush() # Flush to get section_obj.id
total_sections += 1
# 导入小节(Subsection,作为知识点)
for subsection_data in section_data.get("subsections", []):
subsection_obj = Subsection(
section_id=section_obj.id,
subsection_number=subsection_data["subsection_number"],
title=subsection_data["title"],
file_path=chapter_file_path.name,
start_line=subsection_data["start_line"],
end_line=subsection_data["end_line"],
display_order=subsection_data["subsection_number"]
)
db.add(subsection_obj)
total_subsections += 1
db.commit()
print(f"\n[SUCCESS] 书籍结构导入成功!")
print(f" - 书籍: {book.title}")
print(f" - 章节: {total_chapters}")
print(f" - 节: {total_sections}")
print(f" - 小节(知识点): {total_subsections}")
except Exception as e:
db.rollback()
print(f"[ERROR] 导入失败: {e}")
import traceback
traceback.print_exc()
raise
finally:
db.close()
if __name__ == "__main__":
import_book_structure()