commit 834cad729f7274f37536603381ea050f8d2b1a95 Author: xiaopeng <1509442308@qq.com> Date: Sat May 30 00:52:36 2026 +0800 init: KG_ICH 项目初始化 - data/: 非遗地理编码数据(GIS shapefile + CSV) - dofile/kg_project/: 知识图谱构建代码(纳入主仓库) - dofile/visulization/: 可视化数据与路线图 - officefile/: 文献、草稿、bib 文档 - officefile/latex/: Overleaf 同步目录(独立管理,不纳入) - output/: 输出目录 - logs/: 日志目录 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..5e15575 --- /dev/null +++ b/.gitignore @@ -0,0 +1,29 @@ +# === 工具与缓存 === +.claude/ +.playwright-mcp/ +*.log + +# === Python 虚拟环境 === +.venv/ +venv/ +__pycache__/ +*.py[cod] + +# === Obsidian & Pandoc === +.obsidian/ +.pandoc/ + +# === Overleaf Git(仅 latex/ 同步)=== +officefile/latex/.git/ + +# === IDE / OS === +.vscode/ +.idea/ +.DS_Store +Thumbs.db + +# === 旧的嵌套 Git 历史 === +.git-old-local/ + +# === 大型下载文件 === +downloads/ diff --git a/data/exp1.xlsx b/data/exp1.xlsx new file mode 100644 index 0000000..2257916 Binary files /dev/null and b/data/exp1.xlsx differ diff --git a/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.dbf b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.dbf new file mode 100644 index 0000000..8533f8c Binary files /dev/null and b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.dbf differ diff --git a/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.prj b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.prj new file mode 100644 index 0000000..f45cbad --- /dev/null +++ b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.prj @@ -0,0 +1 @@ +GEOGCS["GCS_WGS_1984",DATUM["D_WGS_1984",SPHEROID["WGS_1984",6378137.0,298.257223563]],PRIMEM["Greenwich",0.0],UNIT["Degree",0.0174532925199433]] \ No newline at end of file diff --git a/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.sbn b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.sbn new file mode 100644 index 0000000..1b969db Binary files /dev/null and b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.sbn differ diff --git a/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.sbx b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.sbx new file mode 100644 index 0000000..ce8dee7 Binary files /dev/null and b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.sbx differ diff --git a/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.shp b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.shp new file mode 100644 index 0000000..1b91047 Binary files /dev/null and b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.shp differ diff --git a/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.shx b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.shx new file mode 100644 index 0000000..a5b98c7 Binary files /dev/null and b/data/gis/XY黑龙江国家级和省级非遗名单-地理编码-最终.shx differ diff --git a/data/gis/黑龙江国家级和省级非遗名单-地理编码-最终.csv b/data/gis/黑龙江国家级和省级非遗名单-地理编码-最终.csv new file mode 100644 index 0000000..ee44062 --- /dev/null +++ b/data/gis/黑龙江国家级和省级非遗名单-地理编码-最终.csv @@ -0,0 +1,269 @@ +,Ŀλ,bd09_lat,bd09_lon,geocode_status,wgs84_lon,wgs84_lat +1,ʡо,45.75582106,126.6438083,success,126.6312812,45.74801106 +2,÷˹ӶĻ,47.31554957,123.7595409,success,123.7461629,47.3074024 +3,ںо,50.25127231,127.5354899,success,127.5217199,50.24367861 +4,ʡȺ,45.77151098,126.6480975,success,126.6355371,45.76377588 +5,ϷԺ,47.73331846,128.8475464,success,128.8339473,47.72523303 +6,Ļ,45.85775844,128.8356337,success,128.8222747,45.85006293 +7,Ļ,44.0673227,131.1294821,success,131.1152579,44.05891173 +8,̨зĻŲ,45.78213277,131.0196313,success,131.005397,45.77381386 +9,ԴطĻŲ,45.52415291,125.0845726,success,125.0711585,45.51602736 +10,˫зĻŲ,45.38811152,126.3196231,success,126.3067329,45.38019948 +11,峣Ļ,44.94382191,127.1776152,success,127.1647082,44.93562481 +12,Ժ,47.45690384,126.9365086,success,126.9236728,47.44873835 +13,绯бĹ,46.63862198,126.97987,success,126.9671481,46.6304274 +14,طĻŲ,45.71009294,125.2906327,success,125.2775591,45.70163614 +15,ʡ,47.18896666,124.8840662,success,124.8709297,47.18117849 +16,˫Ļ,45.38916578,126.320916,success,126.3080285,45.38123422 +17,绯Ļ,47.24251579,127.1205151,success,127.1070965,47.23415207 +18,,48.24741953,126.4973797,success,126.4843479,48.23891986 +19,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +20,׸Ⱥ,47.34926432,130.2840214,success,130.2696228,47.34108801 +21,ʡ޹˾,45.76809,126.66036,success,126.6477152,45.76047629 +22,봨Ļ,47.03658466,130.7293638,success,130.7149988,47.02857188 +23,ʡӼ,45.73284876,126.6106653,success,126.5983848,45.72455485 +24,Ϸ,47.36444108,123.9339047,success,123.9208401,47.35644797 +25,жŶɹز,46.86876776,124.4493588,success,124.4361789,46.86024699 +26,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +27,峣Ļ,44.94382191,127.1776152,success,127.1647082,44.93562481 +28,,45.52847968,126.9746364,success,126.9621138,45.52006942 +29,Ļ,44.34538409,129.476914,success,129.4632135,44.33677404 +30,˫ѼɽĺĻ,46.81581498,134.0176875,success,134.0037288,46.80749762 +31,ʡͳѧо,45.74792984,126.6696528,success,126.6569785,45.74030715 +32,ʡϻ,45.74338857,126.6775305,success,126.664857,45.73569396 +33,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +34,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +35,ŶɹĻ,46.87300908,124.4529011,success,124.4397277,46.86451077 +36,ʡо,45.75582106,126.6438083,success,126.6312812,45.74801106 +37,峣зĻŲ,44.93784286,127.1735288,success,127.1605796,44.92971861 +38,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +39,ռЭ,45.5542753,126.9643565,success,126.9519054,45.54586951 +40,Ļ,44.34538409,129.476914,success,129.4632135,44.33677404 +41,Ⱥ,47.26608314,132.0412614,success,132.0268239,47.25766497 +42,Ļ,44.0673227,131.1294821,success,131.1152579,44.05891173 +43,˫Ļ,45.38916578,126.320916,success,126.3080285,45.38123422 +44,Ļ,46.33126029,129.5745197,success,129.560623,46.32343594 +45,ԣĻ,47.80583728,124.4834558,success,124.4700126,47.79765568 +46,Ļ,46.26732644,126.2978939,success,126.2848964,46.25969498 +47,ԣĻ,47.80583728,124.4834558,success,124.4700126,47.79765568 +48,ĵзĻŲЭ,44.58185651,129.6111013,success,129.597687,44.57353907 +49,طĻŲ,45.71009294,125.2906327,success,125.2775591,45.70163614 +50,ɽĻ,45.55448623,131.8864766,success,131.8728149,45.5459851 +51,ҩȪо,48.66257739,126.1514718,success,126.1379422,48.6542552 +52,ګзĻŲ,48.47252804,124.8891678,success,124.8758643,48.46479829 +53,ĵȺ,44.58853502,129.625767,success,129.6122832,44.58038211 +54,Ļ,47.34701886,123.9844157,success,123.9713027,47.3386483 +55,ԴĻ,45.52415291,125.0845726,success,125.0711585,45.51602736 +56,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +57,зĻŲ,45.771343,126.648475,success,126.6359114,45.76361335 +58,ľ˹нĻŲ,46.80568999,130.3273591,success,130.3132017,46.79707441 +59,ľ˹Ⱥ,46.825782,130.369945,success,130.3554282,46.81771885 +60,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +61,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +62,˰Ⱥ,50.42823532,124.138259,success,124.1240089,50.42089502 +63,Ļ,45.71622277,125.2790967,success,125.2660297,45.70779699 +64,о,47.73331846,128.8475464,success,128.8339473,47.72523303 +65,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +66,жŶɹز,46.86876776,124.4493588,success,124.4361789,46.86024699 +67,˫ѼɽĺĻ,46.81581498,134.0176875,success,134.0037288,46.80749762 +68,ľ˹нĻŲ,46.80568999,130.3273591,success,130.3132017,46.79707441 +69,ŶɹĻ,46.87300908,124.4529011,success,124.4397277,46.86451077 +70,аּЭ,45.54991475,126.9827822,success,126.9702047,45.54158713 +71,÷˹ӶĻ,47.31554957,123.7595409,success,123.7461629,47.3074024 +72,ںаĻ,50.25261678,127.5213824,success,127.5074536,50.2452366 +73,ľ˹Ⱥ,47.26608314,132.0412614,success,132.0268239,47.25766497 +74,ĵг,44.58301369,129.6111989,success,129.5977853,44.57469723 +75,ĵг,44.58301369,129.6111989,success,129.5977853,44.57469723 +76,зĻŲЭ,44.34698358,129.489368,success,129.4757255,44.3383656 +77,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +78,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +79,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +80,ֵĻ,47.18968604,124.8832976,success,124.8701555,47.1819085 +81,̨ӺĻ,45.79123818,131.0744806,success,131.06004,45.78282072 +82,÷˹ӶĻ,47.31554957,123.7595409,success,123.7461629,47.3074024 +83,˰Ⱥ,50.42823532,124.138259,success,124.1240089,50.42089502 +84,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +85,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +86,˰Ⱥ,50.42823532,124.138259,success,124.1240089,50.42089502 +87,˫ѼɽĺĻ,46.81581498,134.0176875,success,134.0037288,46.80749762 +88,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +89,̩Ļ,46.39782929,123.4236263,success,123.4101272,46.39017044 +90,а,45.52847968,126.9746364,success,126.9621138,45.52006942 +91,ګĻ,48.47252804,124.8891678,success,124.8758643,48.46479829 +92,ͬȺ,47.64798068,132.5175095,success,132.5034909,47.63953006 +93,ľ˹нĻŲ,46.80568999,130.3273591,success,130.3132017,46.79707441 +94,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +95,ĵг,44.58301369,129.6111989,success,129.5977853,44.57469723 +96,Ⱥ,48.89498347,130.4105555,success,130.3955959,48.88668004 +97,Ļ,48.89407887,130.4071028,success,130.3921184,48.88584037 +98,ĵг,44.58301369,129.6111989,success,129.5977853,44.57469723 +99,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +100,ĵг,44.58301369,129.6111989,success,129.5977853,44.57469723 +101,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +102,г,45.76766336,126.6190306,success,126.6067046,45.75944138 +103,Ļ,44.0673227,131.1294821,success,131.1152579,44.05891173 +104,Ļ,47.87611042,123.7061596,success,123.6928614,47.86778014 +105,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +106,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +107,峣Ļ,44.94382191,127.1776152,success,127.1647082,44.93562481 +108,Ļ,44.34538409,129.476914,success,129.4632135,44.33677404 +109,Ⱥ,46.59363318,125.1086576,success,125.0949471,46.58589537 +110,Ļ,46.68862512,126.1085948,success,126.0953877,46.68024011 +111,Ⱥ,45.30677605,130.9438395,success,130.9300798,45.29843869 +112,찲Ļ,46.88574447,127.5146122,success,127.5012918,46.87802058 +113,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +114,ɽĻ,48.04038692,125.8798207,success,125.8667532,48.03198628 +115,ú·ٺչ,46.64173096,124.8669754,success,124.8538876,46.63433587 +116,о,47.73331846,128.8475464,success,128.8339473,47.72523303 +117,Ļ,46.84411964,126.4760837,success,126.463194,46.83564973 +118,ͯԺ,45.79403096,126.6624385,success,126.6498027,45.78643997 +119,˫ƤӰ,45.38811152,126.3196231,success,126.3067329,45.38019948 +120,ʡ,45.78924165,126.6522311,success,126.6396516,45.78157271 +121,ľ˹,46.80568999,130.3273591,success,130.3132017,46.79707441 +122,ʡԺ,45.75884697,126.6553175,success,126.6426998,45.75119447 +123,зĻ,45.85775844,128.8356337,success,128.8222747,45.85006293 +124,Ļ,47.45690384,126.9365086,success,126.9236728,47.44873835 +125,绯Ļ,46.26732644,126.2978939,success,126.2848964,46.25969498 +126,Ļ,51.73086307,126.6704831,success,126.6566811,51.72372518 +127,ľ˹Э,46.80568999,130.3273591,success,130.3132017,46.79707441 +128,Ļо,46.599149,125.175051,success,125.1617004,46.59078096 +129,ɳĻ,47.32357698,123.9643762,success,123.9513764,47.31510979 +130,Ӣֽ,46.45796584,125.3141247,success,125.3007241,46.44995417 +131,,48.24741953,126.4973797,success,126.4843479,48.23891986 +132,,48.5145725,126.200389,success,126.1868924,48.50670807 +133,˴ĻоԺ,45.76049549,126.6534938,success,126.6408885,45.75282515 +134,Ļ,44.34538409,129.476914,success,129.4632135,44.33677404 +135,ںаĻ,50.25261678,127.5213824,success,127.5074536,50.2452366 +136,ľ˹Ⱥ,47.26608314,132.0412614,success,132.0268239,47.25766497 +137,Ⱥ,45.77151098,126.6480975,success,126.6355371,45.76377588 +138,Ļ,47.24251579,127.1205151,success,127.1070965,47.23415207 +139,,48.24741953,126.4973797,success,126.4843479,48.23891986 +140,˰Ⱥ,52.34030508,124.7165125,success,124.7022813,52.33289247 +141,˶رо,48.04824416,126.2553867,success,126.2423338,48.03968479 +142,Ⱥ,47.34737244,123.9514263,success,123.938436,47.33903986 +143,ĵȺ,44.58853502,129.625767,success,129.6122832,44.58038211 +144,,48.24741953,126.4973797,success,126.4843479,48.23891986 +145,зĻŲЭ,44.34698358,129.489368,success,129.4757255,44.3383656 +146,ԴĻ,45.52415291,125.0845726,success,125.0711585,45.51602736 +147,аռЭ,45.5542753,126.9643565,success,126.9519054,45.54586951 +148,Ļ,46.45796584,125.3141247,success,125.3007241,46.44995417 +149,ʷܴ,45.713621,126.669629,success,126.6569888,45.70600004 +150,аռЭ,45.5542753,126.9643565,success,126.9519054,45.54586951 +151,ֹ,45.76574557,126.6797049,success,126.667029,45.75802654 +152,йι˾,45.76574557,126.6797049,success,126.667029,45.75802654 +153,ͬȺ,47.64798068,132.5175095,success,132.5034909,47.63953006 +154,Ⱥ,47.34737244,123.9514263,success,123.938436,47.33903986 +155,Ⱥ,47.34737244,123.9514263,success,123.938436,47.33903986 +156,о,47.73331846,128.8475464,success,128.8339473,47.72523303 +157,ԭĻ,46.7370677,129.9221257,success,129.9080552,46.729408 +158,зĻŲ,46.58901729,125.1623722,success,125.1490055,46.58060067 +159,Ļ,47.24251579,127.1205151,success,127.1070965,47.23415207 +160,Ϸ,45.77145132,126.6473829,success,126.6348284,45.76370546 +161,Ļ,48.89407887,130.4071028,success,130.3921184,48.88584037 +162,֮Ļι˾,45.753032,126.687176,success,126.6745198,45.74517987 +163,ʡ岩,45.78172497,126.6814856,success,126.6688227,45.77398752 +164,ںа,50.25550311,127.515815,success,127.5018354,50.24817418 +165,˰Ⱥ,50.42823532,124.138259,success,124.1240089,50.42089502 +166,˫ѼɽĺĻ,46.81581498,134.0176875,success,134.0037288,46.80749762 +167,ںзĻŲ,50.2480745,127.5204212,success,127.5064827,50.24070154 +168,˰Ⱥ,50.42823532,124.138259,success,124.1240089,50.42089502 +169,ĵȺ,44.58853502,129.625767,success,129.6122832,44.58038211 +170,׹˾,45.80882583,126.5416151,success,126.5289594,45.80121771 +171,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +172,˫Ļ,45.38916578,126.320916,success,126.3080285,45.38123422 +173,ʳƷ,45.67775619,126.616705,success,126.6044307,45.66949712 +174,˹ʳƷι˾,45.63337857,126.817157,success,126.8045267,45.62528875 +175,޹˾,45.53244496,126.5343041,success,126.5216586,45.52468904 +176,϶ʳƷ޹˾,45.91133879,126.5535234,success,126.5408838,45.9037732 +177,϶һι˾,45.80882583,126.5416151,success,126.5289594,45.80121771 +178,ϳҵ̨ʳ,45.70992736,126.6667568,success,126.6541261,45.70231692 +179,,45.52847968,126.9746364,success,126.9621138,45.52006942 +180,˶ظҵЭ,48.0400769,126.2541175,success,126.2410659,48.03150272 +181,ʡζƷҵЭ,45.74792984,126.6696528,success,126.6569785,45.74030715 +182,Ļ,47.24251579,127.1205151,success,127.1070965,47.23415207 +183,ĵȺ,44.58853502,129.625767,success,129.6122832,44.58038211 +184,о,47.73331846,128.8475464,success,128.8339473,47.72523303 +185,˫а׾ƹо,45.38811152,126.3196231,success,126.3067329,45.38019948 +186,ּ޹˾,45.751576,126.676666,success,126.6639869,45.74389303 +187,ʡԣϽѾҵι˾,45.7401911,126.6824096,success,126.6697471,45.73242424 +188,찲Ļ,46.88574447,127.5146122,success,127.5012918,46.87802058 +189,ʡȪҵι˾,45.42151265,127.167268,success,127.1541858,45.41341576 +190,˫йضҵ޹˾,45.38811152,126.3196231,success,126.3067329,45.38019948 +191,ĵȺ,44.58853502,129.625767,success,129.6122832,44.58038211 +192,ʡ;ҵ޹˾,47.06606258,126.6818311,success,126.6689398,47.0582359 +193,Ļ,44.92398753,130.5362661,success,130.5225358,44.91590669 +194,̨Ⱥ,45.78174652,131.0466262,success,131.0323887,45.77302433 +195,ĵȺݡĻվ,44.12687529,129.1759754,success,129.1623219,44.11882468 +196,Ļ,44.92398753,130.5362661,success,130.5225358,44.91590669 +197,ĺطĻŲ,46.80418274,134.0204689,success,134.006533,46.79580832 +198,Ļ,47.24251579,127.1205151,success,127.1070965,47.23415207 +199,ʡ˲Ʒ޹˾,45.74792984,126.6696528,success,126.6569785,45.74030715 +200,ϺĻо,45.631599,126.52717,success,126.5145109,45.62373057 +201,Ļ,47.91489855,125.3204008,success,125.3066585,47.90666647 +202,绯Ļ޹˾,47.3142979,128.007295,success,127.9939082,47.30614389 +203,,48.5145725,126.200389,success,126.1868924,48.50670807 +204,峣Ļ,44.94382191,127.1776152,success,127.1647082,44.93562481 +205,ɻ˲Ʒ޹˾,45.80882583,126.5416151,success,126.5289594,45.80121771 +206,̨Ӣ۹յ޹˾,45.77630032,131.0115446,success,130.997287,45.76816744 +207,乤Э,45.5542753,126.9643565,success,126.9519054,45.54586951 +208,ʡĻչ,45.30842136,130.9898146,success,130.9756641,45.30060499 +209,еĻ,45.77853084,126.6146262,success,126.6023267,45.77027229 +210,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +211,ںаĻ,50.25261678,127.5213824,success,127.5074536,50.2452366 +212,Ļ,51.73086307,126.6704831,success,126.6566811,51.72372518 +213,Ļ,45.76361288,127.4922455,success,127.479025,45.75572369 +214,̨ӻ޹˾,45.77630032,131.0115446,success,130.997287,45.76816744 +215,ԵռҾ߳,47.707191,128.958087,success,128.9443187,47.69924788 +216,̨оÿ¼ľշ,45.77630032,131.0115446,success,130.997287,45.76816744 +217,ԶĻ,48.37363711,134.3129708,success,134.2987716,48.36515205 +218,һĻý޹˾,45.8815795,126.5441568,success,126.5314593,45.87399685 +219,ʡװо,45.74792984,126.6696528,success,126.6569785,45.74030715 +220,ʡͼ,45.36780947,126.3090899,success,126.2962097,45.36000527 +221,ںаĻ,50.25261678,127.5213824,success,127.5074536,50.2452366 +222,ںط羰ĻŲ,48.70259497,126.2714585,success,126.2582276,48.69441304 +223,ګĻ,48.47252804,124.8891678,success,124.8758643,48.46479829 +224,ĵг,44.583022,129.611217,success,129.5978034,44.57470563 +225,г,45.30254683,130.9731084,success,130.9590567,45.29468866 +226,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +227,ŶɹĻ,46.87300908,124.4529011,success,124.4397277,46.86451077 +228,,45.52847968,126.9746364,success,126.9621138,45.52006942 +229,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +230,ͬȺ,47.64798068,132.5175095,success,132.5034909,47.63953006 +231,Ļ,46.25812558,126.2964168,success,126.2834192,46.25047767 +232,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +233,,45.52847968,126.9746364,success,126.9621138,45.52006942 +234,ռЭ,45.5542753,126.9643565,success,126.9519054,45.54586951 +235,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +236,иĻ,47.21590169,123.6458875,success,123.6327132,47.2080677 +237,÷˹ӶĻ,47.31554957,123.7595409,success,123.7461629,47.3074024 +238,ںаĻ,50.25261678,127.5213824,success,127.5074536,50.2452366 +239,иԣ,47.77758936,124.4766437,success,124.4632343,47.76925486 +240,ԴĻ,45.52415291,125.0845726,success,125.0711585,45.51602736 +241,зĻŲЭ,44.59987197,129.3874268,success,129.3735685,44.59140926 +242,ĺĻ,46.80853529,134.031975,success,134.0180638,46.79991581 +243,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +244,зĻŲЭ,44.34698358,129.489368,success,129.4757255,44.3383656 +245,÷˹ӶĻ,47.31554957,123.7595409,success,123.7461629,47.3074024 +246,ľĻ,45.9484543,128.0610095,success,128.047872,45.93995843 +247,Ļ,46.26732644,126.2978939,success,126.2848964,46.25969498 +248,ںڽо,50.25127231,127.5354899,success,127.5217199,50.24367861 +249,˰Ⱥ,50.42823532,124.138259,success,124.1240089,50.42089502 +250,ŶɹĻ,46.87300908,124.4529011,success,124.4397277,46.86451077 +251,жŶɹز,46.86876776,124.4493588,success,124.4361789,46.86024699 +252,ͬȺ,47.64798068,132.5175095,success,132.5034909,47.63953006 +253,,45.52847968,126.9746364,success,126.9621138,45.52006942 +254,ѧ,45.5542753,126.9643565,success,126.9519054,45.54586951 +255,Ļ,45.76361288,127.4922455,success,127.479025,45.75572369 +256,ʡ岩,45.78172497,126.6814856,success,126.6688227,45.77398752 +257,,45.52847968,126.9746364,success,126.9621138,45.52006942 +258,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +259,Ļ,44.34538409,129.476914,success,129.4632135,44.33677404 +260,,45.52847968,126.9746364,success,126.9621138,45.52006942 +261,Ļ,51.73086307,126.6704831,success,126.6566811,51.72372518 +262,Ļ,51.73086307,126.6704831,success,126.6566811,51.72372518 +263,Ļ,51.73086307,126.6704831,success,126.6566811,51.72372518 +264,ѷز,49.56949144,128.4855846,success,128.4716087,49.56148024 +265,Ļ,44.35308924,129.4761375,success,129.4624379,44.34449078 +266,ľ˹нĻŲ,46.80568999,130.3273591,success,130.3132017,46.79707441 +267,Ļ,44.5780916,129.3953204,success,129.3813797,44.56976716 +268,ĵȺ,44.58853502,129.625767,success,129.6122832,44.58038211 diff --git a/data/retry_geocoding.py b/data/retry_geocoding.py new file mode 100644 index 0000000..951f479 --- /dev/null +++ b/data/retry_geocoding.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python3 +# -*- coding: utf-8 -*- +import pandas as pd +import json +import time +import os +from urllib.request import urlopen, quote + +# 添加skill scripts到路径 +skill_dir = r'C:\Users\xiaopeng\.claude\skills\geocoding-cn\scripts' +if skill_dir not in os.sys.path: + os.sys.path.insert(0, skill_dir) + +from coordinate_transform import CoordinateTransformer + +# 读取数据 +input_file = r'E:\Project\2026_KG_ICH\data\黑龙江国家级和省级非遗名单-地理编码-最终.xlsx' +df = pd.read_excel(input_file) + +# 百度地图API配置 +AK = "L5SlQ1Kwmg6zaESvmc6RKG37yK2va7Ry" +BASE_URL = 'https://api.map.baidu.com/geocoding/v3/' + +def geocode_with_retry(address): + """尝试多次地理编码,使用不同的地址格式""" + attempts = [ + address, # 原始地址 + f"黑龙江省{address}", # 添加省名 + f"{address}黑龙江", # 省名在后 + f"中国黑龙江省{address}", # 添加国家 + ] + + for attempt in attempts: + try: + encoded_address = quote(attempt) + url = f'{BASE_URL}?address={encoded_address}&output=json&ak={AK}' + req = urlopen(url, timeout=10) + response = req.read().decode() + result = json.loads(response) + + if result['status'] == 0: + location = result['result']['location'] + return location['lat'], location['lng'], attempt + except Exception as e: + continue + + time.sleep(0.2) + + return None, None, None + +# 问题记录索引(0-based) +problematic_indices = [20, 127, 148, 161, 185, 199, 214, 58, 223] + +print("=== 重新地理编码问题记录 ===\n") + +success_count = 0 +for idx in problematic_indices: + if idx >= len(df): + continue + + row = df.iloc[idx] + original_address = str(row.iloc[1]) # 项目保护单位列 + + print(f"Row {idx+1}: {original_address}") + print(f" Old coords: {row['wgs84_lat']:.4f}N, {row['wgs84_lon']:.4f}E") + + # 尝试重新地理编码 + lat, lon, used_address = geocode_with_retry(original_address) + + if lat and lon: + # 转换为WGS84 + wgs_lon, wgs_lat = CoordinateTransformer.bd09_to_wgs84(lon, lat) + + print(f" New BD09: {lat:.4f}N, {lon:.4f}E") + print(f" New WGS84: {wgs_lat:.4f}N, {wgs_lon:.4f}E") + print(f" Used address: {used_address}") + + # 检查是否在合理范围内 + if 43 <= wgs_lat <= 53 and 121 <= wgs_lon <= 135: + print(f" OK: Within Heilongjiang range") + # 更新数据 + df.at[idx, 'bd09_lat'] = lat + df.at[idx, 'bd09_lon'] = lon + df.at[idx, 'wgs84_lat'] = wgs_lat + df.at[idx, 'wgs84_lon'] = wgs_lon + df.at[idx, 'geocode_status'] = 'success' + success_count += 1 + else: + print(f" WARNING: Still outside range (43-53N, 121-135E)") + else: + print(f" FAILED: Geocoding failed") + + print() + time.sleep(0.3) + +# 保存更新后的文件 +output_file = r'E:\Project\2026_KG_ICH\data\黑龙江国家级和省级非遗名单-地理编码-最终-修正.xlsx' +df.to_excel(output_file, index=False) +print(f"\n{'='*60}") +print(f"Correction complete!") +print(f"Successfully corrected: {success_count}/{len(problematic_indices)} records") +print(f"Saved to: {output_file}") diff --git a/data/黑龙江国家级和省级非遗名单-地理编码-初次尝试.xlsx b/data/黑龙江国家级和省级非遗名单-地理编码-初次尝试.xlsx new file mode 100644 index 0000000..abd7d7f Binary files /dev/null and b/data/黑龙江国家级和省级非遗名单-地理编码-初次尝试.xlsx differ diff --git a/data/黑龙江国家级和省级非遗名单-地理编码-最终-更新.xlsx b/data/黑龙江国家级和省级非遗名单-地理编码-最终-更新.xlsx new file mode 100644 index 0000000..b41771a Binary files /dev/null and b/data/黑龙江国家级和省级非遗名单-地理编码-最终-更新.xlsx differ diff --git a/data/黑龙江国家级和省级非遗名单.xlsx b/data/黑龙江国家级和省级非遗名单.xlsx new file mode 100644 index 0000000..388d19b Binary files /dev/null and b/data/黑龙江国家级和省级非遗名单.xlsx differ diff --git a/dofile/kg_project/.gitignore b/dofile/kg_project/.gitignore new file mode 100644 index 0000000..9bd1f0e --- /dev/null +++ b/dofile/kg_project/.gitignore @@ -0,0 +1,57 @@ +# Python +__pycache__/ +*.py[cod] +*$py.class +*.so +.Python + +# Virtual Environment +.venv/ +venv/ +ENV/ +env/ + +# IDE +.vscode/ +.idea/ +*.swp +*.swo +*~ + +# Logs +*.log +logs/ + +# OS +.DS_Store +Thumbs.db + +# Data files +data/ +output/ +logs/ + +# Configuration files (entire directory) +config/ + +# Neo4j (entire directory) +neo4j/ + +# Ontology files +ontology/ + +# Documentation +*.md + +# Batch files +*.bat + +# Jupyter +.ipynb_checkpoints/ +*.ipynb + +# Testing +.pytest_cache/ +.coverage +htmlcov/ +.tox/ diff --git a/dofile/kg_project/merge_kg_data.py b/dofile/kg_project/merge_kg_data.py new file mode 100644 index 0000000..d5bca1e --- /dev/null +++ b/dofile/kg_project/merge_kg_data.py @@ -0,0 +1,155 @@ +# -*- coding: utf-8 -*- +""" +整合基础政务实体和深层语义实体 +方案3:合并项目节点,完全整合 +""" + +import pandas as pd +import json +from pathlib import Path + +def merge_kg_data(): + """整合知识图谱数据""" + + output_dir = Path(__file__).parent / 'output' + + # 读取4个文件 + print("=== 读取原始文件 ===") + nodes_basic = pd.read_csv(output_dir / 'nodes.csv', encoding='utf-8-sig') + nodes_deep = pd.read_csv(output_dir / 'nodes_desc.csv', encoding='utf-8-sig') + rels_basic = pd.read_csv(output_dir / 'rels.csv', encoding='utf-8-sig') + rels_deep = pd.read_csv(output_dir / 'rels_desc.csv', encoding='utf-8-sig') + + print(f"基础节点: {len(nodes_basic)}") + print(f"深层节点: {len(nodes_deep)}") + print(f"基础关系: {len(rels_basic)}") + print(f"深层关系: {len(rels_deep)}") + + # 1. 合并ICH项目节点 + print("\n=== 合并ICH项目节点 ===") + basic_projects = nodes_basic[nodes_basic['type'] == 'ICH_Project'].copy() + deep_projects = nodes_deep[nodes_deep['type'] == 'ICH_Project'].copy() + + print(f"基础项目节点: {len(basic_projects)}") + print(f"深层项目节点: {len(deep_projects)}") + + # 合并项目节点的属性 + merged_projects = [] + + for pid in basic_projects['id']: + basic_row = basic_projects[basic_projects['id'] == pid].iloc[0] + + # 查找深层项目节点 + deep_row = deep_projects[deep_projects['id'] == pid] + + if len(deep_row) > 0: + # 合并属性 + deep_row = deep_row.iloc[0] + + # 解析属性JSON + basic_props = json.loads(basic_row['properties']) if basic_row['properties'] else {} + deep_props = json.loads(deep_row['properties']) if deep_row['properties'] else {} + + # 合并属性(深层属性优先) + merged_props = {**basic_props, **deep_props} + + merged_projects.append({ + 'id': pid, + 'label': basic_row['label'], # 保留基础标签 + 'type': 'ICH_Project', + 'properties': json.dumps(merged_props, ensure_ascii=False) + }) + else: + # 只在基础中存在 + merged_projects.append({ + 'id': pid, + 'label': basic_row['label'], + 'type': 'ICH_Project', + 'properties': basic_row['properties'] + }) + + print(f"合并后项目节点: {len(merged_projects)}") + + # 2. 合并其他节点(去除ICH_Project) + print("\n=== 合并其他节点 ===") + other_basic_nodes = nodes_basic[nodes_basic['type'] != 'ICH_Project'] + other_deep_nodes = nodes_deep[nodes_deep['type'] != 'ICH_Project'] + + all_nodes = pd.concat([ + pd.DataFrame(merged_projects), + other_basic_nodes, + other_deep_nodes + ], ignore_index=True) + + print(f"合并后总节点数: {len(all_nodes)}") + print(f" - ICH_Project: {len(merged_projects)}") + print(f" - 其他节点: {len(all_nodes) - len(merged_projects)}") + + # 3. 合并关系 + print("\n=== 合并关系 ===") + all_rels = pd.concat([rels_basic, rels_deep], ignore_index=True) + + # 去重关系 + all_rels = all_rels.drop_duplicates(subset=['source', 'target', 'type'], keep='first') + + print(f"合并后关系数: {len(all_rels)}") + + # 4. 统计信息 + print("\n=== 整合后统计 ===") + print(f"总节点数: {len(all_nodes)}") + print(f"总关系数: {len(all_rels)}") + + print("\n节点类型分布:") + for ntype, count in all_nodes.groupby('type').size().items(): + print(f" {ntype}: {count}") + + print(f"\n关系类型数量: {len(all_rels['type'].unique())}") + + # 5. 保存整合后的文件 + print("\n=== 保存整合文件 ===") + output_file_nodes = output_dir / 'kg_merged_nodes.csv' + output_file_rels = output_dir / 'kg_merged_rels.csv' + + all_nodes.to_csv(output_file_nodes, index=False, encoding='utf-8-sig') + all_rels.to_csv(output_file_rels, index=False, encoding='utf-8-sig') + + print(f"节点已保存: {output_file_nodes}") + print(f"关系已保存: {output_file_rels}") + + # 6. 验证 + print("\n=== 验证 ===") + + # 检查项目节点完整性 + project_count = len(all_nodes[all_nodes['type'] == 'ICH_Project']) + print(f"ICH项目节点数: {project_count} (应该是268)") + + # 检查关系完整性 + rel_sources = set(all_rels['source'].unique()) + node_ids = set(all_nodes['id'].unique()) + + missing_sources = rel_sources - node_ids + if missing_sources: + print(f"警告: {len(missing_sources)} 个关系的源节点不在节点文件中") + else: + print("所有关系的源节点都存在于节点文件中") + + # 统计每个项目的关系数 + project_rels = all_rels[all_rels['source'].str.startswith('ICH-')].groupby('source').size() + print(f"\n有关系的项目数: {len(project_rels)}/268") + + # 展示几个示例项目的统计 + print("\n示例项目关系统计:") + sample_projects = ['ICH-1', 'ICH-118', 'ICH-229'] + for pid in sample_projects: + basic_rel_count = len(rels_basic[rels_basic['source'] == pid]) + deep_rel_count = len(rels_deep[rels_deep['source'] == pid]) + total_rel_count = len(all_rels[all_rels['source'] == pid]) + print(f" {pid}: 基础{basic_rel_count} + 深层{deep_rel_count} = 总计{total_rel_count} 条关系") + + print("\n完成!") + + return all_nodes, all_rels + + +if __name__ == '__main__': + merge_kg_data() diff --git a/dofile/kg_project/merge_retry_results.py b/dofile/kg_project/merge_retry_results.py new file mode 100644 index 0000000..8a6de48 --- /dev/null +++ b/dofile/kg_project/merge_retry_results.py @@ -0,0 +1,203 @@ +# -*- coding: utf-8 -*- +""" +将重新抽取的结果合并到现有的节点和关系文件中 +""" + +import json +import pandas as pd +from pathlib import Path +import sys + +# 添加模块路径 +sys.path.append(str(Path(__file__).parent / 'src')) + +from data_processing.entity_normalizer import EntityNormalizer +from data_processing.relationship_builder import RelationshipBuilder + +def merge_retry_results(): + """合并重新抽取的结果""" + + # 文件路径 + retry_file = Path(__file__).parent / 'output' / 'retry_projects.json' + nodes_file = Path(__file__).parent / 'output' / 'nodes_llm.csv' + rels_file = Path(__file__).parent / 'output' / 'rels_llm.csv' + ontology_file = Path(__file__).parent / 'config' / 'entity_ontology.yaml' + + print("=== 加载数据 ===") + + # 读取重新抽取的结果 + with open(retry_file, 'r', encoding='utf-8') as f: + retry_data = json.load(f) + + print(f"重新抽取的项目数: {len(retry_data)}") + + # 读取现有的节点和关系 + existing_nodes_df = pd.read_csv(nodes_file, encoding='utf-8-sig') + existing_rels_df = pd.read_csv(rels_file, encoding='utf-8-sig') + + print(f"现有节点数: {len(existing_nodes_df)}") + print(f"现有关系数: {len(existing_rels_df)}") + + # 初始化规范化器和关系构建器 + normalizer = EntityNormalizer(str(ontology_file)) + builder = RelationshipBuilder(str(ontology_file)) + + # 规范化重新抽取的实体 + print("\n=== 规范化实体 ===") + + # 收集所有需要规范化的实体 + all_entities_to_normalize = [] + for item in retry_data: + result = item['extraction_result'] + entities = result.get('entities', []) + # 添加项目ID以便跟踪 + for entity in entities: + entity['_project_id'] = item['project_id'] + all_entities_to_normalize.extend(entities) + + # 批量规范化 + extraction_results = [] + for item in retry_data: + extraction_results.append(item['extraction_result']) + + normalizer.normalize_batch(extraction_results) + + # 生成新的实体节点 + new_entity_nodes = normalizer.get_entity_nodes() + print(f"新生成实体节点数: {len(new_entity_nodes)}") + + # 添加项目节点 + new_project_nodes = [] + for item in retry_data: + new_project_nodes.append({ + 'id': item['project_id'], + 'label': item['project_name'], + 'type': 'ICH_Project', + 'properties': '{}' + }) + + print(f"新增项目节点数: {len(new_project_nodes)}") + + # 构建关系 + print("\n=== 构建关系 ===") + all_new_relationships = [] + + # 获取当前所有节点的ID到标签映射(用于查找目标实体ID) + all_nodes_for_lookup = pd.concat([ + existing_nodes_df, + pd.DataFrame(new_entity_nodes) + ], ignore_index=True) + + # 创建名称->ID的映射字典 + name_to_id = {} + for idx, row in all_nodes_for_lookup.iterrows(): + if row['type'] != 'ICH_Project': # 只映射实体节点 + name_to_id[(row['type'], row['label'])] = row['id'] + + for item in retry_data: + project_id = item['project_id'] + result = item['extraction_result'] + + entities = result.get('entities', []) + relationships = result.get('relationships', []) + + # 构建实体名称到ID的映射(仅限当前项目) + entity_name_to_id = {} + for entity in entities: + # 实体名称可能在name字段或attributes.name字段 + entity_name = entity.get('name') or entity.get('attributes', {}).get('name', '') + entity_type = entity.get('type', '') + normalized_name = normalizer.normalize_text(entity_name) + + # 在规范化器的映射中查找(键是normalized_text,不是元组) + entity_id = normalizer.text_to_id_map.get(normalized_name) + if not entity_id: + # 在所有节点中查找(可能是已存在的实体) + entity_id = name_to_id.get((entity_type, entity_name)) + + if entity_id: + entity_name_to_id[entity_name] = entity_id + + # 手动构建关系 + for rel in relationships: + # 关系中的字段名是source_entity和target_entity + source_name = rel.get('source_entity') or rel.get('source') + target_name = rel.get('target_entity') or rel.get('target') + rel_type = rel.get('type') + rel_props = rel.get('properties', {}) + + # 查找源实体ID + if source_name == project_id: + source_id = project_id + else: + # 先尝试在当前项目实体中查找 + source_id = entity_name_to_id.get(source_name) + if not source_id: + # 在规范化器的映射中查找 + normalized_name = normalizer.normalize_text(source_name) + source_id = normalizer.text_to_id_map.get(normalized_name) + + # 查找目标实体ID + # 先尝试在当前项目实体中查找 + target_id = entity_name_to_id.get(target_name) + if not target_id: + # 在规范化器的映射中查找 + normalized_name = normalizer.normalize_text(target_name) + target_id = normalizer.text_to_id_map.get(normalized_name) + + # 如果都找到了,添加关系 + if source_id and target_id: + all_new_relationships.append({ + 'source': source_id, + 'target': target_id, + 'type': rel_type, + 'properties': json.dumps(rel_props, ensure_ascii=False) if rel_props else '{}' + }) + + print(f"项目 {project_id}: {len([r for r in all_new_relationships if r['source'] == project_id])} 条关系") + + print(f"新增关系总数: {len(all_new_relationships)}") + + # 合并节点 + print("\n=== 合并节点 ===") + # 过滤掉已经存在的项目节点 + existing_project_ids = set(existing_nodes_df[existing_nodes_df['type'] == 'ICH_Project']['id'].tolist()) + new_project_nodes_filtered = [n for n in new_project_nodes if n['id'] not in existing_project_ids] + + all_nodes = pd.concat([ + existing_nodes_df, + pd.DataFrame(new_entity_nodes), + pd.DataFrame(new_project_nodes_filtered) + ], ignore_index=True) + + print(f"合并后节点数: {len(all_nodes)} (新增 {len(new_entity_nodes) + len(new_project_nodes_filtered)} 个)") + + # 合并关系 + print("\n=== 合并关系 ===") + all_rels = pd.concat([ + existing_rels_df, + pd.DataFrame(all_new_relationships) + ], ignore_index=True) + + print(f"合并后关系数: {len(all_rels)} (新增 {len(all_new_relationships)} 条)") + + # 保存结果 + print("\n=== 保存结果 ===") + all_nodes.to_csv(nodes_file, index=False, encoding='utf-8-sig') + all_rels.to_csv(rels_file, index=False, encoding='utf-8-sig') + + print(f"节点已保存: {nodes_file}") + print(f"关系已保存: {rels_file}") + + # 验证 + print("\n=== 验证 ===") + for item in retry_data: + project_id = item['project_id'] + rel_count = len(all_rels[all_rels['source'] == project_id]) + print(f"{project_id}: {rel_count} 条关系") + + print("\n完成!") + + +if __name__ == '__main__': + merge_retry_results() diff --git a/dofile/kg_project/requirements.txt b/dofile/kg_project/requirements.txt new file mode 100644 index 0000000..10c2380 --- /dev/null +++ b/dofile/kg_project/requirements.txt @@ -0,0 +1,38 @@ +# 黑龙江省非物质文化遗产知识图谱构建 - 依赖包 + +# DeepSeek和LLM相关 +langchain-deepseek>=0.1.0 +langchain-core>=0.1.0 +openai>=1.0.0 # 备用 + +# Neo4j数据库 +neo4j>=5.15.0 +py2neo>=2021.2.4 + +# 数据处理 +pandas>=2.0.0 +numpy>=1.24.0 +openpyxl>=3.1.0 # Excel读取 + +# NLP和文本处理 +jieba>=0.42.0 +# 可选:深度学习框架 +# torch>=2.0.0 +# transformers>=4.30.0 + +# Web框架(用于后续开发) +# fastapi>=0.100.0 +# uvicorn>=0.23.0 +# flask>=3.0.0 + +# 可视化(用于后续开发) +# matplotlib>=3.7.0 +# seaborn>=0.12.0 + +# 配置和日志 +pyyaml>=6.0 +python-dotenv>=1.0.0 + +# 其他工具 +tqdm>=4.65.0 +requests>=2.31.0 diff --git a/dofile/kg_project/retry_failed_projects.py b/dofile/kg_project/retry_failed_projects.py new file mode 100644 index 0000000..6ed9697 --- /dev/null +++ b/dofile/kg_project/retry_failed_projects.py @@ -0,0 +1,92 @@ +# -*- coding: utf-8 -*- +""" +重新抽取失败项目的脚本 +""" + +import asyncio +import pandas as pd +import yaml +import json +from pathlib import Path +from datetime import datetime +import sys + +# 添加模块路径 +sys.path.append(str(Path(__file__).parent / 'src')) + +from knowledge_extraction.deep_entity_extractor import DeepEntityExtractor + +async def retry_failed_projects(): + """重新抽取失败的项目""" + + # 加载配置 + config_file = Path(__file__).parent / 'config' / 'deep_extraction_config.yaml' + with open(config_file, 'r', encoding='utf-8') as f: + config = yaml.safe_load(f) + + # 读取原始数据 + data_file = Path(__file__).parent.parent.parent / 'data' / '黑龙江国家级和省级非遗名单.xlsx' + df = pd.read_excel(data_file, engine='openpyxl') + + # 找到失败的项目 + # 第0列是序号,需要拼接成ICH-xxx格式 + failed_data = df[df.iloc[:, 0].isin([118, 229])] + + print(f"找到 {len(failed_data)} 个失败项目") + print("=" * 60) + + # 初始化抽取器 + extractor = DeepEntityExtractor(str(config_file)) + + results = [] + + for idx, row in failed_data.iterrows(): + project_num = int(row.iloc[0]) # 序号(118, 229) + project_id = f'ICH-{project_num}' # 拼接成ICH-xxx格式 + project_name = row.iloc[3] # 项目名称(第4列) + description = row.iloc[7] if len(row) > 7 else "" # 完整描述(第8列备注) + + print(f"\n正在抽取: {project_id} - {project_name}") + print(f"描述长度: {len(description)} 字符") + + # 准备输入数据 + input_data = { + 'project_id': project_id, + 'project_name': project_name, + 'description': description + } + + # 抽取实体和关系 + try: + result = await extractor.extract_from_remark( + project_id=project_id, + project_name=project_name, + remark_text=description + ) + + if result: + results.append({ + 'project_id': project_id, + 'project_name': project_name, + 'extraction_result': result + }) + print(f"[OK] 抽取成功: {len(result.get('entities', []))} 个实体, {len(result.get('relationships', []))} 条关系") + else: + print(f"[FAIL] 抽取失败") + + except Exception as e: + print(f"[ERROR] 抽取异常: {str(e)}") + + # 保存结果 + output_file = Path(__file__).parent / 'output' / 'retry_projects.json' + with open(output_file, 'w', encoding='utf-8') as f: + json.dump(results, f, ensure_ascii=False, indent=2) + + print(f"\n结果已保存到: {output_file}") + print(f"成功: {len(results)}/{len(failed_data)}") + + return results + + +if __name__ == '__main__': + asyncio.run(retry_failed_projects()) diff --git a/dofile/kg_project/retry_projects.py b/dofile/kg_project/retry_projects.py new file mode 100644 index 0000000..462f7d3 --- /dev/null +++ b/dofile/kg_project/retry_projects.py @@ -0,0 +1,241 @@ +# -*- coding: utf-8 -*- +""" +重新抽取失败项目的脚本 +""" + +import asyncio +import pandas as pd +import yaml +import json +from pathlib import Path +from datetime import datetime +import sys + +# 添加模块路径 +sys.path.append(str(Path(__file__).parent / 'src')) + +from knowledge_extraction.deep_entity_extractor import DeepEntityExtractor +from data_processing.entity_normalizer import EntityNormalizer +from data_processing.relationship_builder import RelationshipBuilder + +async def retry_and_merge(): + """重新抽取失败项目并直接合并到现有文件""" + + # 加载配置 + config_file = Path(__file__).parent / 'config' / 'deep_extraction_config.yaml' + with open(config_file, 'r', encoding='utf-8') as f: + config = yaml.safe_load(f) + + # 读取原始数据 + data_file = Path(__file__).parent.parent.parent / 'data' / '黑龙江国家级和省级非遗名单.xlsx' + df = pd.read_excel(data_file, engine='openpyxl') + + # 找到失败的项目 + failed_data = df[df.iloc[:, 0].isin([118, 229])] + + print(f"找到 {len(failed_data)} 个失败项目") + print("=" * 60) + + # 初始化组件 + extractor = DeepEntityExtractor(str(config_file)) + ontology_file = Path(__file__).parent / 'config' / 'entity_ontology.yaml' + normalizer = EntityNormalizer(str(ontology_file)) + builder = RelationshipBuilder(str(ontology_file)) + + # 读取现有的实体注册表(如果存在) + existing_nodes_file = Path(__file__).parent / 'output' / 'nodes_llm.csv' + existing_rels_file = Path(__file__).parent / 'output' / 'rels_llm.csv' + + existing_nodes_df = pd.read_csv(existing_nodes_file, encoding='utf-8-sig') + existing_rels_df = pd.read_csv(existing_rels_file, encoding='utf-8-sig') + + print(f"现有节点数: {len(existing_nodes_df)}") + print(f"现有关系数: {len(existing_rels_df)}") + + # 将现有实体加载到规范化器中 + print("\n=== 加载现有实体到规范化器 ===") + for idx, row in existing_nodes_df.iterrows(): + if row['type'] != 'ICH_Project': + # 将现有实体添加到规范化器的注册表 + entity_type = row['type'] + entity_text = row['label'] + entity_id = row['id'] + + # 规范化文本 + normalized_text = normalizer.normalize_text(entity_text) + + # 添加到映射表 + if normalized_text not in normalizer.text_to_id_map: + normalizer.text_to_id_map[normalized_text] = entity_id + normalizer.entity_registry[entity_id] = { + 'type': entity_type, + 'text': entity_text, + 'normalized_text': normalized_text, + 'canonical_name': normalized_text # 添加这个字段 + } + + print(f"已加载 {len(normalizer.entity_registry)} 个现有实体") + + # 重新抽取失败的项目 + extraction_results = [] + + for idx, row in failed_data.iterrows(): + project_num = int(row.iloc[0]) + project_id = f'ICH-{project_num}' + project_name = row.iloc[3] + description = row.iloc[7] if len(row) > 7 else "" + + print(f"\n正在抽取: {project_id} - {project_name}") + print(f"描述长度: {len(description)} 字符") + + try: + result = await extractor.extract_from_remark( + project_id=project_id, + project_name=project_name, + remark_text=description + ) + + if result: + extraction_results.append({ + 'project_id': project_id, + 'project_name': project_name, + 'extraction_result': result + }) + print(f"[OK] 抽取成功: {len(result.get('entities', []))} 个实体, {len(result.get('relationships', []))} 条关系") + else: + print(f"[FAIL] 抽取失败") + + except Exception as e: + print(f"[ERROR] 抽取异常: {str(e)}") + + if not extraction_results: + print("\n没有成功抽取的项目") + return + + # 规范化新抽取的实体(会自动去重) + print("\n=== 规范化新抽取的实体 ===") + for item in extraction_results: + result = item['extraction_result'] + entities = result.get('entities', []) + + for entity in entities: + # 实体名称可能在name字段或attributes.name字段 + entity_name = entity.get('name') or entity.get('attributes', {}).get('name', '') + entity_type = entity.get('type', '') + + # 规范化实体(会自动去重) + normalizer.normalize_entity(entity, similarity_threshold=0.85) + + # 获取所有实体节点(包括新增的) + all_entity_nodes = normalizer.get_entity_nodes() + print(f"规范化后实体节点数: {len(all_entity_nodes)}") + + # 构建关系 + print("\n=== 构建关系 ===") + all_new_relationships = [] + + for item in extraction_results: + project_id = item['project_id'] + result = item['extraction_result'] + + entities = result.get('entities', []) + relationships = result.get('relationships', []) + + # 构建实体名称到ID的映射 + entity_name_to_id = {} + for entity in entities: + entity_name = entity.get('name') or entity.get('attributes', {}).get('name', '') + entity_type = entity.get('type', '') + normalized_name = normalizer.normalize_text(entity_name) + entity_id = normalizer.text_to_id_map.get(normalized_name) + if entity_id: + entity_name_to_id[entity_name] = entity_id + + # 手动构建关系 + for rel in relationships: + source_name = rel.get('source_entity') or rel.get('source') + target_name = rel.get('target_entity') or rel.get('target') + rel_type = rel.get('type') + rel_props = rel.get('properties', {}) + + # 查找源实体ID + if source_name == project_id: + source_id = project_id + else: + normalized_name = normalizer.normalize_text(source_name) + source_id = normalizer.text_to_id_map.get(normalized_name) + + # 查找目标实体ID + normalized_name = normalizer.normalize_text(target_name) + target_id = normalizer.text_to_id_map.get(normalized_name) + + # 如果都找到了,添加关系 + if source_id and target_id: + all_new_relationships.append({ + 'source': source_id, + 'target': target_id, + 'type': rel_type, + 'properties': json.dumps(rel_props, ensure_ascii=False) if rel_props else '{}' + }) + + print(f"项目 {project_id}: {len([r for r in all_new_relationships if r['source'] == project_id])} 条关系") + + # 合并节点和关系 + print("\n=== 合并数据 ===") + + # 项目节点:只添加缺失的项目 + existing_project_ids = set(existing_nodes_df[existing_nodes_df['type'] == 'ICH_Project']['id'].tolist()) + new_project_nodes = [ + {'id': item['project_id'], 'label': item['project_name'], 'type': 'ICH_Project', 'properties': '{}'} + for item in extraction_results + if item['project_id'] not in existing_project_ids + ] + + # 合并所有节点 + all_nodes_df = pd.concat([ + existing_nodes_df[existing_nodes_df['type'] != 'ICH_Project'], # 现有实体节点 + pd.DataFrame(all_entity_nodes), # 所有实体节点(包括新增和去重后的现有) + pd.DataFrame(new_project_nodes), # 新增项目节点 + existing_nodes_df[existing_nodes_df['type'] == 'ICH_Project'] # 现有项目节点 + ], ignore_index=True) + + # 去重节点(按ID) + all_nodes_df = all_nodes_df.drop_duplicates(subset=['id'], keep='first') + + # 合并关系 + all_rels_df = pd.concat([ + existing_rels_df, + pd.DataFrame(all_new_relationships) + ], ignore_index=True) + + # 去重关系 + all_rels_df = all_rels_df.drop_duplicates(subset=['source', 'target', 'type'], keep='first') + + print(f"合并后节点数: {len(all_nodes_df)} (新增 {len(all_entity_nodes) - len(existing_nodes_df[existing_nodes_df['type'] != 'ICH_Project'])} 个实体)") + print(f"合并后关系数: {len(all_rels_df)} (新增 {len(all_new_relationships)} 条)") + + # 保存结果 + print("\n=== 保存结果 ===") + all_nodes_df.to_csv(existing_nodes_file, index=False, encoding='utf-8-sig') + all_rels_df.to_csv(existing_rels_file, index=False, encoding='utf-8-sig') + + print(f"节点已保存: {existing_nodes_file}") + print(f"关系已保存: {existing_rels_file}") + + # 验证 + print("\n=== 验证 ===") + for item in extraction_results: + project_id = item['project_id'] + rel_count = len(all_rels_df[all_rels_df['source'] == project_id]) + print(f"{project_id}: {rel_count} 条关系") + + print("\n完成!") + + # 保存抽取结果以供检查 + output_file = Path(__file__).parent / 'output' / 'retry_projects.json' + with open(output_file, 'w', encoding='utf-8') as f: + json.dump(extraction_results, f, ensure_ascii=False, indent=2) + + +if __name__ == '__main__': + asyncio.run(retry_and_merge()) diff --git a/dofile/kg_project/scripts/excel_reader.py b/dofile/kg_project/scripts/excel_reader.py new file mode 100644 index 0000000..efe32f1 --- /dev/null +++ b/dofile/kg_project/scripts/excel_reader.py @@ -0,0 +1,375 @@ +""" +数据预处理脚本 - 读取黑龙江非遗Excel数据 +""" + +import pandas as pd +import json +from pathlib import Path +from typing import Dict, List, Any +import logging + +class ExcelDataReader: + """Excel数据读取器""" + + def __init__(self, excel_path: str): + """ + 初始化数据读取器 + + Args: + excel_path: Excel文件路径 + """ + self.excel_path = Path(excel_path) + self.data = None + self.logger = self._setup_logger() + + def _setup_logger(self): + """设置日志""" + logging.basicConfig( + level=logging.INFO, + format='%(asctime)s - %(name)s - %(levelname)s - %(message)s' + ) + return logging.getLogger(__name__) + + def read_excel(self) -> pd.DataFrame: + """ + 读取Excel文件 + + Returns: + DataFrame: 数据框 + """ + try: + self.logger.info(f"开始读取Excel文件: {self.excel_path}") + + # 读取Excel文件 + self.data = pd.read_excel(self.excel_path) + + self.logger.info(f"成功读取 {len(self.data)} 行数据") + self.logger.info(f"列名: {list(self.data.columns)}") + + # 显示前5行 + self.logger.info("\n前5行数据:") + self.logger.info(self.data.head()) + + return self.data + + except Exception as e: + self.logger.error(f"读取Excel文件失败: {str(e)}") + raise + + def analyze_data(self) -> Dict[str, Any]: + """ + 分析数据 + + Returns: + Dict: 分析结果 + """ + if self.data is None: + raise ValueError("请先读取Excel文件") + + analysis = { + "total_records": len(self.data), + "columns": list(self.data.columns), + "column_types": {col: str(dtype) for col, dtype in self.data.dtypes.items()}, + "missing_values": self.data.isnull().sum().to_dict(), + "statistics": {} + } + + # 分析类别分布 + if '类别' in self.data.columns: + category_counts = self.data['类别'].value_counts() + analysis['category_distribution'] = category_counts.to_dict() + + # 分析级别分布 + if '项目级别' in self.data.columns: + level_counts = self.data['项目级别'].value_counts() + analysis['level_distribution'] = level_counts.to_dict() + + # 分析地域分布 + if '项目申报单位/地区' in self.data.columns: + location_counts = self.data['项目申报单位/地区'].value_counts() + analysis['location_distribution'] = location_counts.head(20).to_dict() + + # 传承人覆盖率 + if '代表性传承人' in self.data.columns: + has_inheritor = self.data['代表性传承人'].notna().sum() + analysis['inheritor_coverage'] = { + "total": len(self.data), + "has_inheritor": int(has_inheritor), + "coverage_rate": float(has_inheritor / len(self.data) * 100) + } + + return analysis + + def clean_data(self) -> pd.DataFrame: + """ + 清洗数据 + + Returns: + DataFrame: 清洗后的数据 + """ + if self.data is None: + raise ValueError("请先读取Excel文件") + + self.logger.info("开始清洗数据") + + # 去除空行 + original_len = len(self.data) + self.data = self.data.dropna(how='all') + self.logger.info(f"去除空行: {original_len} -> {len(self.data)}") + + # 填充缺失值 + for col in self.data.columns: + if self.data[col].dtype == 'object': + self.data[col] = self.data[col].fillna('') + else: + self.data[col] = self.data[col].fillna(0) + + # 去重(基于项目名称、项目批次、项目保护单位、代表性传承人四个字段) + dedup_cols = ['项目名称', '项目批次', '项目保护单位', '代表性传承人'] + existing_cols = [col for col in dedup_cols if col in self.data.columns] + + if existing_cols: + before_dedup = len(self.data) + self.data = self.data.drop_duplicates(subset=existing_cols, keep='first') + duplicate_count = before_dedup - len(self.data) + dedup_rate = (duplicate_count / before_dedup * 100) if before_dedup > 0 else 0 + self.logger.info(f"去重(基于{len(existing_cols)}个字段: {', '.join(existing_cols)}): {before_dedup} -> {len(self.data)} (删除{duplicate_count}条,去重率{dedup_rate:.1f}%)") + + return self.data + + def convert_to_kg_format(self) -> List[Dict[str, Any]]: + """ + 转换为知识图谱格式 + + Returns: + List[Dict]: 知识图谱节点列表 + """ + if self.data is None: + raise ValueError("请先读取Excel文件") + + self.logger.info("转换为知识图谱格式") + + kg_nodes = [] + + for idx, row in self.data.iterrows(): + try: + # 创建非遗项目节点 + project_node = { + "id": f"ICH-{idx:04d}", + "type": "ICH_Project", + "properties": { + "project_id": f"ICH-{idx:04d}", + "name": str(row.get('项目名称', '')), + "level": str(row.get('项目级别', '')), + "batch": str(row.get('批次号', '')), + "category": str(row.get('类别', '')), + "declaration_area": str(row.get('项目申报单位/地区', '')), + "description": str(row.get('备注', '')), + "source_row": idx + 2 # Excel行号(从2开始,第一行是表头) + } + } + + kg_nodes.append(project_node) + + except Exception as e: + self.logger.warning(f"转换第{idx}行数据失败: {str(e)}") + continue + + self.logger.info(f"成功转换 {len(kg_nodes)} 个节点") + return kg_nodes + + def extract_inheritors(self) -> List[Dict[str, Any]]: + """ + 提取传承人信息 + + Returns: + List[Dict]: 传承人节点列表 + """ + if self.data is None: + raise ValueError("请先读取Excel文件") + + self.logger.info("提取传承人信息") + + inheritors = [] + inheritor_id = 0 + + for idx, row in self.data.iterrows(): + inheritor_names = row.get('代表性传承人', '') + if pd.isna(inheritor_names) or not str(inheritor_names).strip(): + continue + + # 处理多个传承人(用顿号分隔) + names = str(inheritor_names).replace('、', ',').replace(',', ',').split(',') + + for name in names: + name = name.strip() + if not name: + continue + + inheritor_id += 1 + inheritor_node = { + "id": f"INH-{inheritor_id:04d}", + "type": "Inheritor", + "properties": { + "inheritor_id": f"INH-{inheritor_id:04d}", + "name": name, + "project_name": str(row.get('项目名称', '')), + "source_row": idx + 2 + } + } + + inheritors.append(inheritor_node) + + self.logger.info(f"提取了 {len(inheritors)} 个传承人") + return inheritors + + def extract_relations(self) -> List[Dict[str, Any]]: + """ + 提取关系 + + Returns: + List[Dict]: 关系列表 + """ + if self.data is None: + raise ValueError("请先读取Excel文件") + + self.logger.info("提取关系") + + relations = [] + relation_id = 0 + + for idx, row in self.data.iterrows(): + project_id = f"ICH-{idx:04d}" + + # 项目-类别关系 + category = row.get('类别', '') + if pd.notna(category) and str(category).strip(): + relation_id += 1 + relations.append({ + "id": f"REL-{relation_id:04d}", + "type": "belongs_to", + "from": project_id, + "to": f"CAT-{str(category)}", + "properties": {} + }) + + # 项目-传承人关系 + inheritor_names = row.get('代表性传承人', '') + if pd.notna(inheritor_names) and str(inheritor_names).strip(): + names = str(inheritor_names).replace('、', ',').replace(',', ',').split(',') + for name in names: + name = name.strip() + if name: + relation_id += 1 + relations.append({ + "id": f"REL-{relation_id:04d}", + "type": "has_inheritor", + "from": project_id, + "to": f"INH-{name}", # 简化处理 + "properties": {} + }) + + self.logger.info(f"提取了 {len(relations)} 个关系") + return relations + + def save_analysis_report(self, output_path: str): + """ + 保存分析报告 + + Args: + output_path: 输出路径 + """ + analysis = self.analyze_data() + + with open(output_path, 'w', encoding='utf-8') as f: + json.dump(analysis, f, indent=2, ensure_ascii=False) + + self.logger.info(f"分析报告已保存到: {output_path}") + + def save_kg_data(self, nodes: List[Dict], relations: List[Dict], output_dir: str): + """ + 保存知识图谱数据 + + Args: + nodes: 节点列表 + relations: 关系列表 + output_dir: 输出目录 + """ + output_path = Path(output_dir) + output_path.mkdir(parents=True, exist_ok=True) + + # 保存节点 + nodes_file = output_path / "nodes.json" + with open(nodes_file, 'w', encoding='utf-8') as f: + json.dump(nodes, f, indent=2, ensure_ascii=False) + self.logger.info(f"节点已保存到: {nodes_file}") + + # 保存关系 + relations_file = output_path / "relations.json" + with open(relations_file, 'w', encoding='utf-8') as f: + json.dump(relations, f, indent=2, ensure_ascii=False) + self.logger.info(f"关系已保存到: {relations_file}") + + +def main(): + """主函数""" + # 配置 + excel_path = r"E:\Project\2026_KG_ICH\data\黑龙江国家级和省级非遗名单.xlsx" + output_dir = r"E:\Project\2026_KG_ICH\dofile\kg_project\output" + + # 创建读取器 + reader = ExcelDataReader(excel_path) + + # 读取数据 + reader.read_excel() + + # 分析数据 + print("\n" + "="*60) + print("数据分析") + print("="*60) + analysis = reader.analyze_data() + print(f"\n总记录数: {analysis['total_records']}") + print(f"\n类别分布:") + for cat, count in analysis.get('category_distribution', {}).items(): + print(f" {cat}: {count}") + + print(f"\n传承人覆盖率: {analysis.get('inheritor_coverage', {}).get('coverage_rate', 0):.1f}%") + + # 保存分析报告 + Path(output_dir).mkdir(parents=True, exist_ok=True) + reader.save_analysis_report(f"{output_dir}/data_analysis.json") + + # 清洗数据 + print("\n" + "="*60) + print("数据清洗") + print("="*60) + reader.clean_data() + + # 转换为知识图谱格式 + print("\n" + "="*60) + print("转换为知识图谱格式") + print("="*60) + + nodes = reader.convert_to_kg_format() + inheritors = reader.extract_inheritors() + relations = reader.extract_relations() + + # 合并所有节点 + all_nodes = nodes + inheritors + + print(f"\n节点统计:") + print(f" 非遗项目节点: {len(nodes)}") + print(f" 传承人节点: {len(inheritors)}") + print(f" 总节点数: {len(all_nodes)}") + print(f" 关系数: {len(relations)}") + + # 保存知识图谱数据 + reader.save_kg_data(all_nodes, relations, output_dir) + + print("\n" + "="*60) + print("处理完成!") + print("="*60) + + +if __name__ == '__main__': + main() diff --git a/dofile/kg_project/scripts/extract_kg_csv.py b/dofile/kg_project/scripts/extract_kg_csv.py new file mode 100644 index 0000000..b8efce2 --- /dev/null +++ b/dofile/kg_project/scripts/extract_kg_csv.py @@ -0,0 +1,438 @@ +# -*- coding: utf-8 -*- +""" +从Excel提取数据生成知识图谱CSV文件 +保持原始数据表述不变 +""" + +import pandas as pd +import re +import json +from pathlib import Path + + +def extract_batches(df): + """提取所有唯一的批次(保持原始表述)""" + batches = {} + batch_counter = {} + + for batch_str in df['项目批次'].dropna().unique(): + if not batch_str or str(batch_str).strip() == '': + continue + + # 解析批次字符串(可能包含多个批次) + batch_items = re.split(r'[,、,]', str(batch_str)) + + for item in batch_items: + item = item.strip() + if not item: + continue + + # 生成批次ID(使用原始表述的哈希) + if item not in batch_counter: + batch_counter[item] = 1 + else: + batch_counter[item] += 1 + + batch_id = f"BATCH-{abs(hash(item)) % 100000:05d}" + + if batch_id not in batches: + batches[batch_id] = { + 'name': item, # 保持原始表述 + 'original_string': item + } + + return batches + + +def parse_inheritor_field(inheritor_str): + """解析传承人字段,保持原始表述(如"吴明新(国)")""" + if not inheritor_str or str(inheritor_str).strip() in ['无', '']: + return [] + + # 按顿号、逗号分割,保持原始表述 + inheritors = re.split(r'[、,,\n]', str(inheritor_str)) + inheritors = [inh.strip() for inh in inheritors if inh.strip() and inh.strip() != '无'] + return inheritors + + +def extract_inheritors(df): + """提取所有唯一传承人(保持原始表述)""" + inheritors = {} + inheritor_counter = {} + + for idx, row in df.iterrows(): + inheritor_str = row.get('代表性传承人', '') + if not inheritor_str or str(inheritor_str).strip() in ['无', '']: + continue + + # 解析传承人列表 + inheritor_names = parse_inheritor_field(inheritor_str) + + for name in inheritor_names: + # 使用原始名称(包括括号)作为key + if name not in inheritor_counter: + inheritor_counter[name] = 1 + else: + inheritor_counter[name] += 1 + + # 生成传承人ID + inheritor_id = f"INH-{abs(hash(name)) % 100000:05d}-{inheritor_counter[name]}" + + if inheritor_id not in inheritors: + inheritors[inheritor_id] = { + 'name': name # 保持原始表述,如"吴明新(国)" + } + + return inheritors + + +def extract_institutions(df): + """提取所有唯一保护机构(保持原始表述)""" + institutions = {} + inst_counter = {} + + for idx, row in df.iterrows(): + inst_str = row.get('项目保护单位', '') + if not inst_str or str(inst_str).strip() in ['', '无']: + continue + + inst_name = str(inst_str).strip() + + # 使用原始机构名 + if inst_name not in inst_counter: + inst_counter[inst_name] = 1 + else: + inst_counter[inst_name] += 1 + + # 生成机构ID + inst_id = f"INST-{abs(hash(inst_name)) % 100000:05d}-{inst_counter[inst_name]}" + + if inst_id not in institutions: + institutions[inst_id] = { + 'name': inst_name # 保持原始表述 + } + + return institutions + + +def parse_batch_field(batch_str, batches_dict): + """解析批次字段,返回批次ID列表""" + if not batch_str or str(batch_str).strip() == '': + return [] + + batches = re.split(r'[,、,]', str(batch_str)) + batch_ids = [] + + for batch in batches: + batch = batch.strip() + if batch: + # 查找对应的批次ID + for batch_id, batch_info in batches_dict.items(): + if batch_info['name'] == batch: + batch_ids.append(batch_id) + break + + return batch_ids + + +def get_institution_id(institution_name, institutions_dict): + """获取机构ID""" + if not institution_name or str(institution_name).strip() in ['', '无']: + return None + + # 从已提取的机构中查找 + for inst_id, inst_info in institutions_dict.items(): + if inst_info['name'] == str(institution_name).strip(): + return inst_id + + return None + + +def get_inheritor_id(inheritor_name, inheritors_dict): + """获取传承人ID""" + if not inheritor_name or not inheritor_name.strip(): + return None + + # 从已提取的传承人中查找 + for inheritor_id, inheritor_info in inheritors_dict.items(): + if inheritor_info['name'] == inheritor_name: + return inheritor_id + + return None + + +def clean_data(df): + """数据清洗""" + # 去除完全空白的行 + df = df.dropna(how='all') + + # 填充缺失值为空字符串 + for col in df.columns: + if df[col].dtype == 'object': + df[col] = df[col].fillna('') + + return df + + +def build_nodes(df): + """构建节点数据(保持原始表述)""" + nodes_list = [] + + print("正在构建节点...") + + # 1. ICH_Project节点 + print(" - 构建ICH_Project节点...") + for idx, row in df.iterrows(): + seq_num = row.get('总序号', idx + 1) + project_name = row.get('项目名称', '') + + # 确定级别(从批次字段推断) + batch_str = str(row.get('项目批次', '')) + level = '国家级' if '国家级' in batch_str else ('省级' if '省级' in batch_str else '') + + nodes_list.append({ + 'id': f"ICH-{int(seq_num)}", + 'label': str(project_name), + 'type': 'ICH_Project', + 'properties': json.dumps({ + 'level': level, + 'category': row.get('类别', ''), # 保持原始表述,如"曲艺类" + 'batch': batch_str, + 'protection_unit': row.get('项目保护单位', '') + }, ensure_ascii=False) + }) + + # 2. Category节点(从数据中动态提取,保持原始表述) + print(" - 构建Category节点...") + unique_categories = df['类别'].dropna().unique() + + for cat_name in unique_categories: + # 使用类别名称作为ID(使用哈希避免特殊字符) + cat_id = f"CAT-{abs(hash(cat_name)) % 100000:05d}" + + nodes_list.append({ + 'id': cat_id, + 'label': cat_name, # 保持原始表述,如"曲艺类" + 'type': 'Category', + 'properties': '{}' + }) + + # 3. Batch节点(动态识别,保持原始表述) + print(" - 构建Batch节点...") + batches = extract_batches(df) + for batch_id, batch_info in batches.items(): + nodes_list.append({ + 'id': batch_id, + 'label': batch_info['name'], # 保持原始表述,如"国家级第1批" + 'type': 'Batch', + 'properties': json.dumps({ + 'original_string': batch_info['original_string'] + }, ensure_ascii=False) + }) + + # 4. Inheritor节点(动态识别,保持原始表述) + print(" - 构建Inheritor节点...") + inheritors = extract_inheritors(df) + for inheritor_id, inheritor_info in inheritors.items(): + nodes_list.append({ + 'id': inheritor_id, + 'label': inheritor_info['name'], # 保持原始表述,如"吴明新(国)" + 'type': 'Inheritor', + 'properties': '{}' + }) + + # 5. Institution节点(动态识别,保持原始表述) + print(" - 构建Institution节点...") + institutions = extract_institutions(df) + for inst_id, inst_info in institutions.items(): + nodes_list.append({ + 'id': inst_id, + 'label': inst_info['name'], # 保持原始表述 + 'type': 'Institution', + 'properties': '{}' + }) + + return nodes_list, batches, inheritors, institutions + + +def build_relations(df, batches, inheritors, institutions): + """构建关系数据""" + relations_list = [] + + print("正在构建关系...") + + # 建立名称到ID的快速查找映射 + category_map = {} + for cat_name in df['类别'].dropna().unique(): + cat_id = f"CAT-{abs(hash(cat_name)) % 100000:05d}" + category_map[cat_name] = cat_id + + for idx, row in df.iterrows(): + seq_num = row.get('总序号', idx + 1) + project_id = f"ICH-{int(seq_num)}" + + # 1. 项目 → 类别 + category_name = row.get('类别', '') + if category_name and category_name in category_map: + category_id = category_map[category_name] + relations_list.append({ + 'source': project_id, + 'target': category_id, + 'type': 'BELONGS_TO', + 'properties': '{}' + }) + + # 2. 项目 → 批次(支持多个批次) + batch_str = row.get('项目批次', '') + if batch_str: + batch_ids = parse_batch_field(batch_str, batches) + for batch_id in batch_ids: + relations_list.append({ + 'source': project_id, + 'target': batch_id, + 'type': 'SELECTED_IN_BATCH', + 'properties': '{}' + }) + + # 3. 项目 → 保护机构 + institution_name = row.get('项目保护单位', '') + if institution_name and str(institution_name).strip() not in ['', '无']: + institution_id = get_institution_id(institution_name, institutions) + if institution_id: + relations_list.append({ + 'source': project_id, + 'target': institution_id, + 'type': 'PROTECTED_BY', + 'properties': '{}' + }) + + # 4. 项目 → 传承人(支持多个传承人) + inheritor_str = row.get('代表性传承人', '') + if inheritor_str: + inheritor_names = parse_inheritor_field(inheritor_str) + for name in inheritor_names: + inheritor_id = get_inheritor_id(name, inheritors) + if inheritor_id: + relations_list.append({ + 'source': project_id, + 'target': inheritor_id, + 'type': 'HAS_INHERITOR', + 'properties': '{}' + }) + + return relations_list + + +def generate_report(nodes_df, relations_df, output_dir): + """生成统计报告""" + report_lines = [] + report_lines.append("# 知识图谱CSV提取报告\n") + report_lines.append(f"生成时间: {pd.Timestamp.now().strftime('%Y-%m-%d %H:%M:%S')}\n") + report_lines.append("---\n\n") + + # 节点统计 + report_lines.append("## 节点统计\n\n") + report_lines.append(f"**节点总数**: {len(nodes_df)}\n\n") + + node_type_counts = nodes_df['type'].value_counts().sort_index() + report_lines.append("| 节点类型 | 数量 | 占比 |\n") + report_lines.append("|---------|------|------|\n") + for node_type, count in node_type_counts.items(): + percentage = (count / len(nodes_df) * 100) + report_lines.append(f"| {node_type} | {count} | {percentage:.1f}% |\n") + + # 关系统计 + report_lines.append("\n## 关系统计\n\n") + report_lines.append(f"**关系总数**: {len(relations_df)}\n\n") + + rel_type_counts = relations_df['type'].value_counts().sort_index() + report_lines.append("| 关系类型 | 数量 | 占比 |\n") + report_lines.append("|---------|------|------|\n") + for rel_type, count in rel_type_counts.items(): + percentage = (count / len(relations_df) * 100) + report_lines.append(f"| {rel_type} | {count} | {percentage:.1f}% |\n") + + # 保存报告 + report_path = output_dir / 'extraction_report.md' + with open(report_path, 'w', encoding='utf-8') as f: + f.writelines(report_lines) + + print(f"\n报告已生成: {report_path}") + + # 打印统计信息 + print("\n" + "="*50) + print("数据提取完成") + print("="*50) + print(f"\n节点总数: {len(nodes_df)}") + for node_type, count in node_type_counts.items(): + print(f" - {node_type}: {count}") + print(f"\n关系总数: {len(relations_df)}") + for rel_type, count in rel_type_counts.items(): + print(f" - {rel_type}: {count}") + print("="*50) + + +def extract_data_from_excel(): + """从Excel提取数据并生成知识图谱CSV文件""" + + print("开始从Excel提取数据...") + + # 1. 读取Excel + excel_path = r'E:\Project\2026_KG_ICH\data\黑龙江国家级和省级非遗名单.xlsx' + print(f"读取文件: {excel_path}") + + df = pd.read_excel(excel_path) + print(f"原始数据: {len(df)} 行 x {len(df.columns)} 列") + + # 2. 提取所需列 + columns = ['总序号', '类别', '项目名称', '项目批次', '项目保护单位', '代表性传承人'] + print(f"\n提取列: {', '.join(columns)}") + + # 检查列是否存在 + available_cols = [col for col in columns if col in df.columns] + if len(available_cols) < len(columns): + missing = set(columns) - set(available_cols) + print(f"警告: 以下列不存在: {missing}") + + df = df[available_cols] + print(f"提取后数据: {len(df)} 行 x {len(df.columns)} 列") + + # 3. 数据清洗 + print("\n数据清洗...") + df = clean_data(df) + print(f"清洗后数据: {len(df)} 行") + + # 4. 构建节点 + nodes_list, batches, inheritors, institutions = build_nodes(df) + nodes_df = pd.DataFrame(nodes_list) + print(f"节点数据: {len(nodes_df)} 行") + + # 5. 构建关系 + relations_list = build_relations(df, batches, inheritors, institutions) + relations_df = pd.DataFrame(relations_list) + print(f"关系数据: {len(relations_df)} 行") + + # 6. 保存CSV + output_dir = Path(r'E:\Project\2026_KG_ICH\dofile\kg_project\output') + output_dir.mkdir(parents=True, exist_ok=True) + + print(f"\n保存文件到: {output_dir}") + + # 保存为UTF-8-BOM编码(Excel友好) + nodes_path = output_dir / 'nodes.csv' + relations_path = output_dir / 'rels.csv' + + nodes_df.to_csv(nodes_path, index=False, encoding='utf-8-sig') + relations_df.to_csv(relations_path, index=False, encoding='utf-8-sig') + + print(f" - nodes.csv: {nodes_path}") + print(f" - rels.csv: {relations_path}") + + # 7. 生成统计报告 + generate_report(nodes_df, relations_df, output_dir) + + return nodes_df, relations_df + + +if __name__ == '__main__': + extract_data_from_excel() diff --git a/dofile/kg_project/scripts/visualize_kg.py b/dofile/kg_project/scripts/visualize_kg.py new file mode 100644 index 0000000..c5c362f --- /dev/null +++ b/dofile/kg_project/scripts/visualize_kg.py @@ -0,0 +1,382 @@ +# -*- coding: utf-8 -*- +""" +知识图谱可视化工具 +使用networkx和matplotlib绘制知识图谱 +""" + +import pandas as pd +import networkx as nx +import matplotlib.pyplot as plt +from matplotlib import font_manager +import matplotlib.patches as mpatches +from pathlib import Path +import numpy as np + + +# 设置中文字体 +def setup_chinese_font(): + """设置中文字体""" + # 尝试多种中文字体 + chinese_fonts = [ + 'Microsoft YaHei', + 'SimHei', + 'SimSun', + 'KaiTi', + 'FangSong', + 'STXihei', + 'STSong', + 'STKaiti', + 'STFangsong' + ] + + for font in chinese_fonts: + try: + plt.rcParams['font.sans-serif'] = [font] + plt.rcParams['axes.unicode_minus'] = False + break + except: + continue + + print(f"使用字体: {plt.rcParams['font.sans-serif'][0]}") + + +# 节点类型颜色映射 +NODE_TYPE_COLORS = { + 'ICH_Project': '#FF6B6B', # 红色 - 项目 + 'Category': '#4ECDC4', # 青色 - 类别 + 'Batch': '#95E1D3', # 绿色 - 批次 + 'Inheritor': '#FFD93D', # 黄色 - 传承人 + 'Institution': '#6C5CE7' # 紫色 - 机构 +} + +# 节点类型大小映射 +NODE_TYPE_SIZES = { + 'ICH_Project': 300, + 'Category': 500, + 'Batch': 350, + 'Inheritor': 200, + 'Institution': 250 +} + + +def load_graph_data(nodes_csv, rels_csv): + """加载图谱数据""" + print("正在加载数据...") + + # 读取节点和关系 + nodes_df = pd.read_csv(nodes_csv, encoding='utf-8-sig') + rels_df = pd.read_csv(rels_csv, encoding='utf-8-sig') + + print(f" - 节点: {len(nodes_df)}") + print(f" - 关系: {len(rels_df)}") + + return nodes_df, rels_df + + +def build_networkx_graph(nodes_df, rels_df): + """构建NetworkX图""" + print("正在构建图...") + + G = nx.DiGraph() # 有向图 + + # 添加节点 + for idx, row in nodes_df.iterrows(): + node_id = row['id'] + label = row['label'] + node_type = row['type'] + + # 截断过长的标签 + if len(label) > 10: + display_label = label[:10] + '...' + else: + display_label = label + + G.add_node( + node_id, + label=display_label, + full_label=label, + node_type=node_type, + color=NODE_TYPE_COLORS.get(node_type, '#CCCCCC'), + size=NODE_TYPE_SIZES.get(node_type, 200) + ) + + # 添加边 + for idx, row in rels_df.iterrows(): + source = row['source'] + target = row['target'] + rel_type = row['type'] + + if source in G.nodes() and target in G.nodes(): + G.add_edge(source, target, rel_type=rel_type) + + print(f" - 节点数: {G.number_of_nodes()}") + print(f" - 边数: {G.number_of_edges()}") + + return G + + +def filter_graph_by_type(G, include_types=None): + """按节点类型过滤图""" + if include_types is None: + return G + + nodes_to_keep = [n for n, d in G.nodes(data=True) + if d.get('node_type') in include_types] + + return G.subgraph(nodes_to_keep).copy() + + +def draw_graph(G, output_path, title="知识图谱", layout='spring'): + """绘制知识图谱""" + print(f"正在绘制图谱: {title}") + + plt.figure(figsize=(20, 16)) + + # 选择布局算法 + if layout == 'spring': + pos = nx.spring_layout(G, k=2, iterations=50, seed=42) + elif layout == 'circular': + pos = nx.circular_layout(G) + elif layout == 'kamada_kawai': + pos = nx.kamada_kawai_layout(G) + elif layout == 'random': + pos = nx.random_layout(G) + else: + pos = nx.spring_layout(G, k=2, iterations=50, seed=42) + + # 按节点类型分组 + node_types = {} + for node, data in G.nodes(data=True): + node_type = data.get('node_type', 'Unknown') + if node_type not in node_types: + node_types[node_type] = [] + node_types[node_type].append(node) + + # 绘制边 + nx.draw_networkx_edges( + G, pos, + alpha=0.3, + width=0.5, + edge_color='gray', + arrows=True, + arrowsize=10, + arrowstyle='->,head_width=0.2,head_length=0.3' + ) + + # 按类型绘制节点 + for node_type, nodes in node_types.items(): + color = NODE_TYPE_COLORS.get(node_type, '#CCCCCC') + size = NODE_TYPE_SIZES.get(node_type, 200) + + nx.draw_networkx_nodes( + G, pos, + nodelist=nodes, + node_color=color, + node_size=size, + alpha=0.8, + edgecolors='white', + linewidths=1 + ) + + # 绘制标签(只对重要节点) + important_nodes = [] + important_labels = {} + + for node, data in G.nodes(data=True): + node_type = data.get('node_type') + # 只显示类别、批次和部分重要节点的标签 + if node_type in ['Category', 'Batch'] or ( + node_type == 'ICH_Project' and data.get('size', 0) > 400 + ): + important_nodes.append(node) + important_labels[node] = data.get('label', node) + + if len(important_nodes) <= 100: # 节点不多时显示所有标签 + nx.draw_networkx_labels( + G, pos, + labels=important_labels, + font_size=8, + font_weight='bold', + font_family='sans-serif' + ) + else: + # 节点太多时只显示类别标签 + category_labels = {n: d['label'] for n, d in G.nodes(data=True) + if d.get('node_type') == 'Category'} + nx.draw_networkx_labels( + G, pos, + labels=category_labels, + font_size=10, + font_weight='bold' + ) + + # 图例 + legend_patches = [] + for node_type, color in NODE_TYPE_COLORS.items(): + if node_type in node_types: + patch = mpatches.Patch(color=color, label=node_type) + legend_patches.append(patch) + + plt.legend( + handles=legend_patches, + loc='upper right', + fontsize=12, + framealpha=0.9 + ) + + plt.title(title, fontsize=16, fontweight='bold', pad=20) + plt.axis('off') + plt.tight_layout() + + # 保存图片 + plt.savefig(output_path, dpi=150, bbox_inches='tight') + print(f" - 保存到: {output_path}") + plt.close() + + +def draw_subgraphs(G, output_dir): + """绘制子图(按节点类型分组)""" + print("\n正在绘制子图...") + + output_dir = Path(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + + # 1. 只显示项目和类别 + print(" 1. 项目-类别关系图...") + G1 = filter_graph_by_type(G, ['ICH_Project', 'Category']) + draw_graph( + G1, + output_dir / 'kg_project_category.png', + title='非遗项目与类别关系', + layout='spring' + ) + + # 2. 只显示项目和传承人 + print(" 2. 项目-传承人关系图...") + G2 = filter_graph_by_type(G, ['ICH_Project', 'Inheritor']) + if G2.number_of_nodes() > 0: + draw_graph( + G2, + output_dir / 'kg_project_inheritor.png', + title='非遗项目与传承人关系', + layout='spring' + ) + + # 3. 只显示项目、类别和批次 + print(" 3. 项目-类别-批次关系图...") + G3 = filter_graph_by_type(G, ['ICH_Project', 'Category', 'Batch']) + draw_graph( + G3, + output_dir / 'kg_project_category_batch.png', + title='非遗项目、类别与批次关系', + layout='kamada_kawai' + ) + + # 4. 完整图谱(抽样显示) + print(" 4. 完整知识图谱...") + if G.number_of_nodes() > 500: + # 节点太多时,只显示连接度高的节点 + degrees = dict(G.degree()) + high_degree_nodes = [n for n, d in degrees.items() if d >= 3] + G_sample = G.subgraph(high_degree_nodes).copy() + draw_graph( + G_sample, + output_dir / 'kg_full_sampled.png', + title=f'完整知识图谱(抽样,显示{G_sample.number_of_nodes()}个节点)', + layout='spring' + ) + else: + draw_graph( + G, + output_dir / 'kg_full.png', + title='完整知识图谱', + layout='spring' + ) + + +def print_statistics(G): + """打印图统计信息""" + print("\n" + "="*60) + print("图谱统计信息") + print("="*60) + + print(f"\n节点总数: {G.number_of_nodes()}") + print(f"边总数: {G.number_of_edges()}") + + # 按类型统计节点 + print("\n节点类型分布:") + node_types = {} + for node, data in G.nodes(data=True): + node_type = data.get('node_type', 'Unknown') + node_types[node_type] = node_types.get(node_type, 0) + 1 + + for node_type, count in sorted(node_types.items()): + percentage = (count / G.number_of_nodes() * 100) + print(f" - {node_type}: {count} ({percentage:.1f}%)") + + # 按类型统计边 + print("\n关系类型分布:") + rel_types = {} + for u, v, data in G.edges(data=True): + rel_type = data.get('rel_type', 'Unknown') + rel_types[rel_type] = rel_types.get(rel_type, 0) + 1 + + for rel_type, count in sorted(rel_types.items()): + percentage = (count / G.number_of_edges() * 100) + print(f" - {rel_type}: {count} ({percentage:.1f}%)") + + # 连接度统计 + degrees = [d for n, d in G.degree()] + print(f"\n连接度统计:") + print(f" - 平均连接度: {np.mean(degrees):.2f}") + print(f" - 最大连接度: {max(degrees)}") + print(f" - 最小连接度: {min(degrees)}") + + # 找出连接度最高的节点 + top_nodes = sorted(G.degree(), key=lambda x: x[1], reverse=True)[:10] + print(f"\n连接度最高的10个节点:") + for node, degree in top_nodes: + node_data = G.nodes[node] + label = node_data.get('full_label', node) + node_type = node_data.get('node_type', '') + print(f" - [{node_type}] {label}: {degree}个连接") + + print("="*60) + + +def visualize_kg(nodes_csv, rels_csv, output_dir): + """可视化知识图谱""" + print("="*60) + print("知识图谱可视化工具") + print("="*60) + + # 设置中文字体 + setup_chinese_font() + + # 加载数据 + nodes_df, rels_df = load_graph_data(nodes_csv, rels_csv) + + # 构建图 + G = build_networkx_graph(nodes_df, rels_df) + + # 打印统计信息 + print_statistics(G) + + # 绘制子图 + draw_subgraphs(G, output_dir) + + print("\n" + "="*60) + print("可视化完成!") + print("="*60) + + +if __name__ == '__main__': + # 输入文件 + nodes_csv = r'E:\Project\2026_KG_ICH\dofile\kg_project\output\nodes.csv' + rels_csv = r'E:\Project\2026_KG_ICH\dofile\kg_project\output\rels.csv' + + # 输出目录 + output_dir = r'E:\Project\2026_KG_ICH\dofile\kg_project\output\visualizations' + + # 执行可视化 + visualize_kg(nodes_csv, rels_csv, output_dir) diff --git a/dofile/kg_project/src/__init__.py b/dofile/kg_project/src/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/dofile/kg_project/src/data_processing/__init__.py b/dofile/kg_project/src/data_processing/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/dofile/kg_project/src/data_processing/entity_normalizer.py b/dofile/kg_project/src/data_processing/entity_normalizer.py new file mode 100644 index 0000000..9353c42 --- /dev/null +++ b/dofile/kg_project/src/data_processing/entity_normalizer.py @@ -0,0 +1,344 @@ +# -*- coding: utf-8 -*- +""" +实体规范化器 +负责实体的去重、ID生成和规范化 +""" + +import json +import hashlib +import logging +from typing import Dict, List, Any, Optional +from pathlib import Path +import yaml + + +class EntityNormalizer: + """实体规范化器""" + + def __init__(self, ontology_file: str): + """ + 初始化规范化器 + + Args: + ontology_file: 本体配置文件路径 + """ + # 加载本体配置 + with open(ontology_file, 'r', encoding='utf-8') as f: + self.ontology = yaml.safe_load(f) + + self.logger = logging.getLogger(__name__) + + # 实体注册表:{entity_id: entity_data} + self.entity_registry = {} + + # 文本到ID的映射:{normalized_text: entity_id} + self.text_to_id_map = {} + + # 类型计数器:{entity_type: count} + self.type_counters = {} + + self.logger.info("EntityNormalizer初始化完成") + + def normalize_text(self, text: str) -> str: + """ + 标准化文本 + + Args: + text: 原始文本 + + Returns: + 标准化后的文本 + """ + if not text: + return "" + + # 去除首尾空格 + text = text.strip() + + # 统一全角/半角字符 + text = text.replace(' ', ' ').replace(',', ',').replace('、', ',') + + # 转换为小写进行比较(保持原文用于显示) + return text + + def calculate_similarity(self, text1: str, text2: str) -> float: + """ + 计算两个文本的相似度(基于编辑距离) + + Args: + text1: 文本1 + text2: 文本2 + + Returns: + 相似度(0-1之间) + """ + import Levenshtein + + norm1 = self.normalize_text(text1) + norm2 = self.normalize_text(text2) + + if not norm1 or not norm2: + return 0.0 + + max_len = max(len(norm1), len(norm2)) + if max_len == 0: + return 1.0 + + distance = Levenshtein.distance(norm1, norm2) + similarity = 1.0 - (distance / max_len) + + return similarity + + def generate_entity_id(self, entity_type: str, entity_text: str) -> str: + """ + 生成实体ID + + Args: + entity_type: 实体类型 + entity_text: 实体文本 + + Returns: + 实体ID + """ + # 获取类型配置 + type_config = self.ontology['entity_types'].get(entity_type, {}) + prefix = type_config.get('prefix', 'UNK') + + # 生成哈希值 + text_hash = abs(hash(entity_text)) % 100000 + + # 检查是否需要计数器(某些类型可能需要) + if entity_type not in self.type_counters: + self.type_counters[entity_type] = 0 + + # 根据类型生成ID + if entity_type in ['Ethnic_Group', 'Geographic_Location', 'Geographic_Environment', 'Time_Period']: + # 这些类型使用唯一哈希,不需要计数器 + entity_id = f"{prefix}-{text_hash:05d}" + else: + # 其他类型使用计数器 + self.type_counters[entity_type] += 1 + counter = self.type_counters[entity_type] + entity_id = f"{prefix}-{text_hash:05d}-{counter}" + + return entity_id + + def find_similar_entity( + self, + entity_text: str, + entity_type: str, + similarity_threshold: float = 0.85 + ) -> Optional[str]: + """ + 查找相似实体 + + Args: + entity_text: 实体文本 + entity_type: 实体类型 + similarity_threshold: 相似度阈值 + + Returns: + 相似实体的ID,如果不存在返回None + """ + normalized_text = self.normalize_text(entity_text) + + # 只在相同类型的实体中查找 + for entity_id, entity_data in self.entity_registry.items(): + if entity_data['type'] != entity_type: + continue + + # 检查文本相似度 + similarity = self.calculate_similarity(normalized_text, entity_data['canonical_name']) + + if similarity >= similarity_threshold: + self.logger.info(f"发现相似实体: '{entity_text}' ~ '{entity_data['canonical_name']}' (相似度: {similarity:.2f})") + return entity_id + + return None + + def is_generic_concept(self, entity_text: str, entity_type: str) -> bool: + """ + 检查是否为泛指概念(应该被过滤) + + Args: + entity_text: 实体文本 + entity_type: 实体类型 + + Returns: + True表示是泛指概念,应该过滤 + """ + generic_patterns = [ + '东北少数民族', + '本地土著民族', + '当地民族', + '少数民族', + '土著', + '东北民族', + '本地民族', + '地区民族', + ] + + normalized_text = self.normalize_text(entity_text) + + # 检查是否匹配泛指模式 + for pattern in generic_patterns: + if pattern in normalized_text: + self.logger.info(f"过滤泛指概念: '{entity_text}' (类型: {entity_type})") + return True + + # 对于Ethnic_Group类型,必须是具体的民族名称 + if entity_type == 'Ethnic_Group': + specific_ethnic_groups = [ + '满族', '赫哲族', '鄂伦春族', '鄂温克族', + '达斡尔族', '朝鲜族', '蒙古族', '回族', + '汉族', '锡伯族', '柯尔克孜族' + ] + # 如果不在具体民族列表中,可能是泛指概念 + is_specific = any(group in normalized_text for group in specific_ethnic_groups) + if not is_specific: + self.logger.info(f"过滤非具体民族: '{entity_text}'") + return True + + return False + + def normalize_entity( + self, + raw_entity: Dict[str, Any], + similarity_threshold: float = 0.85, + filter_generic: bool = True + ) -> str: + """ + 规范化单个实体 + + Args: + raw_entity: 原始实体数据(从LLM返回) + similarity_threshold: 相似度阈值 + filter_generic: 是否过滤泛指概念 + + Returns: + 实体ID + """ + entity_text = raw_entity.get('text', '') + entity_type = raw_entity.get('type', '') + attributes = raw_entity.get('attributes', {}) + + if not entity_text or not entity_type: + self.logger.error(f"实体缺少text或type: {raw_entity}") + return None + + # 验证实体类型是否在本体中定义 + if entity_type not in self.ontology['entity_types']: + self.logger.warning(f"未知实体类型 '{entity_type}',跳过: '{entity_text}'") + return None + + # 过滤泛指概念 + if filter_generic and self.is_generic_concept(entity_text, entity_type): + return None + + # 标准化文本 + normalized_text = self.normalize_text(entity_text) + + # 检查是否已存在完全相同的实体 + if normalized_text in self.text_to_id_map: + existing_id = self.text_to_id_map[normalized_text] + self.logger.info(f"实体已存在: '{entity_text}' -> {existing_id}") + return existing_id + + # 查找相似实体 + similar_id = self.find_similar_entity(entity_text, entity_type, similarity_threshold) + if similar_id: + # 合并到相似实体 + self.logger.info(f"合并实体: '{entity_text}' -> {similar_id}") + self.text_to_id_map[normalized_text] = similar_id + return similar_id + + # 生成新实体ID + entity_id = self.generate_entity_id(entity_type, entity_text) + + # 获取标准名称 + standard_name = attributes.get('name') or entity_text + + # 保存到注册表 + self.entity_registry[entity_id] = { + 'canonical_name': normalized_text, + 'display_name': standard_name, + 'type': entity_type, + 'attributes': attributes, + 'source_projects': [] + } + + # 保存文本映射 + self.text_to_id_map[normalized_text] = entity_id + + self.logger.info(f"新建实体: {entity_id} - '{standard_name}' ({entity_type})") + + return entity_id + + def normalize_batch( + self, + extraction_results: List[Dict[str, Any]], + similarity_threshold: float = 0.85 + ) -> Dict[str, str]: + """ + 批量规范化实体 + + Args: + extraction_results: LLM抽取结果列表 + similarity_threshold: 相似度阈值 + + Returns: + 文本到ID的映射字典(包含所有已处理的实体) + """ + self.logger.info(f"开始批量规范化,共{len(extraction_results)}个抽取结果") + + for idx, result in enumerate(extraction_results): + if not result or 'entities' not in result: + continue + + for entity in result['entities']: + self.normalize_entity(entity, similarity_threshold) + + self.logger.info(f"批量规范化完成,生成{len(self.entity_registry)}个唯一实体") + + # 返回完整的文本到ID映射(包含所有已处理的实体,不仅仅是本批次) + return self.text_to_id_map.copy() + + def get_entity_nodes(self) -> List[Dict[str, Any]]: + """ + 获取所有实体节点(用于生成CSV) + + Returns: + 节点列表 + """ + nodes = [] + + for entity_id, entity_data in self.entity_registry.items(): + node = { + 'id': entity_id, + 'label': entity_data['display_name'], + 'type': entity_data['type'], + 'properties': json.dumps(entity_data['attributes'], ensure_ascii=False) + } + nodes.append(node) + + return nodes + + def get_statistics(self) -> Dict[str, Any]: + """ + 获取统计信息 + + Returns: + 统计信息字典 + """ + stats = { + 'total_entities': len(self.entity_registry), + 'entities_by_type': {}, + 'type_counters': self.type_counters.copy() + } + + # 按类型统计 + for entity_data in self.entity_registry.values(): + entity_type = entity_data['type'] + stats['entities_by_type'][entity_type] = stats['entities_by_type'].get(entity_type, 0) + 1 + + return stats diff --git a/dofile/kg_project/src/data_processing/relationship_builder.py b/dofile/kg_project/src/data_processing/relationship_builder.py new file mode 100644 index 0000000..74de4b3 --- /dev/null +++ b/dofile/kg_project/src/data_processing/relationship_builder.py @@ -0,0 +1,280 @@ +# -*- coding: utf-8 -*- +""" +关系构建器 +负责构建非遗项目与深层实体之间的关系 +""" + +import json +import logging +from typing import Dict, List, Any, Optional +from pathlib import Path +import yaml + + +class RelationshipBuilder: + """关系构建器""" + + def __init__(self, ontology_file: str): + """ + 初始化关系构建器 + + Args: + ontology_file: 本体配置文件路径 + """ + # 加载本体配置 + with open(ontology_file, 'r', encoding='utf-8') as f: + self.ontology = yaml.safe_load(f) + + self.logger = logging.getLogger(__name__) + + # 关系注册表:用于去重 + self.relationship_registry = set() + + self.logger.info("RelationshipBuilder初始化完成") + + def build_relationship( + self, + source_id: str, + target_id: str, + rel_type: str, + properties: Dict[str, Any] = None + ) -> Optional[Dict[str, Any]]: + """ + 构建单个关系 + + Args: + source_id: 源实体ID + target_id: 目标实体ID + rel_type: 关系类型 + properties: 关系属性 + + Returns: + 关系字典,如果验证失败返回None + """ + # 验证关系类型 + if rel_type not in self.ontology['relationship_types']: + self.logger.warning(f"未知关系类型: {rel_type}") + return None + + # 验证ID不为空 + if not source_id or not target_id: + self.logger.error(f"关系ID不能为空: source={source_id}, target={target_id}") + return None + + # 生成关系唯一键(用于去重) + rel_key = f"{source_id}-{target_id}-{rel_type}" + + # 检查是否重复 + if rel_key in self.relationship_registry: + self.logger.debug(f"关系已存在,跳过: {rel_key}") + return None + + # 添加到注册表 + self.relationship_registry.add(rel_key) + + # 构建关系 + relationship = { + 'source': source_id, + 'target': target_id, + 'type': rel_type, + 'properties': json.dumps(properties or {}, ensure_ascii=False) + } + + return relationship + + def build_relationships_from_extraction( + self, + project_id: str, + extraction_result: Dict[str, Any], + entity_id_map: Dict[str, str] + ) -> List[Dict[str, Any]]: + """ + 从抽取结果构建关系 + + Args: + project_id: 项目ID + extraction_result: LLM抽取结果 + entity_id_map: 实体文本到ID的映射 + + Returns: + 关系列表 + """ + relationships = [] + + if not extraction_result or 'relationships' not in extraction_result: + return relationships + + for rel in extraction_result['relationships']: + # 获取源实体(应该是项目ID) + source_entity = rel.get('source_entity', '') + + # 验证源实体是否匹配当前项目 + if source_entity != project_id: + self.logger.warning(f"关系源实体不匹配: 期望{project_id}, 实际{source_entity}") + # 如果不匹配,尝试修正 + source_entity = project_id + + # 获取目标实体文本 + target_text = rel.get('target_entity', '') + + # 查找目标实体ID + target_id = entity_id_map.get(target_text) + + if not target_id: + self.logger.warning(f"未找到目标实体ID: {target_text}") + continue + + # 获取关系类型 + rel_type = rel.get('type', '') + rel_properties = rel.get('properties', {}) + + # 构建关系 + relationship = self.build_relationship( + source_entity, + target_id, + rel_type, + rel_properties + ) + + if relationship: + relationships.append(relationship) + + return relationships + + def build_batch_relationships( + self, + projects_data: List[Dict[str, Any]], + extraction_results: List[Dict[str, Any]], + entity_id_map: Dict[str, str] + ) -> List[Dict[str, Any]]: + """ + 批量构建关系 + + Args: + projects_data: 项目数据列表 + extraction_results: 抽取结果列表 + entity_id_map: 实体文本到ID的映射 + + Returns: + 关系列表 + """ + self.logger.info(f"开始批量构建关系,共{len(projects_data)}个项目") + + all_relationships = [] + + for idx, (project_data, extraction_result) in enumerate(zip(projects_data, extraction_results)): + if not extraction_result: + continue + + project_id = project_data.get('project_id', '') + + if not project_id: + self.logger.warning(f"项目{idx}缺少project_id") + continue + + # 构建该项目的关系 + relationships = self.build_relationships_from_extraction( + project_id, + extraction_result, + entity_id_map + ) + + all_relationships.extend(relationships) + + self.logger.info(f"项目 {project_id} 构建了{len(relationships)}个关系") + + self.logger.info(f"批量构建完成,共{len(all_relationships)}个关系") + + return all_relationships + + def validate_relationships( + self, + relationships: List[Dict[str, Any]], + node_ids: set + ) -> Dict[str, Any]: + """ + 验证关系完整性 + + Args: + relationships: 关系列表 + node_ids: 节点ID集合 + + Returns: + 验证报告 + """ + report = { + 'total_relationships': len(relationships), + 'valid_relationships': 0, + 'invalid_relationships': 0, + 'broken_links': [], + 'invalid_types': [], + 'relationships_by_type': {} + } + + for rel in relationships: + source_id = rel.get('source', '') + target_id = rel.get('target', '') + rel_type = rel.get('type', '') + + # 统计关系类型 + report['relationships_by_type'][rel_type] = \ + report['relationships_by_type'].get(rel_type, 0) + 1 + + is_valid = True + + # 验证关系类型 + if rel_type not in self.ontology['relationship_types']: + report['invalid_types'].append({ + 'source': source_id, + 'target': target_id, + 'type': rel_type + }) + is_valid = False + + # 验证链接完整性 + if source_id not in node_ids: + report['broken_links'].append({ + 'source': source_id, + 'target': target_id, + 'type': rel_type, + 'issue': 'source_not_found' + }) + is_valid = False + + if target_id not in node_ids: + report['broken_links'].append({ + 'source': source_id, + 'target': target_id, + 'type': rel_type, + 'issue': 'target_not_found' + }) + is_valid = False + + if is_valid: + report['valid_relationships'] += 1 + else: + report['invalid_relationships'] += 1 + + return report + + def get_statistics(self, relationships: List[Dict[str, Any]]) -> Dict[str, Any]: + """ + 获取关系统计信息 + + Args: + relationships: 关系列表 + + Returns: + 统计信息 + """ + stats = { + 'total_relationships': len(relationships), + 'relationships_by_type': {} + } + + for rel in relationships: + rel_type = rel.get('type', 'Unknown') + stats['relationships_by_type'][rel_type] = \ + stats['relationships_by_type'].get(rel_type, 0) + 1 + + return stats diff --git a/dofile/kg_project/src/deep_extraction_pipeline.py b/dofile/kg_project/src/deep_extraction_pipeline.py new file mode 100644 index 0000000..9ea1809 --- /dev/null +++ b/dofile/kg_project/src/deep_extraction_pipeline.py @@ -0,0 +1,406 @@ +# -*- coding: utf-8 -*- +""" +深度文化实体抽取主流程 +从Excel备注字段抽取深层文化实体并生成知识图谱CSV文件 +""" + +import asyncio +import pandas as pd +import json +import logging +from pathlib import Path +from datetime import datetime +from typing import Dict, List, Any + +# 添加模块路径 +import sys +sys.path.append(str(Path(__file__).parent)) + +from knowledge_extraction.deep_entity_extractor import DeepEntityExtractor +from data_processing.entity_normalizer import EntityNormalizer +from data_processing.relationship_builder import RelationshipBuilder + + +class DeepExtractionPipeline: + """深度文化实体抽取流程""" + + def __init__(self, config_file: str): + """ + 初始化流程 + + Args: + config_file: 配置文件路径 + """ + self.config_file = config_file + + # 设置日志 + logging.basicConfig( + level=logging.INFO, + format='%(asctime)s - %(name)s - %(levelname)s - %(message)s' + ) + self.logger = logging.getLogger(__name__) + + # 初始化组件 + self.extractor = DeepEntityExtractor(config_file) + self.normalizer = EntityNormalizer( + str(Path(config_file).parent / 'entity_ontology.yaml') + ) + self.builder = RelationshipBuilder( + str(Path(config_file).parent / 'entity_ontology.yaml') + ) + + self.logger.info("DeepExtractionPipeline初始化完成") + + def load_excel_data(self, excel_file: str, max_rows: int = None) -> List[Dict[str, str]]: + """ + 加载Excel数据 + + Args: + excel_file: Excel文件路径 + max_rows: 最大读取行数(用于测试) + + Returns: + 项目数据列表 + """ + self.logger.info(f"读取Excel文件: {excel_file}") + + df = pd.read_excel(excel_file) + + # 限制行数 + if max_rows: + df = df.head(max_rows) + self.logger.info(f"限制读取行数: {max_rows}") + + # 检查必需列 + required_columns = ['总序号', '项目名称', '备注'] + missing_columns = [col for col in required_columns if col not in df.columns] + + if missing_columns: + self.logger.error(f"Excel缺少必需列: {missing_columns}") + raise ValueError(f"缺少必需列: {missing_columns}") + + # 构建项目数据列表 + projects_data = [] + for idx, row in df.iterrows(): + seq_num = row.get('总序号', idx + 1) + project_name = row.get('项目名称', '') + remark_text = row.get('备注', '') + + # 过滤空备注 + if not remark_text or len(str(remark_text).strip()) < 10: + continue + + projects_data.append({ + 'project_id': f"ICH-{int(seq_num)}", + 'project_name': str(project_name), + 'remark_text': str(remark_text), + 'row_number': idx + 2 # Excel行号(1-based + header) + }) + + self.logger.info(f"加载了{len(projects_data)}个项目的数据") + + return projects_data + + async def run_extraction( + self, + excel_file: str, + output_prefix: str = "", + max_rows: int = None + ) -> Dict[str, Any]: + """ + 执行完整的抽取流程(支持增量保存) + + Args: + excel_file: Excel文件路径 + output_prefix: 输出文件前缀(用于测试) + max_rows: 最大读取行数(用于测试) + + Returns: + 处理结果统计 + """ + self.logger.info("="*60) + self.logger.info("开始深度文化实体抽取流程(增量保存模式)") + self.logger.info("="*60) + + start_time = datetime.now() + + # 1. 加载数据 + projects_data = self.load_excel_data(excel_file, max_rows) + + if not projects_data: + self.logger.error("没有可处理的数据") + return {} + + # 准备输出目录 + output_dir = Path('output') + output_dir.mkdir(parents=True, exist_ok=True) + + # 确定输出文件名 + if output_prefix: + nodes_file = output_dir / f"{output_prefix}_nodes.csv" + rels_file = output_dir / f"{output_prefix}_rels.csv" + report_file = output_dir / f"{output_prefix}_report.md" + else: + nodes_file = output_dir / Path(self.extractor.config['output']['nodes_file']).name + rels_file = output_dir / Path(self.extractor.config['output']['relationships_file']).name + report_file = output_dir / Path(self.extractor.config['output']['report_file']).name + + # 2. 定义增量保存回调函数 + async def save_progress(batch_num, total_batches, extraction_results, processed_projects): + """每批次完成后保存进度(仅保存节点,不构建关系)""" + try: + self.logger.info(f"\n[增量保存] 批次 {batch_num}/{total_batches} 开始保存...") + + # 实体规范化(累积到实体注册表) + self.normalizer.normalize_batch(extraction_results) + + # 生成节点(累积所有已处理的实体) + entity_nodes = self.normalizer.get_entity_nodes() + + # 添加项目节点(仅当前批次) + project_nodes = [ + { + 'id': item['project_id'], + 'label': item['project_name'], + 'type': 'ICH_Project', + 'properties': '{}' + } + for item in processed_projects + ] + + all_nodes = entity_nodes + project_nodes + + # 保存节点 + nodes_df = pd.DataFrame(all_nodes) + nodes_df.to_csv(nodes_file, index=False, encoding='utf-8-sig') + self.logger.info(f"[增量保存] 节点已保存: {nodes_file} ({len(all_nodes)}行)") + + except Exception as e: + self.logger.error(f"[增量保存] 批次 {batch_num} 保存失败: {str(e)}", exc_info=True) + + # 3. LLM批量抽取(带增量保存回调) + self.logger.info("\n步骤1: LLM批量抽取(增量保存模式)") + extraction_results = await self.extractor.batch_extract( + projects_data, + progress_callback=save_progress + ) + + # 4. 最终数据验证和报告 + self.logger.info("\n步骤2: 最终数据验证") + + # 重新规范化所有实体(确保一致性) + entity_id_map = self.normalizer.normalize_batch(extraction_results) + + # 重新构建所有关系 + relationships = self.builder.build_batch_relationships( + projects_data, + extraction_results, + entity_id_map + ) + + # 生成最终节点 + entity_nodes = self.normalizer.get_entity_nodes() + project_nodes = [ + { + 'id': item['project_id'], + 'label': item['project_name'], + 'type': 'ICH_Project', + 'properties': '{}' + } + for item in projects_data + ] + all_nodes = entity_nodes + project_nodes + + # 数据验证 + node_ids = set(node['id'] for node in all_nodes) + validation_report = self.builder.validate_relationships(relationships, node_ids) + + # 保存最终结果 + self.logger.info("\n步骤3: 保存最终结果") + nodes_df = pd.DataFrame(all_nodes) + nodes_df.to_csv(nodes_file, index=False, encoding='utf-8-sig') + self.logger.info(f"最终节点已保存: {nodes_file} ({len(all_nodes)}行)") + + rels_df = pd.DataFrame(relationships) + rels_df.to_csv(rels_file, index=False, encoding='utf-8-sig') + self.logger.info(f"最终关系已保存: {rels_file} ({len(relationships)}行)") + + # 生成报告 + self.logger.info("\n步骤4: 生成报告") + self.generate_report( + report_file, + projects_data, + all_nodes, + relationships, + validation_report, + start_time + ) + + end_time = datetime.now() + duration = (end_time - start_time).total_seconds() + + # 返回统计信息 + result_stats = { + 'total_projects': len(projects_data), + 'total_nodes': len(all_nodes), + 'total_relationships': len(relationships), + 'entity_nodes': len(entity_nodes), + 'project_nodes': len(project_nodes), + 'duration_seconds': duration, + 'validation_report': validation_report + } + + self.logger.info("\n"+"="*60) + self.logger.info(f"抽取完成!耗时: {duration:.2f}秒") + self.logger.info(f"项目数: {result_stats['total_projects']}") + self.logger.info(f"节点数: {result_stats['total_nodes']}") + self.logger.info(f"关系数: {result_stats['total_relationships']}") + self.logger.info("="*60) + + return result_stats + + def generate_report( + self, + report_file: Path, + projects_data: List[Dict[str, str]], + nodes: List[Dict[str, Any]], + relationships: List[Dict[str, Any]], + validation_report: Dict[str, Any], + start_time: datetime + ): + """ + 生成抽取报告 + + Args: + report_file: 报告文件路径 + projects_data: 项目数据 + nodes: 节点列表 + relationships: 关系列表 + validation_report: 验证报告 + start_time: 开始时间 + """ + report_lines = [] + + # 标题 + report_lines.append("# 深度文化实体抽取报告\n") + report_lines.append(f"**生成时间**: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n") + report_lines.append(f"**耗时**: {(datetime.now() - start_time).total_seconds():.2f}秒\n") + report_lines.append("---\n\n") + + # 1. 处理统计 + report_lines.append("## 1. 处理统计\n\n") + report_lines.append(f"- **处理项目数**: {len(projects_data)}\n") + report_lines.append(f"- **总节点数**: {len(nodes)}\n") + report_lines.append(f"- **总关系数**: {len(relationships)}\n") + report_lines.append(f"- **实体节点数**: {len([n for n in nodes if n['type'] != 'ICH_Project'])}\n") + report_lines.append(f"- **项目节点数**: {len([n for n in nodes if n['type'] == 'ICH_Project'])}\n") + + # 2. 节点类型分布 + report_lines.append("\n## 2. 节点类型分布\n\n") + node_types = {} + for node in nodes: + node_type = node['type'] + node_types[node_type] = node_types.get(node_type, 0) + 1 + + report_lines.append("| 节点类型 | 数量 | 占比 |\n") + report_lines.append("|---------|------|------|\n") + for node_type, count in sorted(node_types.items()): + percentage = (count / len(nodes) * 100) if len(nodes) > 0 else 0 + report_lines.append(f"| {node_type} | {count} | {percentage:.1f}% |\n") + + # 3. 关系类型分布 + report_lines.append("\n## 3. 关系类型分布\n\n") + rel_types = {} + for rel in relationships: + rel_type = rel['type'] + rel_types[rel_type] = rel_types.get(rel_type, 0) + 1 + + report_lines.append("| 关系类型 | 数量 | 占比 |\n") + report_lines.append("|---------|------|------|\n") + for rel_type, count in sorted(rel_types.items()): + percentage = (count / len(relationships) * 100) if len(relationships) > 0 else 0 + report_lines.append(f"| {rel_type} | {count} | {percentage:.1f}% |\n") + + # 4. 数据质量 + report_lines.append("\n## 4. 数据质量\n\n") + report_lines.append(f"- **有效关系**: {validation_report['valid_relationships']}\n") + report_lines.append(f"- **无效关系**: {validation_report['invalid_relationships']}\n") + + if validation_report['broken_links']: + report_lines.append(f"\n**断链警告**: {len(validation_report['broken_links'])}个\n") + report_lines.append("```json\n") + report_lines.append(json.dumps(validation_report['broken_links'][:10], ensure_ascii=False, indent=2)) + if len(validation_report['broken_links']) > 10: + report_lines.append(f"\n... (还有{len(validation_report['broken_links'])-10}个)") + report_lines.append("\n```\n") + + # 5. 项目详情(前5个) + report_lines.append("\n## 5. 项目抽取详情(前5个)\n\n") + + for idx, (project_data, node) in enumerate(zip(projects_data[:5], nodes[:5])): + if node['type'] != 'ICH_Project': + continue + + report_lines.append(f"### {idx+1}. {project_data['project_name']}\n\n") + report_lines.append(f"**项目ID**: {project_data['project_id']}\n\n") + report_lines.append(f"**备注**: {project_data['remark_text'][:200]}...\n\n") + + # 查找相关关系 + related_rels = [r for r in relationships if r['source'] == project_data['project_id']] + if related_rels: + report_lines.append(f"**关系数**: {len(related_rels)}\n\n") + report_lines.append("| 关系类型 | 目标实体 |\n") + report_lines.append("|---------|---------|\n") + for rel in related_rels[:10]: + target_node = next((n for n in nodes if n['id'] == rel['target']), None) + if target_node: + report_lines.append(f"| {rel['type']} | {target_node['label']} |\n") + if len(related_rels) > 10: + report_lines.append(f"| ... | 还有{len(related_rels)-10}个关系 |\n") + + report_lines.append("\n") + + # 6. 输出文件 + report_lines.append("## 6. 输出文件\n\n") + report_lines.append(f"- **节点文件**: `{report_file.parent / (report_file.stem.replace('_report', '') + '_nodes.csv')}`\n") + report_lines.append(f"- **关系文件**: `{report_file.parent / (report_file.stem.replace('_report', '') + '_rels.csv')}`\n") + + # 保存报告 + with open(report_file, 'w', encoding='utf-8') as f: + f.writelines(report_lines) + + self.logger.info(f"报告已保存: {report_file}") + + +async def main(): + """主函数""" + import sys + + # 配置文件 + config_file = r"E:\Project\2026_KG_ICH\dofile\kg_project\config\deep_extraction_config.yaml" + + # 数据文件:从命令行参数获取,或使用默认完整数据文件 + if len(sys.argv) > 1: + excel_file = sys.argv[1] + else: + excel_file = r"E:\Project\2026_KG_ICH\data\黑龙江国家级和省级非遗名单.xlsx" + + # 创建流程 + pipeline = DeepExtractionPipeline(config_file) + + # 执行抽取 + results = await pipeline.run_extraction( + excel_file=excel_file, + output_prefix="", # 不使用前缀(完整抽取) + max_rows=None # 读取所有行 + ) + + print("\n" + "="*60) + print("抽取完成!") + print("="*60) + print(f"\n结果统计:") + print(json.dumps(results, indent=2, ensure_ascii=False)) + + +if __name__ == '__main__': + asyncio.run(main()) diff --git a/dofile/kg_project/src/knowledge_extraction/__init__.py b/dofile/kg_project/src/knowledge_extraction/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/dofile/kg_project/src/knowledge_extraction/deep_entity_extractor.py b/dofile/kg_project/src/knowledge_extraction/deep_entity_extractor.py new file mode 100644 index 0000000..5239762 --- /dev/null +++ b/dofile/kg_project/src/knowledge_extraction/deep_entity_extractor.py @@ -0,0 +1,420 @@ +# -*- coding: utf-8 -*- +""" +深度文化实体抽取器 +利用LLM从非遗项目备注中抽取深层文化实体和关系 +""" + +import json +import asyncio +import logging +from typing import Dict, List, Any, Optional +from pathlib import Path +import yaml +from langchain_deepseek import ChatDeepSeek +from langchain_core.messages import HumanMessage, SystemMessage + + +class DeepEntityExtractor: + """深度文化实体抽取器""" + + def __init__(self, config_file: str): + """ + 初始化抽取器 + + Args: + config_file: 配置文件路径 + """ + # 加载配置 + with open(config_file, 'r', encoding='utf-8') as f: + self.config = yaml.safe_load(f) + + # 设置日志 + log_config = self.config.get('logging', {}) + log_file = log_config.get('log_file', 'logs/deep_extraction.log') + Path(log_file).parent.mkdir(parents=True, exist_ok=True) + + # 创建logger + self.logger = logging.getLogger(__name__) + self.logger.setLevel(getattr(logging, log_config.get('level', 'INFO'))) + + # 清除已有的handlers + self.logger.handlers.clear() + + # 文件handler(实时刷新) + file_handler = logging.FileHandler(log_file, encoding='utf-8') + file_handler.setLevel(getattr(logging, log_config.get('level', 'INFO'))) + file_formatter = logging.Formatter(log_config.get('format', '%(asctime)s - %(name)s - %(levelname)s - %(message)s')) + file_handler.setFormatter(file_formatter) + # 强制实时刷新 + file_handler.flush = lambda: file_handler.stream.flush() + self.logger.addHandler(file_handler) + + # 控制台handler + console_handler = logging.StreamHandler() + console_handler.setLevel(getattr(logging, log_config.get('level', 'INFO'))) + console_formatter = logging.Formatter(log_config.get('format', '%(asctime)s - %(name)s - %(levelname)s - %(message)s')) + console_handler.setFormatter(console_formatter) + self.logger.addHandler(console_handler) + + # 加载实体本体配置 + ontology_file = Path(config_file).parent / 'entity_ontology.yaml' + with open(ontology_file, 'r', encoding='utf-8') as f: + self.ontology = yaml.safe_load(f) + + # 初始化LLM + self._init_llm() + + # 构建提示词 + self.system_prompt = self._build_system_prompt() + + self.logger.info("DeepEntityExtractor初始化完成") + + def _init_llm(self): + """初始化LLM模型""" + llm_config = self.config['llm'] + + # 读取API密钥(从现有的api_keys.yaml) + api_keys_file = Path(__file__).parent.parent.parent / 'config' / 'api_keys.yaml' + with open(api_keys_file, 'r', encoding='utf-8') as f: + api_keys = yaml.safe_load(f) + + # 适配API密钥格式 + api_key = api_keys.get('deepseek_api_key', '') + if not api_key: + # 尝试嵌套格式 + api_key = api_keys.get('deepseek', {}).get('api_key', '') + + self.llm = ChatDeepSeek( + model=llm_config['model'], + temperature=llm_config['temperature'], + max_tokens=llm_config['max_tokens'], + api_key=api_key + ) + + self.logger.info(f"LLM初始化完成: {llm_config['model']}") + + def _build_system_prompt(self) -> str: + """构建系统提示词""" + entity_prompts = self.config.get('entity_type_prompts', {}) + relationship_prompts = self.config.get('relationship_type_prompts', {}) + + prompt = """你是一位非物质文化遗产领域的专家,擅长从项目描述中识别深层文化实体和关系。 + +【任务】 +请从非遗项目描述中识别以下9类深层文化实体和9种深层关系: + +=== 实体类型 === +""" + + # 添加实体类型说明 + for entity_type, description in entity_prompts.items(): + prompt += f"{entity_type}:{description}\n" + + prompt += "\n=== 关系类型 ===\n" + + # 添加关系类型说明 + for rel_type, description in relationship_prompts.items(): + prompt += f"{rel_type}:{description}\n" + + prompt += """ +【输出要求】 +1. 严格按JSON格式输出,不要包含任何其他文本 +2. 只输出明确的、文本中提到的实体和关系,不要臆测 +3. 实体text必须从原文中提取,不要自行改写 +4. 同一个实体只识别一次,避免重复 +5. 关系必须基于文本中的明确描述 +6. 如果某类实体或关系不存在,相应数组为空 +7. 确保JSON格式正确,可以被Python解析 + +【JSON输出格式】 +```json +{ + "entities": [ + { + "text": "实体文本(如:鱼皮)", + "type": "实体类型(如:Material)", + "attributes": { + "name": "标准名称", + "category": "类别(如:动物材料)", + "description": "详细描述" + } + } + ], + "relationships": [ + { + "source_entity": "ICH-{项目ID}", + "target_entity": "目标实体文本", + "type": "关系类型", + "properties": { + "description": "关系描述", + "context": "上下文信息" + } + } + ], + "summary": { + "total_entities": 0, + "total_relationships": 0 + } +} +``` +""" + + return prompt + + async def extract_from_remark( + self, + project_id: str, + project_name: str, + remark_text: str + ) -> Optional[Dict[str, Any]]: + """ + 从单条备注抽取实体和关系 + + Args: + project_id: 项目ID(如ICH-1) + project_name: 项目名称 + remark_text: 备注文本 + + Returns: + 抽取结果(包含entities, relationships, summary) + """ + if not remark_text or len(remark_text.strip()) < 10: + self.logger.warning(f"项目 {project_id} 备注文本过短,跳过抽取") + return None + + # 构建用户提示词 + user_prompt = f"""【项目信息】 +项目名称:{project_name} +项目ID:{project_id} + +【描述文本】 +{remark_text} + +请从上述描述中识别深层文化实体和关系,严格按JSON格式输出。""" + + try: + # 调用LLM + messages = [ + SystemMessage(content=self.system_prompt), + HumanMessage(content=user_prompt) + ] + + max_retries = self.config['llm']['max_retries'] + retry_delay = self.config['llm']['retry_delay'] + request_timeout = self.config['llm'].get('request_timeout', 120) + + for attempt in range(max_retries): + try: + self.logger.info(f"项目 {project_id} 开始LLM调用(尝试{attempt+1}/{max_retries})") + + # 添加超时控制 + response = await asyncio.wait_for( + self.llm.ainvoke(messages), + timeout=request_timeout + ) + result_text = response.content + + self.logger.info(f"项目 {project_id} LLM调用成功,开始提取JSON") + + # 提取JSON + extraction_result = self._extract_json(result_text) + + if extraction_result: + # 验证结果 + if self._validate_extraction(extraction_result): + self.logger.info(f"项目 {project_id} 抽取成功:{extraction_result['summary']['total_entities']}个实体,{extraction_result['summary']['total_relationships']}个关系") + return extraction_result + else: + self.logger.warning(f"项目 {project_id} 抽取结果验证失败") + else: + self.logger.warning(f"项目 {project_id} JSON提取失败") + + except asyncio.TimeoutError: + self.logger.error(f"项目 {project_id} LLM调用超时({request_timeout}秒)(尝试{attempt+1}/{max_retries})") + if attempt < max_retries - 1: + await asyncio.sleep(retry_delay) + else: + self.logger.error(f"项目 {project_id} 达到最大重试次数,放弃抽取") + raise + except Exception as e: + self.logger.error(f"项目 {project_id} LLM调用失败(尝试{attempt+1}/{max_retries}): {type(e).__name__}: {str(e)}") + if attempt < max_retries - 1: + await asyncio.sleep(retry_delay) + else: + raise + + except Exception as e: + self.logger.error(f"项目 {project_id} 抽取失败: {str(e)}", exc_info=True) + return None + + def _extract_json(self, text: str) -> Optional[Dict[str, Any]]: + """ + 从文本中提取JSON + + Args: + text: LLM返回的文本 + + Returns: + 解析后的JSON对象,失败返回None + """ + # 尝试直接解析 + try: + return json.loads(text) + except json.JSONDecodeError: + pass + + # 尝试提取JSON块 + import re + json_pattern = r'```json\s*(.*?)\s*```' + match = re.search(json_pattern, text, re.DOTALL) + + if match: + try: + return json.loads(match.group(1)) + except json.JSONDecodeError: + pass + + # 尝试提取花括号内容 + brace_pattern = r'\{.*\}' + match = re.search(brace_pattern, text, re.DOTALL) + + if match: + try: + return json.loads(match.group(0)) + except json.JSONDecodeError: + pass + + self.logger.error(f"无法从文本中提取有效JSON: {text[:200]}...") + return None + + def _validate_extraction(self, result: Dict[str, Any]) -> bool: + """ + 验证抽取结果 + + Args: + result: 抽取结果 + + Returns: + 是否有效 + """ + # 检查必需字段 + required_fields = ['entities', 'relationships', 'summary'] + for field in required_fields: + if field not in result: + self.logger.error(f"缺少必需字段: {field}") + return False + + # 检查entities格式 + if not isinstance(result['entities'], list): + self.logger.error("entities必须是列表") + return False + + for entity in result['entities']: + if not all(k in entity for k in ['text', 'type', 'attributes']): + self.logger.error(f"实体缺少必需字段: {entity}") + return False + + # 检查实体类型是否合法 + entity_type = entity['type'] + if entity_type not in self.ontology['entity_types']: + self.logger.warning(f"未知实体类型: {entity_type}") + + # 检查relationships格式 + if not isinstance(result['relationships'], list): + self.logger.error("relationships必须是列表") + return False + + for rel in result['relationships']: + if not all(k in rel for k in ['source_entity', 'target_entity', 'type', 'properties']): + self.logger.error(f"关系缺少必需字段: {rel}") + return False + + # 检查关系类型是否合法 + rel_type = rel['type'] + if rel_type not in self.ontology['relationship_types']: + self.logger.warning(f"未知关系类型: {rel_type}") + + return True + + async def batch_extract( + self, + projects_data: List[Dict[str, str]], + batch_size: int = None, + progress_callback=None + ) -> List[Optional[Dict[str, Any]]]: + """ + 批量抽取实体和关系 + + Args: + projects_data: 项目数据列表,每项包含project_id, project_name, remark_text + batch_size: 批次大小(从配置文件读取) + progress_callback: 进度回调函数,每批次完成后调用 + + Returns: + 抽取结果列表 + """ + if batch_size is None: + batch_size = self.config['batch_processing']['batch_size'] + + batch_timeout = self.config['llm'].get('batch_timeout', 300) + incremental_save = self.config['batch_processing'].get('incremental_save', False) + + results = [] + total = len(projects_data) + + self.logger.info(f"="*60) + self.logger.info(f"开始批量抽取,共{total}个项目,批次大小{batch_size}") + self.logger.info(f"增量保存: {'启用' if incremental_save else '禁用'}") + self.logger.info(f"请求超时: {self.config['llm'].get('request_timeout', 120)}秒") + self.logger.info(f"批次超时: {batch_timeout}秒") + self.logger.info(f"="*60) + + for i in range(0, total, batch_size): + batch = projects_data[i:i+batch_size] + batch_num = i // batch_size + 1 + total_batches = (total + batch_size - 1) // batch_size + + self.logger.info(f"-"*60) + self.logger.info(f"开始处理批次 {batch_num}/{total_batches},包含{len(batch)}个项目") + for item in batch: + self.logger.info(f" - {item['project_id']}: {item['project_name']}") + + # 并发处理当前批次 + batch_tasks = [ + self.extract_from_remark( + item['project_id'], + item['project_name'], + item['remark_text'] + ) + for item in batch + ] + + try: + # 添加批次级别的超时控制 + batch_results = await asyncio.wait_for( + asyncio.gather(*batch_tasks, return_exceptions=True), + timeout=batch_timeout + ) + results.extend(batch_results) + + # 统计本批次结果 + success_count = sum(1 for r in batch_results if r is not None and not isinstance(r, Exception)) + error_count = len(batch_results) - success_count + + self.logger.info(f"批次 {batch_num}/{total_batches} 完成 - 成功: {success_count}, 失败: {error_count}") + + # 调用进度回调(用于增量保存) + if progress_callback: + await progress_callback(batch_num, total_batches, results, projects_data[:i+len(batch)]) + + except asyncio.TimeoutError: + self.logger.error(f"批次 {batch_num}/{total_batches} 超时({batch_timeout}秒),部分请求失败") + # 将未完成的结果标记为None + for item in batch[len(results):]: + results.append(None) + + self.logger.info(f"="*60) + self.logger.info(f"批量抽取完成,共处理{len(results)}个项目") + self.logger.info(f"="*60) + + return results diff --git a/dofile/kg_project/src/knowledge_extraction/llm_extractor.py b/dofile/kg_project/src/knowledge_extraction/llm_extractor.py new file mode 100644 index 0000000..28a3234 --- /dev/null +++ b/dofile/kg_project/src/knowledge_extraction/llm_extractor.py @@ -0,0 +1,486 @@ +""" +DeepSeek实体识别模块 +使用DeepSeek API进行非遗知识抽取 +""" + +import yaml +import asyncio +import json +from pathlib import Path +from typing import Dict, List, Optional, Any +import logging +from datetime import datetime + +try: + from langchain_deepseek import ChatDeepSeek + from langchain_core.messages import HumanMessage, SystemMessage +except ImportError: + print("错误: 请先安装依赖包") + print("运行: pip install langchain-deepseek langchain-core") + raise + + +class DeepSeekExtractor: + """基于DeepSeek的知识抽取器""" + + def __init__(self, config_file: str = 'config/ich_config.yaml'): + """ + 初始化抽取器 + + Args: + config_file: 配置文件路径 + """ + # 加载配置 + self.config = self._load_config(config_file) + self.api_key = self._load_api_key() + + # 初始化模型 + self.model = ChatDeepSeek( + model=self.config['api']['model'], + api_key=self.api_key, + temperature=self.config['api']['temperature'], + max_tokens=self.config['api']['max_tokens'] + ) + + # 设置日志 + self.logger = self._setup_logger() + + # 加载本体配置 + self.entity_types = self.config['extraction']['entity_types'] + self.relation_types = self.config['extraction']['relation_types'] + + def _setup_logger(self): + """设置日志""" + log_dir = Path(self.config['paths']['logs_dir']) + log_dir.mkdir(parents=True, exist_ok=True) + + log_file = log_dir / f"extraction_{datetime.now().strftime('%Y%m%d')}.log" + logging.basicConfig( + level=getattr(logging, self.config['logging']['level']), + format=self.config['logging']['format'], + handlers=[ + logging.FileHandler(log_file, encoding='utf-8'), + logging.StreamHandler() + ] + ) + return logging.getLogger(__name__) + + def _load_config(self, config_file: str) -> Dict: + """加载配置""" + config_path = Path(config_file) + if not config_path.exists(): + raise FileNotFoundError(f"配置文件不存在: {config_file}") + + with open(config_path, 'r', encoding='utf-8') as f: + return yaml.safe_load(f) + + def _load_api_key(self) -> str: + """加载API密钥""" + api_config_file = Path(self.config['api']['config_file']) + if not api_config_file.exists(): + raise FileNotFoundError(f"API配置文件不存在: {api_config_file}") + + with open(api_config_file, 'r', encoding='utf-8') as f: + api_config = yaml.safe_load(f) + + return api_config['deepseek_api_key'] + + def extract_json_from_response(self, response_text: str) -> str: + """ + 从API响应中提取JSON内容 + + Args: + response_text: API响应文本 + + Returns: + str: 提取的JSON字符串 + """ + # 如果响应包含markdown代码块,提取其中的JSON + if '```json' in response_text: + start = response_text.find('```json') + 7 + end = response_text.find('```', start) + if start > 6 and end > start: + return response_text[start:end].strip() + elif '```' in response_text: + start = response_text.find('```') + 3 + end = response_text.find('```', start) + if start > 2 and end > start: + content = response_text[start:end].strip() + if not content.startswith('json'): + return content + return content[5:].strip() if content.startswith('json') else content.strip() + + # 否则直接返回原始文本 + return response_text.strip() + + async def extract_entities_from_text( + self, + text: str, + project_name: str = "", + max_retries: int = 3 + ) -> Optional[Dict[str, Any]]: + """ + 从文本中抽取实体 + + Args: + text: 输入文本 + project_name: 项目名称(可选) + max_retries: 最大重试次数 + + Returns: + Dict: 抽取结果 + """ + # 构建提示词 + entity_types_str = "、".join(self.entity_types) + + prompt = f""" +你是一位非物质文化遗产领域的专家。请从以下文本中识别非遗相关实体,并提取属性。 + +项目名称:{project_name} + +文本内容: +{text} + +请识别以下类型的实体: +{entity_types_str} + +对于每个实体,请提取以下信息: +1. 实体文本 +2. 实体类型 +3. 相关属性(如:民族、地点、时间、技艺特点等) + +输出格式(JSON): +{{ + "entities": [ + {{ + "text": "实体文本", + "type": "实体类型", + "attributes": {{ + "ethnic_group": "民族(如果适用)", + "location": "地点(如果适用)", + "time_period": "时期(如果适用)", + "skill_feature": "技艺特点(如果适用)", + "cultural_value": "文化价值(如果适用)" + }} + }} + ], + "relationships": [ + {{ + "from": "实体1", + "to": "实体2", + "type": "关系类型", + "description": "关系描述" + }} + ] +}} + +请确保输出是有效的JSON格式。 +""" + + # 调用API + for attempt in range(max_retries): + try: + self.logger.info(f"开始抽取实体 (尝试 {attempt + 1}/{max_retries})") + + messages = [HumanMessage(content=prompt)] + response = await self.model.ainvoke(messages) + result_text = response.content + + # 提取JSON + json_text = self.extract_json_from_response(result_text) + + # 解析JSON + try: + result = json.loads(json_text) + + # 添加元数据 + result['metadata'] = { + 'project_name': project_name, + 'extraction_time': datetime.now().isoformat(), + 'text_length': len(text), + 'model': self.config['api']['model'] + } + + self.logger.info(f"成功抽取 {len(result.get('entities', []))} 个实体") + return result + + except json.JSONDecodeError as e: + self.logger.warning(f"JSON解析失败: {str(e)}") + self.logger.debug(f"响应文本: {result_text[:500]}") + if attempt == max_retries - 1: + raise + + except Exception as e: + self.logger.error(f"API调用失败 (尝试 {attempt + 1}/{max_retries}): {str(e)}") + if attempt == max_retries - 1: + return None + + # 等待后重试 + await asyncio.sleep(self.config['processing']['retry_delay']) + + return None + + async def extract_relationships( + self, + entities: List[Dict], + context: str = "", + max_retries: int = 3 + ) -> Optional[List[Dict[str, Any]]]: + """ + 抽取实体间的关系 + + Args: + entities: 实体列表 + context: 上下文文本 + max_retries: 最大重试次数 + + Returns: + List[Dict]: 关系列表 + """ + if not entities: + return [] + + # 构建实体描述 + entity_descriptions = [] + for entity in entities: + desc = f"- {entity.get('text', '')} ({entity.get('type', '')})" + if entity.get('attributes'): + attrs = ", ".join([f"{k}={v}" for k, v in entity['attributes'].items() if v]) + if attrs: + desc += f" [{attrs}]" + entity_descriptions.append(desc) + + entities_str = "\n".join(entity_descriptions) + relations_str = "、".join(self.relation_types) + + prompt = f""" +基于以下实体和上下文,识别实体间的关系: + +实体列表: +{entities_str} + +上下文: +{context} + +可能的关系类型: +{relations_str} + +对于每个关系,请提供: +1. 头实体(from) +2. 尾实体(to) +3. 关系类型 +4. 关系描述 + +输出格式(JSON): +{{ + "relationships": [ + {{ + "from": "实体1文本", + "to": "实体2文本", + "type": "关系类型", + "description": "关系描述", + "confidence": 0.9 + }} + ] +}} + +请确保输出是有效的JSON格式。 +""" + + # 调用API + for attempt in range(max_retries): + try: + messages = [HumanMessage(content=prompt)] + response = await self.model.ainvoke(messages) + result_text = response.content + + # 提取JSON + json_text = self.extract_json_from_response(result_text) + + # 解析JSON + try: + result = json.loads(json_text) + relationships = result.get('relationships', []) + + self.logger.info(f"成功抽取 {len(relationships)} 个关系") + return relationships + + except json.JSONDecodeError as e: + self.logger.warning(f"JSON解析失败: {str(e)}") + if attempt == max_retries - 1: + raise + + except Exception as e: + self.logger.error(f"API调用失败 (尝试 {attempt + 1}/{max_retries}): {str(e)}") + if attempt == max_retries - 1: + return None + + await asyncio.sleep(self.config['processing']['retry_delay']) + + return None + + async def enrich_project_entity( + self, + project_data: Dict[str, Any], + max_retries: int = 3 + ) -> Optional[Dict[str, Any]]: + """ + 丰富非遗项目实体信息 + + Args: + project_data: 项目数据 + max_retries: 最大重试次数 + + Returns: + Dict: 丰富后的实体信息 + """ + project_name = project_data.get('properties', {}).get('name', '') + description = project_data.get('properties', {}).get('description', '') + + prompt = f""" +你是一位非物质文化遗产领域的专家。请分析以下非遗项目,提取和丰富实体信息。 + +项目名称:{project_name} + +项目描述: +{description} + +请提取以下信息: + +1. **民族特色**:是否与特定少数民族相关?(满族、赫哲族、鄂伦春族等) +2. **地域特征**:体现哪些黑龙江地域特征?(寒地、冰雪、森林、江河等) +3. **技艺特点**:核心技艺特点是什么? +4. **文化价值**:有哪些重要的文化价值? +5. **传承方式**:如何传承?(家族传承、师徒制度、口传心授等) +6. **濒危状况**:是否濒危?原因是什么? +7. **保护措施**:有哪些保护措施? + +输出格式(JSON): +{{ + "ethnic_features": ["民族1", "民族2"], + "regional_characteristics": ["特征1", "特征2"], + "skill_features": ["技艺特点1", "技艺特点2"], + "cultural_values": ["价值1", "价值2"], + "transmission_methods": ["方式1", "方式2"], + "endangerment_status": "濒危状况描述", + "protection_measures": ["措施1", "措施2"], + "summary": "项目总结(50字以内)" +}} + +请确保输出是有效的JSON格式。 +""" + + # 调用API + for attempt in range(max_retries): + try: + messages = [HumanMessage(content=prompt)] + response = await self.model.ainvoke(messages) + result_text = response.content + + # 提取JSON + json_text = self.extract_json_from_response(result_text) + + # 解析JSON + try: + result = json.loads(json_text) + + self.logger.info(f"成功丰富项目实体信息: {project_name}") + return result + + except json.JSONDecodeError as e: + self.logger.warning(f"JSON解析失败: {str(e)}") + if attempt == max_retries - 1: + raise + + except Exception as e: + self.logger.error(f"API调用失败 (尝试 {attempt + 1}/{max_retries}): {str(e)}") + if attempt == max_retries - 1: + return None + + await asyncio.sleep(self.config['processing']['retry_delay']) + + return None + + async def batch_extract( + self, + items: List[Dict[str, Any]], + concurrent: int = 5 + ) -> List[Optional[Dict[str, Any]]]: + """ + 批量抽取 + + Args: + items: 待处理项目列表 + concurrent: 并发数 + + Returns: + List[Dict]: 抽取结果列表 + """ + results = [] + batch_size = concurrent + + for i in range(0, len(items), batch_size): + batch = items[i:i + batch_size] + self.logger.info(f"处理批次 {i//batch_size + 1}: {len(batch)} 个项目") + + # 并发处理 + tasks = [] + for item in batch: + project_name = item.get('properties', {}).get('name', '') + description = item.get('properties', {}).get('description', '') + + if description: + task = self.enrich_project_entity(item) + tasks.append(task) + else: + tasks.append(asyncio.sleep(0)) # 占位任务 + + batch_results = await asyncio.gather(*tasks, return_exceptions=True) + results.extend(batch_results) + + # 显示进度 + completed = min(i + batch_size, len(items)) + self.logger.info(f"进度: {completed}/{len(items)}") + + return results + + +def main(): + """主函数 - 测试实体抽取""" + import sys + + # 添加项目根目录到路径 + sys.path.insert(0, str(Path(__file__).parent.parent.parent)) + + # 创建抽取器 + extractor = DeepSeekExtractor('config/ich_config.yaml') + + # 测试文本 + test_text = """ + 桦树皮制作技艺是鄂伦春族的传统手工艺,利用桦树皮制作各种生活用品。 + 传承人莫桂树2008年被评为国家级传承人。这项技艺体现了鄂伦春族对自然资源的 + 巧妙利用,具有鲜明的渔猎文化特色。 + """ + + # 测试实体抽取 + print("="*60) + print("测试实体抽取") + print("="*60) + + async def test(): + result = await extractor.extract_entities_from_text( + text=test_text, + project_name="桦树皮制作技艺" + ) + + if result: + print("\n抽取结果:") + print(json.dumps(result, indent=2, ensure_ascii=False)) + else: + print("抽取失败") + + asyncio.run(test()) + + +if __name__ == '__main__': + main() diff --git a/dofile/kg_project/src/main.py b/dofile/kg_project/src/main.py new file mode 100644 index 0000000..6ad5349 --- /dev/null +++ b/dofile/kg_project/src/main.py @@ -0,0 +1,276 @@ +""" +黑龙江省非物质文化遗产知识图谱构建 - 主处理脚本 +""" + +import json +import asyncio +import yaml +from pathlib import Path +from typing import Dict, List, Any +import logging +from datetime import datetime +import sys + +# 添加项目路径 +sys.path.insert(0, str(Path(__file__).parent)) + +from knowledge_extraction.llm_extractor import DeepSeekExtractor + + +class ICKnowledgeGraphBuilder: + """非遗知识图谱构建器""" + + def __init__(self, config_file: str = 'config/ich_config.yaml'): + """初始化构建器""" + # 加载配置 + with open(config_file, 'r', encoding='utf-8') as f: + self.config = yaml.safe_load(f) + + # 设置路径 + self.project_root = Path(__file__).parent.parent + self.data_dir = self.project_root / self.config['paths']['data_dir'] + self.output_dir = self.project_root / self.config['paths']['output_dir'] + self.logs_dir = self.project_root / self.config['paths']['logs_dir'] + + # 创建目录 + self.output_dir.mkdir(parents=True, exist_ok=True) + self.logs_dir.mkdir(parents=True, exist_ok=True) + + # 设置日志 + self.logger = self._setup_logger() + + # 初始化DeepSeek抽取器 + self.extractor = DeepSeekExtractor(config_file) + + # 加载进度 + self.progress_file = self.output_dir / 'progress.json' + self.progress = self._load_progress() + + def _setup_logger(self): + """设置日志""" + log_file = self.logs_dir / f"build_{datetime.now().strftime('%Y%m%d_%H%M%S')}.log" + logging.basicConfig( + level=getattr(logging, self.config['logging']['level']), + format=self.config['logging']['format'], + handlers=[ + logging.FileHandler(log_file, encoding='utf-8'), + logging.StreamHandler() + ] + ) + return logging.getLogger(__name__) + + def _load_progress(self) -> Dict: + """加载进度""" + if self.progress_file.exists(): + with open(self.progress_file, 'r', encoding='utf-8') as f: + return json.load(f) + return { + 'total': 0, + 'processed': [], + 'successful': [], + 'failed': [], + 'last_update': None + } + + def _save_progress(self): + """保存进度""" + self.progress['last_update'] = datetime.now().isoformat() + with open(self.progress_file, 'w', encoding='utf-8') as f: + json.dump(self.progress, f, indent=2, ensure_ascii=False) + + def load_nodes(self) -> List[Dict[str, Any]]: + """加载节点数据""" + nodes_file = self.output_dir / 'nodes.json' + if not nodes_file.exists(): + self.logger.error(f"节点文件不存在: {nodes_file}") + return [] + + with open(nodes_file, 'r', encoding='utf-8') as f: + nodes = json.load(f) + + self.logger.info(f"加载了 {len(nodes)} 个节点") + self.progress['total'] = len(nodes) + + return nodes + + async def enrich_single_node(self, node: Dict[str, Any]) -> Dict[str, Any]: + """丰富单个节点""" + node_id = node.get('id') + project_name = node.get('properties', {}).get('name', '') + + self.logger.info(f"开始处理节点: {node_id} - {project_name}") + + try: + # 调用DeepSeek丰富信息 + enriched_data = await self.extractor.enrich_project_entity(node) + + if enriched_data: + # 合并到原节点 + node['properties']['enriched_data'] = enriched_data + node['properties']['enriched_at'] = datetime.now().isoformat() + + self.logger.info(f"成功丰富节点: {node_id}") + return { + 'node_id': node_id, + 'status': 'success', + 'enriched_data': enriched_data + } + else: + self.logger.warning(f"丰富节点失败(无返回数据): {node_id}") + return { + 'node_id': node_id, + 'status': 'failed', + 'error': 'No data returned' + } + + except Exception as e: + self.logger.error(f"丰富节点失败: {node_id} - {str(e)}") + return { + 'node_id': node_id, + 'status': 'failed', + 'error': str(e) + } + + async def batch_enrich_nodes( + self, + nodes: List[Dict[str, Any]], + max_nodes: int = None, + concurrent: int = 5 + ): + """批量丰富节点""" + # 过滤已处理的节点 + pending_nodes = [ + node for node in nodes + if node.get('id') not in self.progress['processed'] + ] + + # 限制处理数量(用于测试) + if max_nodes: + pending_nodes = pending_nodes[:max_nodes] + + self.logger.info(f"开始批量处理: {len(pending_nodes)} 个节点待处理") + + # 分批处理 + batch_size = concurrent + for i in range(0, len(pending_nodes), batch_size): + batch = pending_nodes[i:i + batch_size] + batch_num = i // batch_size + 1 + total_batches = (len(pending_nodes) + batch_size - 1) // batch_size + + self.logger.info(f"处理批次 {batch_num}/{total_batches}: {len(batch)} 个节点") + + # 并发处理 + tasks = [self.enrich_single_node(node) for node in batch] + results = await asyncio.gather(*tasks, return_exceptions=True) + + # 更新进度 + for result in results: + if isinstance(result, Exception): + self.logger.error(f"处理异常: {str(result)}") + continue + + node_id = result.get('node_id') + status = result.get('status') + + if status == 'success': + self.progress['successful'].append(node_id) + else: + self.progress['failed'].append(node_id) + + self.progress['processed'].append(node_id) + + # 保存进度 + self._save_progress() + + # 显示进度 + self._print_progress() + + def _print_progress(self): + """打印进度""" + total = self.progress['total'] + processed = len(self.progress['processed']) + successful = len(self.progress['successful']) + failed = len(self.progress['failed']) + + print("\n" + "="*60) + print("处理进度") + print("="*60) + print(f"总节点数: {total}") + print(f"已处理: {processed} ({processed/total*100:.1f}%)") + print(f"成功: {successful}") + print(f"失败: {failed}") + print(f"成功率: {successful/processed*100:.1f}%" if processed > 0 else "成功率: N/A") + print(f"最后更新: {self.progress['last_update']}") + print("="*60 + "\n") + + def save_enriched_nodes(self, nodes: List[Dict[str, Any]]): + """保存丰富后的节点""" + output_file = self.output_dir / 'nodes_enriched.json' + + with open(output_file, 'w', encoding='utf-8') as f: + json.dump(nodes, f, indent=2, ensure_ascii=False) + + self.logger.info(f"丰富后的节点已保存到: {output_file}") + + # 统计 + enriched_count = sum( + 1 for node in nodes + if 'enriched_data' in node.get('properties', {}) + ) + + print(f"\n统计信息:") + print(f"总节点数: {len(nodes)}") + print(f"已丰富节点数: {enriched_count}") + print(f"丰富率: {enriched_count/len(nodes)*100:.1f}%") + + +def main(): + """主函数""" + import argparse + + parser = argparse.ArgumentParser(description='黑龙江非遗知识图谱构建') + parser.add_argument('--config', default='config/ich_config.yaml', help='配置文件路径') + parser.add_argument('--max-nodes', type=int, help='最大处理节点数(用于测试)') + parser.add_argument('--status', action='store_true', help='查看进度') + parser.add_argument('--resume', action='store_true', help='断点续传') + + args = parser.parse_args() + + # 创建构建器 + builder = ICKnowledgeGraphBuilder(args.config) + + if args.status: + # 显示进度 + builder._print_progress() + return + + # 加载节点数据 + print("加载节点数据...") + nodes = builder.load_nodes() + + if not nodes: + print("没有找到节点数据") + return + + # 批量处理 + print("开始批量处理节点...") + asyncio.run(builder.batch_enrich_nodes( + nodes, + max_nodes=args.max_nodes, + concurrent=builder.config['processing']['concurrent_requests'] + )) + + # 加载并保存丰富后的节点 + print("保存处理结果...") + nodes_file = builder.output_dir / 'nodes.json' + with open(nodes_file, 'r', encoding='utf-8') as f: + all_nodes = json.load(f) + + builder.save_enriched_nodes(all_nodes) + + print("\n处理完成!") + builder._print_progress() + + +if __name__ == '__main__': + main() diff --git a/dofile/kg_project/src/validation/__init__.py b/dofile/kg_project/src/validation/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/dofile/visulization/nodes.csv b/dofile/visulization/nodes.csv new file mode 100644 index 0000000..d69897b --- /dev/null +++ b/dofile/visulization/nodes.csv @@ -0,0 +1,5 @@ +id,name,type +1,张三,Person +2,李四,Person +3,Neo4j,Tool +4,知识图谱,Project diff --git a/dofile/visulization/rels.csv b/dofile/visulization/rels.csv new file mode 100644 index 0000000..a18cfc9 --- /dev/null +++ b/dofile/visulization/rels.csv @@ -0,0 +1,4 @@ +start_id,end_id,relation +1,2,朋友 +1,3,使用 +1,4,研究 diff --git a/dofile/visulization/roadmap.md b/dofile/visulization/roadmap.md new file mode 100644 index 0000000..03d2e9b --- /dev/null +++ b/dofile/visulization/roadmap.md @@ -0,0 +1,94 @@ +在 Docker 环境中使用 Neo4j 创建知识图谱,最关键的步骤是将你的 CSV 文件放入 Docker 容器能够访问的 import 目录下。 + +以下是详细的操作指南,分为准备、导入和验证三个阶段: + +第一步:准备工作(文件存放) +在 Docker 中,Neo4j 默认只能读取容器内 /var/lib/neo4j/import 目录下的文件。你需要将 nodes.csv 和 rels.csv 拷贝到这个位置。 + +确定容器名称:打开终端运行 docker ps,假设你的容器名为 neo4j。 + +拷贝文件:在宿主机(你的电脑)存放这两个文件的目录下运行: + +Bash +docker cp nodes.csv neo4j:/var/lib/neo4j/import/nodes.csv +docker cp rels.csv neo4j:/var/lib/neo4j/import/rels.csv +(注意:如果你在启动容器时挂载了 import 卷,直接把文件丢进对应的宿主机文件夹即可。) + +第二步:执行 Cypher 语句导入数据 +访问 Neo4j 浏览器(通常是 http://localhost:7474),登录后在顶部的输入框分步执行以下代码: + +1. 导入节点 (Nodes) +这段代码会读取 nodes.csv,并根据 type 字段动态创建节点。 + +Cypher +LOAD CSV WITH HEADERS FROM "file:///nodes.csv" AS row +// 使用 MERGE 避免重复创建,根据 id 匹配 +MERGE (n:Entity {id: row.id}) +ON CREATE SET n.name = row.name, n.type = row.type +// 为方便查询,根据 type 字段动态添加标签(可选) +WITH n, row +CALL apoc.create.addLabels(n, [row.type]) YIELD node +RETURN count(node); +提示:如果你的 Neo4j 没有安装 APOC 插件,可以直接简单处理: +LOAD CSV WITH HEADERS FROM "file:///nodes.csv" AS row MERGE (:Entity {id: row.id, name: row.name, type: row.type}); + +2. 创建索引(加速后续匹配) +在导入关系前,建议先为 id 建立索引: + +Cypher +CREATE INDEX FOR (n:Entity) ON (n.id); +3. 导入关系 (Relationships) +这段代码会根据 start_id 和 end_id 将节点连接起来。 + +Cypher +LOAD CSV WITH HEADERS FROM "file:///rels.csv" AS row +MATCH (from:Entity {id: row.start_id}) +MATCH (to:Entity {id: row.end_id}) +// 使用 APOC 创建动态关系类型,或者手动写死关系 +CALL apoc.create.relationship(from, row.relation, {}, to) YIELD rel +RETURN count(rel); +提示:如果没有 APOC,由于 CSV 中关系类型是中文或动态的,建议分类型导入,或先统一用一个关系名: +LOAD CSV WITH HEADERS FROM "file:///rels.csv" AS row MATCH (from:Entity {id: row.start_id}), (to:Entity {id: row.end_id}) MERGE (from)-[:RELATED {type: row.relation}]->(to); + +第三步:验证结果 +执行以下查询来查看你刚刚创建的图谱: + +Cypher +MATCH (n)-[r]->(m) RETURN n, r, m LIMIT 20; + +1. 彻底清库 +在 Neo4j 浏览器中运行以下命令,清空刚才导入的乱码数据: + +Cypher +MATCH (n) DETACH DELETE n; + +# 黑龙江非遗知识图谱导入 +1. 导入节点(修正版) +我们利用 apoc.convert.fromJsonMap 解析 JSON,然后直接用 Neo4j 原生的 += 操作符合并属性,这样比调用存储过程更快、更稳定。 + +Cypher +LOAD CSV WITH HEADERS FROM "file:///kg_merged_nodes.csv" AS row +// 1. 根据 ID 创建或匹配基础节点 +MERGE (n:Entity {id: row.id}) +SET n.name = row.label + +// 2. 解析 JSON 属性并批量赋值 +WITH n, row, apoc.convert.fromJsonMap(row.properties) AS props +SET n += props + +// 3. 动态添加业务标签 (ICH_Project, Performance_Form 等) +WITH n, row +CALL apoc.create.addLabels(n, [row.type]) YIELD node +RETURN count(node); + +2. 导入关系(修正版) +关系文件同理,我们也把 properties 里的描述和上下文解析出来。 + +Cypher +LOAD CSV WITH HEADERS FROM "file:///kg_merged_rels.csv" AS row +MATCH (s:Entity {id: row.source}) +MATCH (t:Entity {id: row.target}) + +// 动态创建关系并赋予 JSON 中的属性 +CALL apoc.create.relationship(s, row.type, apoc.convert.fromJsonMap(row.properties), t) YIELD rel +RETURN count(rel); diff --git a/officefile/.gitignore b/officefile/.gitignore new file mode 100644 index 0000000..baa2e62 --- /dev/null +++ b/officefile/.gitignore @@ -0,0 +1,8 @@ +# Overleaf sync is in latex/ subdirectory, ignore its git state +latex/.git/ + +# Obsidian workspace +.obsidian/ + +# Pandoc cache +.pandoc/ diff --git a/officefile/2026-雒伟群-刘华瑞-基于大语言模型的唐蕃古道文物知识图谱构建研究.pdf b/officefile/2026-雒伟群-刘华瑞-基于大语言模型的唐蕃古道文物知识图谱构建研究.pdf new file mode 100644 index 0000000..bb13eeb Binary files /dev/null and b/officefile/2026-雒伟群-刘华瑞-基于大语言模型的唐蕃古道文物知识图谱构建研究.pdf differ diff --git a/officefile/20260516-大模型驱动的黑龙江省非遗知识图谱构建-en.docx b/officefile/20260516-大模型驱动的黑龙江省非遗知识图谱构建-en.docx new file mode 100644 index 0000000..64f250a Binary files /dev/null and b/officefile/20260516-大模型驱动的黑龙江省非遗知识图谱构建-en.docx differ diff --git a/officefile/20260516-大模型驱动的黑龙江省非遗知识图谱构建.docx b/officefile/20260516-大模型驱动的黑龙江省非遗知识图谱构建.docx new file mode 100644 index 0000000..e61c259 Binary files /dev/null and b/officefile/20260516-大模型驱动的黑龙江省非遗知识图谱构建.docx differ diff --git a/officefile/2026_KG_ICH.bib b/officefile/2026_KG_ICH.bib new file mode 100644 index 0000000..53a48da --- /dev/null +++ b/officefile/2026_KG_ICH.bib @@ -0,0 +1,785 @@ +@article{2F3PRYB4, + title = {大语言模型强化学习驱动的文化遗迹叙事文本语义组织方法研究}, + author = {{张卫} and {高鑫} and {张予歌}}, + date = {2025-12-09}, + journaltitle = {图书情报工作}, + issn = {0252-3116}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CAPJ&dbname=CAPJLAST&filename=TSQB20251208001}, + urldate = {2026-03-13}, + abstract = {[目的 /意义]文化遗迹叙事文本是我国重要的文物数据资源,利用AIGC技术挖掘与组织叙事文本内的事件语义特征,对于文化遗产数字记忆保护与开发具有重要意义。[方法 /过程]在大语言模型强化学习驱动下提出基于事件本体的文化遗迹叙事文本语义组织方法。首先,综合专家智慧与文本特征进行事件概念揭示与知识单元重组,归纳遗迹叙事文本中核心元数据形成遗迹事件本体;基于DeepSeek深度思考范式进行事件抽取冷启动以获取伪标签,引入人智协同思维对各任务问答语料进行质量优化;重点针对事件抽取的任务框架设计高效自适应的奖励函数,并通过组内相对优势计算探索事件提及抽取、事件分类、事件论元识别等大模型强化学习的有效性。[结果 /结论 ]经过群体相对策略优化的DeepSeek-R1-Distill-Qwen-32B蒸馏模型分别在事件提及抽取(BLEU_1=80.21\%、ROUGE-1-F\textsubscript{1}=84.64\%)与事件论元识别(F\textsubscript{1}=78.90\%)任务取得最佳实践,强化后的QwQ-32B在事件分类任务上更具优势(ACC=86.09\%);经过实体消歧与共指消解能够进一步优化事件知识图谱的质量,从历史情节分析、人物知识谱系、时空耦合分析方面实现文化遗迹数字叙事服务。}, + langid = {chinese}, + pubstate = {advance online publication}, + keywords = {大语言模型,强化学习,事件抽取,数字人文,文化遗迹,语义组织}, + annotation = {original-container-title: Library and Information Service\\ +foundation: 国家自然科学基金青年项目“事件情感知识关联驱动下文化遗迹数字记忆重构模式研究”(项目编号:72404131); 江苏省社会科学基金青年项目“历代经典诗作隐喻的跨语言知识组织及应用研究”(项目编号:24TQC009)的研究成果之一;\\ +album: 信息科技;经济与管理科学;哲学与人文科学\\ +CLC: K87;TP391.1;TP18\\ +dbcode: CAPJ\\ +dbname: CAPJLAST\\ +filename: TSQB20251208001\\ +publicationTag: 北大核心, JST, AMI权威, CSSCI\\ +CIF: 4.608\\ +AIF: 3.514}, + file = {F:\Zotero\storage\S65UXEN3\大语言模型强化学习驱动的文化遗迹叙事文本语义组织方法研究_张卫.pdf} +} + +@article{2LWIKRWR, + title = {基于大语言模型的唐蕃古道文物知识图谱构建研究}, + author = {雒伟群, 刘华瑞}, + date = {2026}, + journaltitle = {计算机科学与探索}, + volume = {20}, + number = {3}, + pages = {801}, + doi = {10.3778/j.issn.1673-9418.2510052}, + url = {http://fcst.ceaj.org/CN/10.3778/j.issn.1673-9418.2510052}, + urldate = {2026-03-13}, + abstract = {针对文物领域多源异构数据整合困难,以及传统知识抽取方法依赖人工标注、自动化程度低等问题,提出一种基于大语言模型的文物知识图谱构建方法。融合领域专家知识与文物行业标准,构建涵盖文物核心概念及其关系的知识本体,系统定义其时空属性、文化属性和物理特征,以及文物间的关联关系,形成多维度结构化的知识框架。设计多阶段任务分解的联合抽取策略,利用大语言模型的语义理解与生成能力,将知识抽取拆解为“实体属性提取”与“三元组抽取”两个子任务,并通过提示工程引导模型逐步完成。同时,引入多维度校验机制,从逻辑一致性、领域规范性和事实准确性等三方面对生成的三元组进行验证,确保知识图谱的专业性与可靠性。实验表明,DeepSeek-R1模型在抽取任务中F1值达86.25\%,较BERT-BiLSTM-CRF模型提升3.13个百分点;消融实验也验证了各模块的有效性。该方法为文物数字化保护与跨学科研究提供了高效、自动化的技术支撑。}, + langid = {chinese}, + file = {F:\Zotero\storage\4PSREFEE\2026-雒伟群-刘华瑞-基于大语言模型的唐蕃古道文物知识图谱构建研究.pdf} +} + +@article{49Q5D2DH, + title = {基于Neo4j的中轴线艺术价值数字化知识图谱研究}, + author = {{刘彦超} and {刘键} and {席上琳} and {晁溪蕊} and {侯娜} and {朱文莲}}, + date = {2024}, + journaltitle = {包装工程}, + volume = {45}, + number = {8}, + issn = {1001-3563}, + doi = {10.19554/j.cnki.1001-3563.2024.08.023}, + url = {https://doi.org/10.19554/j.cnki.1001-3563.2024.08.023}, + urldate = {2026-03-13}, + abstract = {目的 以数字技术推动文化遗产价值阐释,以北京中轴线为例,提出了基于人工智能知识图谱的遗产价值挖掘与阐释方法。方法 构建了人工智能阐释遗产艺术价值的知识图谱七步法:1)多源异构的艺术资料整理与数字转化;2)基于Protégé系统的本体系统;3)借助本体与图数据库的映射厘清逻辑关系;4)结合NLP大数据技术进行文本挖掘与抽取;5)基于Neo4j构建数字化资源;6)基于Cypher语言查询与图算法提炼艺术价值研究;7)知识图谱的可视化呈现。结论 跳出了传统的中轴线宫廷艺术的范畴,提出了4个审美维度,并从4维度揭示了中轴线所承载的中华传统思想精髓。}, + langid = {chinese}, + keywords = {北京中轴线,本体,艺术价值,知识图谱,Neo4j}, + annotation = {original-container-title: Packaging Engineering\\ +foundation: 国家社科基金艺术学一般项目(23BH146); 北京市宣传系统高层次人才项目;\\ +album: 工程科技Ⅱ辑\\ +CLC: TB47\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2024\\ +filename: BZGC202408023\\ +publicationTag: 北大核心, CAS, JST, WJCI, AMI扩展\\ +CIF: 2.948\\ +AIF: 1.651}, + file = {F:\Zotero\storage\8MZ39UG7\基于Neo4j的中轴线艺术价值数字化知识图谱研究_刘彦超.pdf} +} + +@article{4GYKXZJ5, + title = {基于知识图谱和大模型的文化遗产展示和查询方法研究——以大运河文化遗产为例}, + author = {{蒋金亮} and {徐云翼} and {杨晗} and {刘志超}}, + date = {2024}, + journaltitle = {中国名城}, + volume = {38}, + number = {12}, + issn = {1674-4144}, + doi = {10.19924/j.cnki.1674-4144.2024.012.001}, + url = {https://doi.org/10.19924/j.cnki.1674-4144.2024.012.001}, + urldate = {2026-03-13}, + abstract = {随着数字技术的不断发展,知识图谱、生成式大模型等技术在历史文化遗产展示、保护领域得到广泛研究和应用。本研究提出基于知识图谱和大模型的历史文化遗产展示和查询方法,以大运河文化遗产作为研究对象,采集文化遗产的管理属性和文化属性信息,通过RDF三元组方法构建文化遗产知识图谱,用于可视化展示和遗产搜索;利用生成式大模型方法,构建大运河文化遗产的自然语言生成方案,最终抽取、生成大运河历史文化遗产的具体知识。本文提出的文化遗产展示和查询方法,拓展了文化遗产数字化保护传承利用的视角,同时为其他相关文化遗产的可视化呈现、数字化管理提供参考。}, + langid = {chinese}, + keywords = {大模型,大运河,文化遗产,知识图谱}, + annotation = {original-container-title: China Ancient City\\ +foundation: 国家自然科学基金面上项目“基于复杂系统模拟的跨区域国土空间韧性耦合机制与规划方法研究——以长三角地区为例”(编号:52178043);\\ +album: 哲学与人文科学;工程科技Ⅱ辑\\ +CLC: TU984.114;K878.4\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2025\\ +filename: ZGMI202412012\\ +CIF: 1.972\\ +AIF: 1.36}, + file = {F:\Zotero\storage\HT8773S9\基于知识图谱和大模型的文化遗产展示和查询方法研究——以大运河文化遗产为例_蒋金亮.pdf} +} + +@article{5VBJ427S, + title = {基于Neo4j的湘西地区旅游知识图谱构建研究}, + author = {{彭纪扬} and {郑昂}}, + date = {2025}, + journaltitle = {科技资讯}, + volume = {23}, + number = {8}, + issn = {1672-3791}, + doi = {10.16661/j.cnki.1672-3791.2409-5042-9147}, + url = {https://doi.org/10.16661/j.cnki.1672-3791.2409-5042-9147}, + urldate = {2026-03-13}, + abstract = {人工智能大模型和知识图谱技术的应用有助于系统化地组织和展示湘西地区的旅游资源信息。首先,通过网络数据采集,获取湘西地区的旅游资源文本数据,并利用生成式人工智能模型进行关系抽取和内容标注。然后,将处理后的数据导入Neo4j数据库,构建涵盖景点、饮食、交通等多维度信息的旅游知识图谱,并通过可视化工具予以直观展示。研究结果为区域旅游资源的数字化管理和精准营销提供了科学支持,并为相关领域的知识图谱构建提供了实践参考。}, + langid = {chinese}, + keywords = {关系抽取,旅游知识图谱,湘西地区,Neo4j数据库}, + annotation = {original-container-title: Science \& Technology Information\\ +foundation: 湖南省自然与文化遗产研究基地开放基金项目“基于自然语言处理的文旅资源知识图谱构建研究”(项目编号:ZRYC2306);\\ +album: 基础科学;信息科技;经济与管理科学\\ +CLC: TP391.1;F592.7\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2025\\ +filename: ZXLJ202508020\\ +publicationTag: JST\\ +CIF: 0.376\\ +AIF: 0.206}, + file = {F:\Zotero\storage\8WM48HSY\基于Neo4j的湘西地区旅游知识图谱构建研究_彭纪扬.pdf} +} + +@article{6JYY2GXD, + title = {论地理知识图谱}, + author = {{陆锋} and {余丽} and {仇培元}}, + date = {2017}, + journaltitle = {地球信息科学学报}, + volume = {19}, + number = {6}, + pages = {723--734}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2017&filename=DQXX201706002}, + urldate = {2026-03-22}, + abstract = {网络文本蕴含大量隐式地理空间信息,为地理知识获取与知识服务提供了巨大潜能。地理知识图谱是将传统地理信息服务拓展到地理知识服务的关键,也是网络文本蕴含地理信息采集与处理的终极目标。本文系统评述了开放地理语义网、开放地理实体及关系抽取、地理语义网对齐、知识图谱存储方法等地理知识图谱相关主题的研究进展,从网络文本蕴含地理空间信息量与质量评价、地理信息语义理解、空间语义计算模型和异构地理语义网对齐等方面剖析了目前亟需解决的关键科学问题。}, + langid = {chinese}, + keywords = {地理信息抽取,语义网,知识图谱,自然语言理解}, + annotation = {Fund: 国家自然科学重点基金项目(41631177); 中国科学院重点部署项目(ZDRW-ZS-2016-6-3);} +} + +@article{72HK557Y, + title = {知识图谱在数字人文中的应用研究}, + author = {{陈涛} and {刘炜} and {单蓉蓉} and {朱庆华}}, + date = {2019}, + journaltitle = {中国图书馆学报}, + shortjournal = {中国图书馆学报}, + volume = {45}, + number = {6}, + pages = {34--49}, + doi = {10.13530/j.cnki.jlis.190046}, + url = {https://link.cnki.net/doi/10.13530/j.cnki.jlis.190046}, + urldate = {2026-03-23}, + abstract = {知识图谱是利用计算机存储、管理和呈现概念及其相互关系的一种技术,一经提出便很快成为工业界和学术界的研究热点,但目前对知识图谱的认知还比较混乱。依据存储方式不同,知识图谱可分为基于RDF存储的语义知识图谱(关联数据)和基于图数据库的广义知识图谱。语义知识图谱(关联数据)侧重于知识的发布和链接,广义知识图谱则更侧重于知识的挖掘和计算,两者之间既有共同点,又有不同之处。本文从概念层面和技术层面详细分析了两者之间的异同,指出语义知识图谱(关联数据)才是谷歌知识图谱的延续和发展。随后,提出了将知识图谱应用于数字人文研究的系统框架,并在此基础上构建了中国历代人物传记资料库的关联数据平台(CBDBLD)。该平台借助知识图谱的理念展现了人物之间丰富的亲属及社会关系,形成了特有的社会关系网络,并可通过设置推理规则来实现人物之间隐性关系的挖掘与呈现。广义知识图谱研究中丰富的图运算和关联数据的结合将会成为数字人文领域研究的下一个热点,从而开启数字人文研究的新时代。图10。表2。参考文献25。}, + langid = {chinese}, + keywords = {/unread,关联数据,数字人文,知识图谱,知识推理,中国历代人物传记资料库}, + annotation = {Fund: 国家社会科学基金项目“数字人文中图像文本资源的语义化建设与开放图谱构建研究”(编号:19BTQ024)的研究成果之一\textasciitilde\textasciitilde ;}, + file = {F:\Zotero\storage\XDFFDI6E\2019-陈涛-刘炜-单蓉蓉-朱庆华-知识图谱在数字人文中的应用研究.pdf} +} + +@article{7ZYWS5KV, + title = {中共中央办公厅 国务院办公厅印发《关于进一步加强非物质文化遗产保护工作的意见》}, + journaltitle = {中华人民共和国国务院公报}, + shortjournal = {中华人民共和国国务院公报}, + number = {24}, + issn = {1004-3438}, + url = {https://www.gov.cn/gongbao/content/2021/content_5633447.htm}, + urldate = {2026-03-23}, + abstract = {近日,中共中央办公厅、国务院办公厅印发了《关于进一步加强非物质文化遗产保护工作的意见》,并发出通知,要求各地区各部门结合实际认真贯彻落实。明确提出,加大非物质文化遗产传播普及力度, 将非物质文化遗产内容贯穿国民教育始终,构建非物质文化 遗产课程体系和教材体系,鼓励非物质文化遗产进校园}, + langid = {chinese}, + keywords = {非物质文化遗产中央办公厅意见国务院办公厅}, + file = {F:\Zotero\storage\ERF9BFWT\content_5633447.html} +} + +@article{8LEDXI3J, + title = {ChatKG:一种基于大语言模型和提示工程的非遗知识图谱构建框架——以中国非遗陶瓷制作工艺为例}, + author = {{周正达} and {王昊} and {汪琳} and {李晓敏} and {周抒} and {姚天辰}}, + date = {2025-02-24}, + journaltitle = {图书馆杂志}, + issn = {1000-4254}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CAPJ&dbname=CAPJLAST&filename=TNGZ20250221003}, + urldate = {2026-03-13}, + abstract = {本文旨在解决当前非遗知识图谱构建中存在的人工成本高、准确率不足的问题,提出利用大规模预训练模型ChatGPT开展非遗知识图谱构建的新思路。具体提出一种基于大语言模型和提示工程的非遗知识图谱构建框架ChatKG(Chat Knowledge Graph),选择复用CIDOC CRM本体模型,结合人工梳理与ChatGPT辅助,实现了非遗本体构建,并提出一种基于思维链(CoT)的提示优化方法实现准确的非遗知识抽取,为缺少可直接复用本体概念模型以及高质量标注数据的非遗领域,提供了一种快速、高效、低成本构建领域本体概念模型和进行知识抽取的方法。本文以中国非遗陶瓷制作工艺为例,引导大语言模型成功识别出了419个工艺实体及763条实体间关系,最终构建了非遗陶瓷工艺知识图谱并进行了应用场景探索,验证了本文方法的有效性。}, + langid = {chinese}, + pubstate = {advance online publication}, + keywords = {本体构建,大语言模型,提示工程,知识抽取,知识图谱}, + annotation = {original-container-title: Library Journal\\ +foundation: 国家自然科学基金面上项目“关联数据驱动下我国非遗文本的语义解析与人文计算研究”(项目编号:72074108); 南京大学“中央高校基本科研业务费专项资金资助”项目“面向人文计算的方志文本的语义分析和知识图谱研究”(项目编号:010814370113)的研究成果之一; 江苏青年社科英才和南京大学仲英青年学者等人才培养计划的支持;\\ +album: 哲学与人文科学;信息科技\\ +CLC: J527;G353.1\\ +dbcode: CAPJ\\ +dbname: CAPJLAST\\ +filename: TNGZ20250221003\\ +publicationTag: 北大核心, AMI核心, CSSCI\\ +CIF: 3.589\\ +AIF: 2.878}, + file = {F:\Zotero\storage\8SYAQFEN\ChatKG:一种基于大语言模型和提示工程的非遗知识图谱构建框架——以中国非遗陶瓷制作工艺为例_周正达.pdf} +} + +@article{B79P45VU, + title = {唐诗知识图谱的构建及其智能知识服务设计}, + author = {{周莉娜} and {洪亮} and {高子阳}}, + date = {2019}, + journaltitle = {图书情报工作}, + volume = {63}, + number = {2}, + pages = {24--33}, + doi = {10.13266/j.issn.0252-3116.2019.02.003}, + url = {https://doi.org/10.13266/j.issn.0252-3116.2019.02.003}, + urldate = {2026-03-22}, + abstract = {[目的/意义]立足于当前大数据环境下的唐诗知识服务需求,以大规模唐诗数据为基础构建唐诗知识图谱并提供智能知识服务,推动人工智能环境下唐诗知识管理和知识服务方式的创新。[方法/过程]本文在对领域知识服务需求调研的基础上,设计领域知识服务驱动的唐诗本体模型,然后利用从Web上爬取的多源异构数据,采用知识抽取、知识融合、知识推理等技术自动构建唐诗知识图谱,统一表示和组织唐诗领域数据,实现对大规模唐诗数据的语义化处理。[结果/结论]本文设计基于唐诗知识图谱的智能知识服务平台KnowPoetry,提供唐诗领域的知识探索、时空轨迹、语义查询等智能化知识服务,推动人工智能环境下唐诗数字人文研究方法的创新转型。}, + langid = {chinese}, + keywords = {数字人文,唐诗知识图谱,知识建模,智能知识服务}, + annotation = {Fund: 国家重点研发计划“科学大数据管理系统”子课题“图数据管理关键技术及系统”(项目编号:2016YFB1000603); 教育部人文社会科学重点研究基地重大项目“大数据资源的语义表示与组织研究——面向文化遗产领域”(项目编号:16JJD870002)研究成果之一;}, + file = {F:\Zotero\storage\7DMY9Q5C\2019-周莉娜-洪亮-高子阳-唐诗知识图谱的构建及其智能知识服务设计.pdf} +} + +@article{D93KE6Y2, + title = {A {{Map}} for {{Big Data Research}} in {{Digital Humanities}}}, + author = {Kaplan, Frédéric}, + date = {2015-05-06}, + journaltitle = {Frontiers in DIGITAL Humanities}, + shortjournal = {Front. DIGIT. Humanit.}, + volume = {2}, + publisher = {Frontiers}, + issn = {2297-2668}, + doi = {10.3389/fdigh.2015.00001}, + url = {https://www.frontiersin.org/journals/digital-humanities/articles/10.3389/fdigh.2015.00001/full}, + urldate = {2026-03-23}, + abstract = {AbstractThis article is an attempt to represent Big Data research in Digital Humanities as a structured research field. A division in three concentric areas of study is presented. Challenges in the first circle - focusing on the processing and interpretations of large cultural datasets - can be organized linearly following the data processing pipeline. Challenges in the second circle - concerning digital culture at large – can be structured around the different relations linking massive datasets, large communities, collective discourses, global actors and the software medium. Challenges in the third circle - dealing with the experience of big data - can be described within a continuous space of possible interfaces organized around three poles: immersion, abstraction and language. By identifying research challenges in all these domains, the article illustrates how this initial cartography could be helpful to organize the exploration of the various dimensions of Big Data Digital Humanities research. Introduction: Big Data Digital Humanities vs. Small Data Digital HumanitiesDefining the nature and the boundaries of Digital Humanities is a long-discussed and unsolved issue (Terras et al 2013), not only because there is no consensus on this question but also because Digital Humanities are currently undergoing a profound transformation that calls for a reconsideration of its fundamental concepts (Gold 2012). For years, Digital humanities have been loosely regrouping computational approaches of humanities research problems and critical reflections of the effects of digital technologies on culture and knowledge (Schreibman et al 2004). Ten years ago, they emerged as a new label, rebranding and enlarging the idea of “humanities computing” (Svensson 2009). Around this new name and under a “big tent”, a progressively larger community of practice thrived (Terras 2011). Each work at the intersection of Computer Science and the Humanities could potentially be part of this welcoming trend. Researchers gathered in national and international meetings, exchanged their views on blogs and mailing lists. If not a well bounded field, Digital Humanities were surely a lively conversation.The welcoming Digital Humanities label opened doors, connected separated academic silos, built bridges between information sciences and the various disciplines loosely forming what is called the humanities. However openness was always associated with a need for introspection, self-reflexive writings, tentative boundaries definitions, the “What are digital humanities” articles and monographs became a genre of its own structured around several narratives of exclusion and inclusion (Rockwell, 2011). Digital Humanities as a research domain define themselves dynamically in the negotiation of these tensions as discussed by several Digital Humanities scholars (Unsworth 2002, Svensson 2009, Rockwell 2011). Table 1 gives a non-exhaustive list of these structuring tensions.The starting point of this article is a relatively new particular structuring tension, opposing Big Data Digital Humanists with Small Data Digital Humanists. Research in Big Data Digital Humanities focuses on large or dense cultural datasets, that call for new processing and interpretation methods. The term Big Data itself has disputed origins (Diebold, 2012, Lohr 2013). The Oxford English Dictionary defines it as “data of a very large size, typically to the extent that its manipulation and management present significant logistical challenges.” In that sense, Big Data are “big” when “manual” analysis becomes cumbersome and new study and interpretation methods must be invented. However, massiveness of Big Data is not tightly linked to a certain number of Terabytes. Boyd and Crowford (2011) note that “Big Data is not notable because of its size, but because of its relationality to other data”. Big Data is “fundamentally networked” and challenges in processing it are linked with its interconnected nature. In comparison, the Small Data Digital Humanities regroup more focused works that do not use massive data processing methods and explore other interdisciplinary dimensions linking computer science and humanities research. In comparison with Big Data, Small Data is small in the sense that it is not only smaller-scale but also well-bounded.This article intends to draw a map for Big Data Digital Humanities showing how it can be organized as a structured field. The ambition of this map is to show that Big Data research in Digital Humanities can be characterized by common methodologies and objects of studies, therefore transcending some of the tensions that have structured Digital Humanities so far. As it focuses only on research that deals with these “large body of information” (Katz 2005), this maps does not cover the Digital Humanities domain as whole. Nevertheless, given the growing importance of massive and networked cultural datasets, it is likely that Big Data Digital Humanities become a significant part of the whole Digital Humanities field. In this context, this map may help institutionalize research and education programs with clearer focuses and objectives. This article presents Big Data research in Digital Humanities as three concentric circles (Fig 1.) The first circle corresponds to research focusing on processing and interpretation big and networked cultural data sets, the first object of study of this field. Most of the methods needed to study these datasets need still to be invented, as they are currently not mastered neither by humanists or computer scientists. However, it is important to consider that data processing and interpretation occur in a larger context of the new digital culture characterized by collective discourses, large community, ubiquitous software and global IT actors. Understanding the relation between these entities could be considered the second object of study for Big Data Digital Humanities. Eventually, the human experience of such datasets through various kinds of interfaces corresponds to a third family of challenges, differing in scope and methodology from the other two. Therefore, these three areas of studies could be represented as three concentric circles, illustrating three levels of contextualization and embodiment of cultural data. In the next sections we will briefly discuss each of the circles in more details.Big Cultural datasetsMassive cultural digital objects include large-scale corpus like the millions of books scanned by Google and the ones produced by numerous other digitization initiatives (Jacquesson 2010), the millions of photos and micro-message shared on social network services (Tushoo et al 2010), giant geographical information systems like Google Earth (Butler 2006) or the ever expanding networks of academic papers citing one another (Shibata et al 2008). These interconnected objects - either digitally born or reconstructed through digitization pipelines - are too big to be read or watched. The traditional 1:1 ratio of a single scholar confronted with one document cannot cope with such abundance. Moreover, their boundaries are sometimes fuzzy, their content partially unknown and, likely to be in continuous expansion. These characteristics make them profoundly different from corpora traditionally studied by humanities researchers, despite surface resemblances. The confrontation with these “massive” objects calls for fundamental questions. What can really be extracted from these huge datasets and what interpretations can be drawn based on these extractions? Will we learn more by analyzing 10 millions books that we cannot read individually or by reading five carefully (Moretti 2005)? What is the role of algorithms for mining, shaping and representing these large digital objects?Some of these challenges can be structured following the specific parts of data processing: digitization, transcription, pattern recognition, simulation and inferences, preservation and curation as show in figure 2 and in the table below. Each step in the data processing pipeline can be associated with questions that are both technical and epistemological. Consider the processing pipeline of mass book digitization projects. Physical books must be transformed into images (digitization step) that are then transformed into texts (transcription step), on which various pattern can be detected (pattern recognition step like text mining or n-gram approaches) or inferred (simulation step) while being preserved and curated for future research (preservation step). This way of presenting the research challenge insists on the fact that data are never given, but taken and transformed (Gitelman 2013). The technical complexity of pipelines involved clearly demonstrate that, at each step of the data processing, choices are made and biases apply. Understanding these technical choices is crucial to develop new interpretive theories.Digital CultureWe discussed the relationship between data processing pipelines and large cultural datasets. However, data processing and interpretation happen in a larger context, which we may call Digital Culture. The study of this large context can be considered to be the second object of study for Digital Humanities research. One way to structure this domain is to replace the relation between software and data (the focus of the first circle) in a network of relations between new entities including large-scale communities (MOOCs classrooms, Wikipedia contributors, etc.), collective discourses (Blogs, data journalism, wiki-style collaborative writing), ubiquitous software medium (auto-completion algorithm, search engine) and global actors (Google, Facebook, GLAM, Universities). Consider the millions of photos shared every hour on Facebook (Huang et al 2013). In this case, large-scale communities produce both the massive digital objects and the collective discourses about massive digital objects. They do so through the mediation of algorithms produced by one giant IT company of the web. Retroactively, collective discourses about the photos have a shaping role on the emergence and structuration of these communities. In addition, as collective discourses reach rapidly a critical mass (e.g. millions of messages or status update) they tend to become themselves massive digital objects, to be archived and studied through specific text and data mining approaches. Understanding photo sharing implies understanding the complexity of this network of interactions. More generally, research about digital culture can be segmented in subdomains corresponding to groups of relations between some of the entities we have been discussing. This structuration summarized in Table 3 and Figure 3, identifies five domains: the processing domain, the discursive domain, the social shaping domain, the algorithmic mediation domain and the control domain. This grouping articulates differently the relations of Big Data Digital Humanities with trad}, + langid = {english}, + keywords = {big data,Cartography.,Challenges,Digital Humanities,Mapping}, + file = {F:\Zotero\storage\FFVP5DKK\2015-kaplan-frédéric-a-map-for-big-data-research-in.pdf} +} + +@article{EC26X7PS, + title = {基于大模型技术的档案文化遗产自动问答平台构建研究}, + author = {{李根}}, + date = {2024}, + journaltitle = {山西档案}, + number = {9}, + issn = {1005-9652}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2024&filename=SXDA202409029}, + urldate = {2026-03-13}, + abstract = {大语言模型的出现为档案工作的数智化转型提供了新契机。在梳理了大模型技术特点的基础上,分析其在图情档领域的应用现状,并剖析了档案文化遗产传承弘扬面临的困境,进而提出档案文化遗产自动问答平台的整体构建框架,并围绕领域知识库构建、大模型与知识库融合的问答、档案知识可视化、问答质量评估等关键技术展开了深入探讨。旨在为档案知识服务创新提供理论视角,也为智能问答系统的实践应用提供借鉴,助力新时代档案事业的高质量发展。}, + langid = {chinese}, + keywords = {大语言模型,档案文化遗产,数字人文,知识库,智能问答}, + annotation = {original-container-title: Shanxi Archives\\ +foundation: 2023年广东省教育厅青年创新人才项目(自然科学)“基于大数据图像处理的工业瑕疵检测系统”(项目编号:2023KQNCX143);\\ +album: 信息科技\\ +CLC: TP18;TP391.1;G270.7\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2024\\ +filename: SXDA202409029\\ +publicationTag: 北大核心, AMI扩展\\ +CIF: 3.591\\ +AIF: 3.339}, + file = {F:\Zotero\storage\A4YJ4GJF\基于大模型技术的档案文化遗产自动问答平台构建研究_李根.pdf} +} + +@article{FMHA6MVH, + title = {基于大模型的非遗知识图谱与智慧问答系统构建研究}, + author = {{徐怀钰} and {赵俊伟} and {彭潇} and {黄梅荣}}, + date = {2025}, + journaltitle = {华东科技}, + number = {6}, + issn = {1006-8465}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2025&filename=HDKJ202506031}, + urldate = {2026-03-13}, + abstract = {{$<$}正{$>$}自党的二十大报告明确提出实施国家文化数字化战略以来,文化数字化已逐步成为建设社会主义文化强国、实现文化产业高质量发展的战略选择。非物质文化遗产(以下简称非遗)作为人类文明的瑰宝,在保护与传承方面仍面临诸多现实挑战。与此同时,数字化技术为非遗的保护与传承提供了创新手段,如通过智能问答技术来高效解答公众对非遗项目的疑问,有效促进中华优秀传统文化的传播。基于此,本文探讨了基于大模型的非遗知识图谱与智慧问答系统的构建路径,以期为非遗的保护与传承提供技术支持。}, + langid = {chinese}, + annotation = {original-container-title: East China Science \& Technology\\ +foundation: 湖南省大学生创新训练项目“内容生成技术赋能非物质文化遗产的知识图谱建设研究——以江永女书为例”(项目编号:S202410554092); 湖南省普通本科高校教学改革研究项目“ChatGPT赋能元宇宙教学资源数字化建设”(项目编号:202401001061);\\ +album: 基础科学;哲学与人文科学;信息科技\\ +CLC: G122;TP391.1;TP18\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2025\\ +filename: HDKJ202506031\\ +CIF: 0.324\\ +AIF: 0.157}, + file = {F:\Zotero\storage\WJJTSAGE\基于大模型的非遗知识图谱与智慧问答系统构建研究_徐怀钰.pdf} +} + +@article{FYB5RXLC, + title = {文化遗产领域知识图谱发展趋势与前沿进展研究}, + author = {{敖若瑶}}, + date = {2025}, + journaltitle = {科技与创新}, + number = {17}, + issn = {2095-6835}, + doi = {10.15913/j.cnki.kjycx.2025.17.004}, + url = {https://doi.org/10.15913/j.cnki.kjycx.2025.17.004}, + urldate = {2026-03-13}, + abstract = {知识图谱作为学界较为成熟的一项技术,为分散且多维度的文化遗产资源提供了结构化与语义化整合的解决方案。运用文献计量法和文献分析法,借助CiteSpace软件对从CNKI期刊库中获取的共计92篇相关文献进行科学知识图谱可视化分析,旨在揭示文化遗产领域知识图谱研究的演进脉络、热点主题以及前沿趋势。研究总结归纳出文化领域知识图谱研究的演进可划分为3个阶段,即萌芽期(2015—2019年)、发展期(2020—2023年)、探索期(2024—2025年),发现大模型与智能体技术显著推动了文化遗产知识图谱在知识生成、多模态交互以及动态服务方面的革新。展望未来,可以重点关注多模态知识图谱的深度整合与动态表达,以及智能体驱动的主动化知识服务系统等方向,以深化文化遗产的智慧化保护与传播。}, + langid = {chinese}, + keywords = {大模型,数字人文,文化遗产,知识图谱}, + annotation = {original-container-title: Science and Technology \& Innovation\\ +album: 工程科技Ⅱ辑;哲学与人文科学;信息科技\\ +CLC: G353.1;G122\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2025\\ +filename: KJYX202517004\\ +publicationTag: JST\\ +CIF: 0.443\\ +AIF: 0.24}, + file = {F:\Zotero\storage\YNNL99H5\文化遗产领域知识图谱发展趋势与前沿进展研究_敖若瑶.pdf} +} + +@article{GCLXKNED, + title = {融合学习扩展的非遗陶瓷工艺领域术语库构建及应用}, + author = {{汪琳} and {王昊} and {李晓敏} and {邓三鸿}}, + date = {2024}, + journaltitle = {图书馆论坛}, + shortjournal = {图书馆论坛}, + volume = {44}, + number = {2}, + pages = {66--78}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2024&filename=TSGL202402008}, + urldate = {2026-03-23}, + abstract = {文章通过学习扩展的机器学习和深度学习,提出针对非物质文化遗产项目语料的术语抽取及新词发现方法,形成领域术语库并探讨在数字人文领域的应用。首先使用自然语言处理方法对非遗陶瓷语料进行预处理,结合领域术语词表对语料进行标注;然后针对Random-CRFs模型,研究词表特征(DICT)、词性特征(POS)、部首特征(Radical)、拼音特征(Pinyin)对术语抽取效果的影响,再对比Random-CRFs、Random-BiLSTM、Random-BiLSTM-CRFs、BERT-BiLSTMCRFs等4个模型对术语抽取效果的影响;最后使用训练完成的模型对测试集语料进行新词识别,对抽取出的候选词进行人工判断,构建包含1,173个术语的非物质文化遗产陶瓷工艺领域术语库,将其应用于非遗项目画像、非遗陶瓷工艺知识图谱和非遗陶瓷工艺术语检索。}, + langid = {chinese}, + keywords = {非物质文化遗产,领域术语,数字人文,新词发现}, + annotation = {Fund: 国家自然科学基金项目“关联数据驱动下我国非遗文本的语义解析与人文计算研究”(项目编号:72074108); 中央高校基本科研业务费项目“面向人文计算的方志文本的语义分析和知识图谱研究”(项目编号:010814370113)研究成果;}, + file = {F:\Zotero\storage\UCTVN2WA\2024-汪琳-王昊-李晓敏-邓三鸿-融合学习扩展的非遗陶瓷工艺领域术语库构建及应用.pdf} +} + +@article{IBVAJNMD, + title = {价值共创视域下中国传统戏曲知识图谱模式层构建及应用研究}, + author = {{王左戎} and {邓三鸿} and {胡畔} and {翟姗姗}}, + date = {2025}, + journaltitle = {情报科学}, + shortjournal = {情报科学}, + volume = {43}, + number = {3}, + pages = {165--175}, + doi = {10.13833/j.issn.1007-7634.2025.03.020}, + url = {https://doi.org/10.13833/j.issn.1007-7634.2025.03.020}, + urldate = {2026-03-23}, + abstract = {【目的/意义】作为中华民族文化的重要组成部分,利用现代语义组织实现对于海量异构中国传统戏曲主题数字资源的组织,有利于实现中国传统戏曲文化价值的挖掘和释放。【方法/过程】本研究在梳理价值共创理论在文化遗产领域应用的基础上,围绕16项具体中国传统戏曲全面搜集了包括政府部门、企业、高校在内多类型主体提供的公开数字资源,按照七步法知识建模流程提出中国传统戏曲知识图谱模式层构建方案,同时利用Protégé和Neo4j工具进行了图谱模式层的具体应用展示。【结果/结论】本研究以充分释放中国传统戏曲文化价值为导向,在人工梳理了120篇主题文献和40项主题网站内容的基础上提出了囊括11项实体类、24项实体子类及7种主要关系类型的模式层设计方案和应用策略,以期从资源组织的角度为我国传统戏曲文化的创新性发展提供支持。【创新/局限】本研究在价值共创视域下面向中国传统戏曲信息组织场景提出领域知识图谱模式层构建方案,但在具体实践的细节层面仍有可以完善的空间,后续将会进一步考虑引入大语言模型等方式提升图谱构建的效率。}, + langid = {chinese}, + keywords = {本体,传统戏曲,价值共创,信息组织,知识图谱}, + annotation = {Fund: 国家社科基金一般项目“数字人文视域下非遗知识图谱自动构建与长期演进研究”(20BTQ071);}, + file = {F:\Zotero\storage\V6ZYTA73\2025-王左戎-邓三鸿-胡畔-翟姗姗-价值共创视域下中国传统戏曲知识图谱模式层构建及应用研究.pdf} +} + +@article{JH5C6HKD, + title = {“博古问津”:知识图谱增强的文化遗产领域多模态大模型}, + shorttitle = {“博古问津”}, + author = {{赵万青} and {徐朝阳} and {谢智伟} and {张少博} and {张晓丹} and {彭进业}}, + date = {2025}, + journaltitle = {西北大学学报(自然科学版)}, + volume = {55}, + number = {6}, + issn = {1000-274X}, + doi = {10.16152/j.cnki.xdxbzr.2025-06-006}, + url = {https://doi.org/10.16152/j.cnki.xdxbzr.2025-06-006}, + urldate = {2026-03-13}, + abstract = {近年来,大语言模型(LLMs)和多模态大模型(MLMs)在自然语言处理和多模态内容理解方面取得了显著成就,然而,这些通用模型在处理文化遗产相关任务时存在明显缺陷,如对领域专业术语的理解存在偏差、缺乏文化历史背景导致回答不够深入以及知识幻觉等问题,使得输出结果难以满足实际需求。针对这些挑战,首次提出了面向文化遗产领域的多模态大模型——“博古问津”。首先,设计半自动化策略构建大规模的多模态文化遗产数据集并形成多模态知识图谱;然后,利用构建的数据集对通用大模型进行图文对齐和指令微调两阶段训练,以适应文化遗产领域的特定需求;此外,还引入了知识图谱作为辅助知识库,通过图文检索和关系检索策略,有效提升了模型在文化遗产领域问答任务上的可信度和可解释性。实验结果表明,“博古问津”在文物图像描述、属性问题解答及关系问题理解等多个方面表现优异,相较于通用多模态大模型,对复杂文化内容的理解和回答能力提升效果显著,分别在文物图像描述、文物属性问题和文物关系问题3个不同任务的综合分值上高出次优模型21.4\%、53\%和20.6\%。}, + langid = {chinese}, + keywords = {多模态大模型,模型微调,视觉问答,文化遗产,知识图谱,知识增强}, + annotation = {original-container-title: Journal of Northwest University(Natural Science Edition)\\ +foundation: 国家重点研发计划(2024YFF0907600); 国家自然科学基金(62273275); 陕西省自然科学基础研究计划青年项目(2025JC-YBQN-847);\\ +album: 基础科学;哲学与人文科学;信息科技\\ +CLC: K85;TP18;TP391.1\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2026\\ +filename: XBDZ202506006\\ +publicationTag: 北大核心, CAS, JST, CSCD, WJCI, 卓越期刊\\ +CIF: 2.156\\ +AIF: 1.458}, + file = {F:\Zotero\storage\2USFSBL6\“博古问津”_知识图谱增强的文化遗产领域多模态大模型_赵万青.pdf} +} + +@article{LDUKQAWH, + title = {数字人文领域的知识图谱:研究进展与未来趋势}, + author = {{朱丽雅} and {张珺} and {洪亮} and {罗绍辉} and {兰度}}, + date = {2022}, + journaltitle = {知识管理论坛}, + shortjournal = {知识管理论坛}, + volume = {7}, + number = {1}, + pages = {87--100}, + doi = {10.13266/j.issn.2095-5472.2022.008}, + url = {https://doi.org/10.13266/j.issn.2095-5472.2022.008}, + urldate = {2026-03-23}, + abstract = {[目的/意义]对数字人文领域的知识图谱研究进行系统性回顾,旨在提供未来可能的研究方向和开放的研究主题。[方法/过程]以国内外会议、期刊发表的相关文献为研究对象,采用综合归纳法,系统梳理数字人文领域知识图谱的理论与实践发展。阐述数字人文领域知识图谱的相关概念,并根据当前的研究热点,从数据资源建设、关键构建技术、平台智能应用3个方面揭示其研究动向,并对未来研究趋势进行展望。[结果/结论]总结数字人文知识图谱研究的未来发展趋势,即未来将呈现出多源数据集成、多模态知识融合、多学科交叉应用的发展趋势。}, + langid = {chinese}, + keywords = {数据资源建设,数字人文,语义挖掘,知识图谱,智慧数据}, + annotation = {Fund: 2020年国家档案局科技项目“基于时空数据的智慧城市档案知识图谱构建及应用服务体系研究”(项目编号:2020-X-053); 湖北省重点研发计划项目“文旅科技大数据关键技术研发与应用示范”(项目编号:2020BAB117); 南宁市科学研究与技术开发计划项目科技重大专项“基于GIS和BIM技术的城建大数据平台研究”(项目编号:20193010)研究成果之一;}, + file = {F:\Zotero\storage\3UB3793C\2022-朱丽雅-张珺-洪亮-罗绍辉-兰度-数字人文领域的知识图谱:研究进展与未来趋势.pdf} +} + +@article{LJ5QBY4I, + title = {“大模型+知识图谱”双轮驱动的公共数字文化资源管理新范式}, + author = {{杨萌} and {张云中} and {赵程程}}, + date = {2025}, + journaltitle = {情报科学}, + volume = {43}, + number = {9}, + issn = {1007-7634}, + doi = {10.13833/j.issn.1007-7634.2025.09.009}, + url = {https://doi.org/10.13833/j.issn.1007-7634.2025.09.009}, + urldate = {2026-03-13}, + abstract = {【目的/意义】“大模型+知识图谱”的双轮驱动范式,可以创造更为强大的公共数字文化资源研究工具和应用,有助于提升公共数字文化资源的保护、研究、利用和传播水平。【方法/过程】本文在对比知识图谱和大模型作为知识库的技术特点及优劣势的基础上,提出了公共数字文化资源管理“大模型+知识图谱”双轮驱动模型,分析了该模型建构的需求与动机、基础与条件,并对资源层、数据层、技术层、应用层和行业层的运行机制进行了阐释。【结果/结论】研究发现,建构在“大模型+知识图谱”双轮驱动模型基础上的公共数字文化行业知识中台,可以更好实现人机协同,有效支撑公共数字文化领域的文化遗产保护修复、文化内涵挖掘与数字展陈、文化资产智慧化管理、文化教育与文化传播等方面的知识服务创新。【创新/局限】将大模型和知识图谱二者相结合,构建公共数字文化资源知识平台框架。}, + langid = {chinese}, + keywords = {大模型,公共数字文化资源,知识服务,知识图谱,知识中台}, + annotation = {original-container-title: Information Science\\ +foundation: 国家社会科学基金项目“智慧数据驱动的公共数字文化资源知识图谱构建与应用研究”(21BTQ105); 上海市教育发展基金会和上海市教育委员会“曙光计划”资助;\\ +album: 信息科技\\ +CLC: G250.7\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2026\\ +filename: QBKX202509009\\ +publicationTag: 北大核心, JST, WJCI, AMI核心, CSSCI\\ +CIF: 5.027\\ +AIF: 3.1}, + file = {F:\Zotero\storage\CPG5RZ24\“大模型+知识图谱”双轮驱动的公共数字文化资源管理新范式_杨萌.pdf} +} + +@article{MLPW6LLL, + title = {基于关联数据的数字人文视觉资源知识组织研究}, + author = {{曾子明} and {周知} and {蒋琳}}, + date = {2018}, + journaltitle = {情报资料工作}, + shortjournal = {情报资料工作}, + number = {6}, + pages = {6--12}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2018&filename=QBZL201806003}, + urldate = {2026-03-23}, + abstract = {文章提出一种基于关联数据的数字人文视觉资源知识组织模型,在分析用户需求的基础上,提出从数据采集到智慧服务的完整流程,构建基于关联数据的知识组织模型,并以敦煌文化遗产为具体案例进行说明,为相关问题的解决提供参考。}, + langid = {chinese}, + keywords = {关联数据,视觉资源,数字人文,知识组织}, + annotation = {Fund: 国家自然科学基金项目“云环境下智慧图书馆移动视觉搜索模型与实现研究”(编号:71673203)的研究成果之一;}, + file = {F:\Zotero\storage\D7B32DXK\2018-曾子明-周知-蒋琳-基于关联数据的数字人文视觉资源知识组织研究.pdf} +} + +@article{N9VGUZD6, + title = {大模型与古籍档案文化遗产数字化:价值、挑战与应对}, + author = {{刘文俏}}, + date = {2024}, + journaltitle = {山西档案}, + number = {1}, + issn = {1005-9652}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2024&filename=SXDA202401023}, + urldate = {2026-03-13}, + abstract = {古籍档案是中华文明重要的物质文化载体,其价值不仅在于承载了丰富的历史文化知识,而且在于彰显了民族的文化自信和价值观念。为了实现古籍档案的长期有效保护和广泛传播利用,将大模型技术应用于古籍档案数字化保护与利用中,发挥其巨大的潜力。旨在深入探讨大模型技术赋能古籍档案文化遗产数字化保护与利用的路径设计,站在理论与实践相结合的高度,充分挖掘大模型技术在传统档案文化遗产保护与传播中的变革性作用,为推动古籍档案资源保护和文化创新利用提供有力的技术支撑。}, + langid = {chinese}, + keywords = {大语言模型,古籍档案,数字化保护,文化创新利用,智慧服务}, + annotation = {original-container-title: Shanxi Archives\\ +album: 信息科技\\ +CLC: G255.1;G270.7\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2024\\ +filename: SXDA202401023\\ +publicationTag: 北大核心, AMI扩展\\ +CIF: 3.591\\ +AIF: 3.339}, + file = {F:\Zotero\storage\4DH3ZTUP\大模型与古籍档案文化遗产数字化:价值、挑战与应对_刘文俏.pdf} +} + +@article{NLCHJIZL, + title = {AIGC视角下非物质文化遗产知识图谱的构建研究}, + author = {{陈昱成} and {黎洋} and {刘江峰} and {杨帆}}, + date = {2024}, + journaltitle = {科技情报研究}, + volume = {6}, + number = {2}, + issn = {2096-7144}, + doi = {10.19809/j.cnki.kjqbyj.2024.02.010}, + url = {https://doi.org/10.19809/j.cnki.kjqbyj.2024.02.010}, + urldate = {2026-03-13}, + abstract = {[目的/意义]非遗是人类文明的重要组成部分,对于保护和弘扬民族精神,增强民族认同感和凝聚力具有重要意义。[方法/过程]文章探讨如何利用AIGC的优势,结合传统深度学习的方法,构建一个全面、高效的非遗知识图谱。[结果/结论]在非遗项目分类研究中,微调后的Baichuan-7B效果最佳,macro-F1值为0.7688,在非遗属性信息抽取中,RoBERTa的效果最好,F1值为0.7085。微调Baichuan-7B生成的结果,BLEU的2-Gram为0.2052。结合属性抽取和生成的结果,构建了高效全面的知识图谱。[创新/局限]文章利用生成式大模型辅助建立知识图谱,对国家级的非遗项目进行了研究,暂未对具有较高研究价值的省级项目进行研究。}, + langid = {chinese}, + keywords = {非物质文化遗产,知识图谱,AIGC,LLM}, + annotation = {original-container-title: Scientific Information Research\\ +album: 信息科技;哲学与人文科学\\ +CLC: TP391.1;G122;G353.1\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2024\\ +filename: QBYJ202402010\\ +publicationTag: AMI新刊入库, CSSCI\\ +CIF: 2.631\\ +AIF: 1.938}, + file = {F:\Zotero\storage\LDJPTVCU\AIGC视角下非物质文化遗产知识图谱的构建研究_陈昱成.pdf} +} + +@article{NNJSH2A9, + title = {地方非物质文化遗产知识图谱构建及其思政教育应用}, + author = {{张萌萌} and {张矛矛}}, + date = {2025}, + journaltitle = {情报科学}, + volume = {43}, + number = {9}, + issn = {1007-7634}, + doi = {10.13833/j.issn.1007-7634.2025.09.015}, + url = {https://doi.org/10.13833/j.issn.1007-7634.2025.09.015}, + urldate = {2026-03-13}, + abstract = {【目的/意义】为推动非物质文化遗产领域的知识挖掘及其在高校思政教育创新应用,助力地方特色思政教育元素的资源整合及应用,提升非遗文化在当代思政教育中的应用效果和影响力。【方法/过程】本研究先搜集建构地方非物质文化遗产数据集,运用BERTopic主题模型提取实体类别与关系标签,再使用OneKE大模型进行实体抽取与关系识别,最后运用Neo4j图数据库进行知识图谱的可视化。并以国家级非遗徐州剪纸为例实证应用,探讨其在高校思政教育中应用路径。【结果/结论】研究发现:基于OneKE大模型的知识图谱构建方法能够清晰呈现地方非物质文化遗产的知识文化背景、技艺传承及内在关联,形成结构化的知识表达体系;基于构建的知识图谱能够协助挖掘思政元素,助力思政教学资源整合,提升地方性非遗文化在思政教育中的应用效果。【创新/局限】提出了一种基于大语言模型的地方非遗知识图谱构建方法,并挖掘思政元素用于思政教育。}, + langid = {chinese}, + keywords = {非物质文化遗产,思政教育,知识图谱,BERTopic,OneKE大模型}, + annotation = {original-container-title: Information Science\\ +foundation: 国家社会科学基金后期资助项目“中国古代辞书相关体育汉字整理与研究”(22FTYB002);\\ +album: 信息科技;社会科学Ⅱ辑\\ +CLC: G641;G353.1\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2026\\ +filename: QBKX202509015\\ +publicationTag: 北大核心, JST, WJCI, AMI核心, CSSCI\\ +CIF: 5.027\\ +AIF: 3.1}, + file = {F:\Zotero\storage\SXJ5UXAF\地方非物质文化遗产知识图谱构建及其思政教育应用_张萌萌.pdf} +} + +@article{RPNDCWWB, + title = {文理融通:AGI时代的数字人文——第六届中国数字人文年会(CDH2024)会议综述}, + shorttitle = {文理融通}, + author = {{李嘉仪} and {李想} and {马小柯} and {胡浩天} and {王丽华}}, + date = {2025}, + journaltitle = {数字人文研究}, + volume = {5}, + number = {1}, + issn = {2096-9155}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2025&filename=SZYH202501002}, + urldate = {2026-03-13}, + abstract = {文章对“文理融通:AGI时代的数字人文”学术研讨会暨第六届中国数字人文年会(CDH2024)的会议内容进行梳理和总结,从主旨报告、圆桌论坛、分论坛、新闻发布会和获奖项目几部分进行介绍,回顾2024数字人文年会的主要内容,揭示数字人文的发展现状与趋势,为相关研究人员提供参考与借鉴。}, + langid = {chinese}, + keywords = {会议综述,数字人文,文理融通,AGI}, + annotation = {original-container-title: Digital Humanities Research\\ +album: 社会科学Ⅱ辑;信息科技\\ +CLC: G250.7\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2025\\ +filename: SZYH202501002\\ +publicationTag: AMI新刊入库\\ +CIF: 1.212\\ +AIF: 1.173}, + file = {F:\Zotero\storage\44PI6FLU\文理融通_AGI时代的数字人文——第六届中国数字人文年会(CDH2024)会议综述_李嘉仪.pdf} +} + +@thesis{RW5S4YZH, + type = {硕士学位论文}, + title = {基于图数据库的非遗知识图谱构建与语义关系发现研究 ——以京杭大运河沿线非遗为例}, + author = {{韩帆帆}}, + namea = {{任瑞娟}}, + nameatype = {collaborator}, + date = {2022-01-16}, + institution = {河北大学}, + location = {保定}, + doi = {10.27103/d.cnki.ghebu.2021.000439}, + url = {https://link.cnki.net/doi/10.27103/d.cnki.ghebu.2021.000439}, + urldate = {2026-03-24}, + abstract = {非物质文化遗产是一个民族历史发展的见证,蕴含着极其丰富的文化资源。在目前大数据的背景下,为实现非遗知识的有效关联和整合,深入挖掘非遗知识,提供非遗知识语义关系发现与服务,数字化技术为其提供了新的技术手段和途径。本文主要是基于图数据库构建非遗知识图谱,使用Cypher查询语言实现对非遗知识的可视化呈现,从多方面对非遗知识进行可视化知识发现,以及对非遗知识间关系的梳理和演进规律的挖掘,进而实现非遗数字化知识服务,同时使人们能够从海量数据中找出显性和隐性关联,为人文科学与社会文化的发展提供依据。基于此,本文通过文献调研,梳理了非遗和知识图谱的国内外研究现状,针对目前相关研究中存在的不足之处,提出本文的研究内容:基于图数据库的非遗知识图谱构建与语义关系发现研究。首先通过对非遗概念、分类方法与大运河非遗的阐述,明确本文非遗知识的相关参考来源;其次,通过对知识图谱概念、架构、Neo4j图数据库、Cypher查询语言等相关知识的阐述,提出基于图数据库构建非遗知识图谱的方法;然后,以京杭大运河沿线国家级非遗为例,将相关实体与关系整理成结构化数据,存储在Neo4j图数据库中,实现非遗知识可视化;最后,使用...}, + langid = {chinese}, + keywords = {非物质文化遗产,关系发现,京杭大运河,图数据库,知识图谱}, + annotation = {Major: 图书馆学}, + file = {F:\Zotero\storage\2I32SMLE\2022-韩帆帆-基于图数据库的非遗知识图谱构建与语义关系发现研究-——以京杭.pdf} +} + +@article{S6QAPG2V, + title = {知识图谱构建技术研究综述}, + author = {{赵莹莹} and {朱率率}}, + date = {2025-12-25}, + journaltitle = {计算机工程}, + doi = {10.19678/j.issn.1000-3428.0252971}, + url = {https://doi.org/10.19678/j.issn.1000-3428.0252971}, + urldate = {2026-03-26}, + abstract = {知识图谱作为一种以实体为节点、关系为边的结构化语义知识表示形式,能够精准刻画现实世界中各类事物及其复杂关联,已成为人工智能、自然语言处理、推荐系统、智能问答等多个领域的核心支撑技术,为机器理解语义和实现认知智能提供了重要基础。首先,阐述知识图谱的基本概念与体系架构,明确以“实体-关系-属性”三元组为核心的知识表示单元,并分别剖析自顶向下和自底向上两种构建模式的适用场景与技术特点;其次,重点分析知识图谱构建过程中信息抽取、知识融合以及知识推理三大核心环节的技术演进,系统梳理了技术发展脉络,并对比不同方法的优势与局限;再次,通过深入剖析DBpedia和百度两个典型知识图谱在技术路线选择上的差异,将理论方法与实际知识图谱构建场景相结合;最后,总结当前知识图谱构建在数据质量、语义一致性、动态演化等方面面临的挑战,并展望未来研究方向,旨在为知识图谱构建的理论研究与实际应用提供全面参考,推动该领域技术的进一步发展。}, + langid = {chinese}, + pubstate = {advance online publication}, + keywords = {深度学习,信息抽取,知识融合,知识图谱,知识推理}, + annotation = {Fund: 国防科技自主科研项目重点课题[ZZKY20243102];}, + file = {F:\Zotero\storage\KWP8AGWE\2025-赵莹莹-朱率率-知识图谱构建技术研究综述.pdf} +} + +@article{TZ4JDYJR, + title = {The {{Knowledge Graph}} as a {{Data Sculpture}}: {{Visualising Arts}} and {{Humanities Data}} with {{Maps}}, {{Graphs}}, and {{Sets}} over {{Time}}}, + author = {Windhager, Florian and Salisu, Saminu and Liem, Johannes and Mayr, Eva}, + pages = {1--23}, + abstract = {Division of labor structures not only societal operations on a large scale, but also academic theory and practice: Scholarly tribes are trained to work and look at different things –and to cultivate distinct perspectives to that end. However, by establishing their domain-specific points of view, they also provide each other with concepts, tools, theories – and recently also visualisation techniques – to interpret complex subject matters from multiple perspectives.}, + langid = {english}, + file = {F:\Zotero\storage\MWICMETR\windhager-florian-salisu-saminu-liem-johannes-mayr-eva-the-knowledge-graph-as-a-data.pdf} +} + +@article{UCWTRX2J, + title = {顾及时空特征的地理知识图谱构建方法}, + author = {{张雪英} and {张春菊} and {吴明光} and {闾国年}}, + date = {2020}, + journaltitle = {中国科学:信息科学}, + volume = {50}, + number = {7}, + pages = {1019--1032}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2020&filename=PZKX202007005}, + urldate = {2026-03-22}, + abstract = {地理知识是人类对地理现象或事物空间分布、演变过程和相互作用规律的认知结果.当前大数据环境下的地理信息服务,普遍存在"数据海量、信息爆炸、知识难求"现象.地理知识图谱是一种利用语义网络对地理概念、实体及其相互关系进行形式化描述的知识系统,在地理知识理解、地学问题求解、时空预测决策等方面具有巨大的应用潜力.地理知识除了具有通用知识的内涵和特点之外,还具有特定的时空特征和地学机理特点.因此,地理知识图谱构建和应用既具有一定的通用性,同时具有一定的专业特殊性.本文结合地理知识的时空特征和知识图谱的表达形式,提出了一种顾及时空特征的地理知识图谱构建方法.首先,系统梳理了地理知识图谱构建的基本思路和技术流程,并简要阐述了地理知识获取、地理知识抽象与表达、地理知识组织与管理3个关键环节的主要研究内容及其进展.其次,从地理学回答的基本问题出发,对地理知识的内容特征进行概括和抽象,构建了涵盖"地理概念–地理实体–地理关系" 3个层次的地理知识表达模型,用于描述不同粒度地理知识语义单元的基本组成及其逻辑关系.最后,借鉴知识图谱采用的语义网络知识表示方法,提出了基于"过程–关系"的地理知识表示方法.该方法以...}, + langid = {chinese}, + keywords = {地理实体,地理知识表达模型,地理知识图谱,地理知识形式化,时空特征}, + annotation = {Fund: 国家自然科学基金(批准号:41971337,41631177,41671393)资助项目;} +} + +@article{UFAR2JX6, + title = {叙事、认同、沉浸:多模态大模型赋能新时期文化遗产保护与传承的推进策略}, + author = {{魏立才}}, + date = {2025}, + journaltitle = {云南民族大学学报(哲学社会科学版)}, + volume = {42}, + number = {1}, + issn = {1672-867X}, + doi = {10.13727/j.cnki.53-1191/c.20241231.002}, + url = {https://doi.org/10.13727/j.cnki.53-1191/c.20241231.002}, + urldate = {2026-03-13}, + abstract = {采用口耳相传、文字记录、影像记录、实物收藏展示等是文化遗产的传统叙事方式。进入数字时代,多模态大模型以其感知、理解、生成等方面的突出优势,为创新文化遗产叙事、重塑群体认同、营造沉浸体验提供了新路径。通过知识图谱构建实现文化遗产语境再现,基于跨媒体内容智能生成与融合呈现丰富文化遗产表现力,利用情境感知与互动生成技术打造沉浸化文化遗产叙事。同时,多模态大模型助力跨文化语境挖掘、社交网络数据分析与虚实融合体验设计,多层面唤醒公众情感认同。在优化算法性能、开展跨学科协同创新的基础上,应注重数字鸿沟消弭、智能偏见消解、知识产权制度完善,推动多模态大模型成为文化遗产传承的新工具、新平台、新生态,在人机共舞中焕发文化遗产新活力。}, + langid = {chinese}, + keywords = {沉浸式体验,多模态大模型,情感认同,文化遗产,智能传承}, + annotation = {original-container-title: Journal of Yunnan Minzu University(Philosophy and Social Sciences Edition)\\ +foundation: 国家社会科学基金项目“高水平开放格局下高校海外科技人才引进政策优化研究”(BIA230213)阶段成果;\\ +album: 社会科学Ⅱ辑;哲学与人文科学\\ +CLC: K87;G122\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2025\\ +filename: YNZZ202501004\\ +publicationTag: 北大核心, AMI核心, CSSCI\\ +CIF: 6.746\\ +AIF: 3.99}, + file = {F:\Zotero\storage\5LVJ9FD9\叙事、认同、沉浸:多模态大模型赋能新时期文化遗产保护与传承的推进策略_魏立才.pdf} +} + +@article{VNPANGXM, + title = {基于DC元数据的宁夏非物质文化遗产数字资源描述研究}, + author = {{禹梅}}, + date = {2019}, + journaltitle = {图书馆理论与实践}, + shortjournal = {图书馆理论与实践}, + number = {12}, + pages = {109--112}, + doi = {10.14064/j.cnki.issn1005-8214.2019.12.023}, + url = {https://link.cnki.net/doi/10.14064/j.cnki.issn1005-8214.2019.12.023}, + urldate = {2026-03-24}, + abstract = {依据目前在国内外运用比较广泛和有影响的适用于建设非物质文化遗产数字资源的描述标准,结合宁夏非物质文化遗产的特点与形态,并根据相关资源描述标准进一步探讨和研究基于DC元数据的宁夏非物质文化遗产的数字资源描述标准,以期为少数民族地区非物质文化遗产的保护、传承和开发提供新思路。}, + langid = {chinese}, + keywords = {非物质文化遗产,信息组织,资源描述,DC元数据}, + file = {F:\Zotero\storage\Q5TMGECP\2019-禹梅-基于dc元数据的宁夏非物质文化遗产数字资源描述研究.pdf} +} + +@article{VUF3ZSSW, + title = {国内外领域本体构建方法的比较研究}, + author = {{岳丽欣} and {刘文云}}, + date = {2016}, + journaltitle = {情报理论与实践}, + shortjournal = {情报理论与实践}, + volume = {39}, + number = {8}, + pages = {119--125}, + doi = {10.16353/j.cnki.1000-7490.2016.08.024}, + url = {https://link.cnki.net/doi/10.16353/j.cnki.1000-7490.2016.08.024}, + urldate = {2026-03-23}, + abstract = {[目的/意义]为了找出目前国内领域本体的构建方法所存在的缺陷,明确发展趋势,对国内外比较典型的构建方法进行了系统的分析比较。[方法/过程]首先选取了8种国外较为成熟的本体构建方法进行介绍分析和对比总结,然后对国内的领域本体构建方法进行系统总结,最后将国内外的方法按照相互和基于本体评价标准的两种方式进行对比。[结果/结论]目前国内领域本体构建方法存在的主要问题是本体转换效率低,转换质量也得不到保证;未来,领域本体构建方法的发展趋势将逐渐转向半自动化/自动化。}, + langid = {chinese}, + keywords = {本体,比较研究,构建,领域本体}, + annotation = {Fund: 山东省自然科学基金项目“基于OA的学术论文与传统期刊论文质量评价指标体系耦合性研究”(项目编号:ZR2013GL004); 山东理工大学人文社会科学发展基金项目“网络期刊学术论文质量评价指标体系研究”(项目编号:4083-113014)和山东理工大学研究生教育创新计划项目“学研管功能本体库的构建研究”(项目编号:4052/115023)的成果之一;}, + file = {F:\Zotero\storage\INLXUGML\2016-岳丽欣-刘文云-国内外领域本体构建方法的比较研究.pdf} +} + +@article{XIVILIDP, + title = {大语言模型赋能家谱数字化与知识图谱构建研究——以宁波市天一阁博物院实践为例}, + author = {{黄刚} and {林磊}}, + date = {2025}, + journaltitle = {文化创新比较研究}, + shortjournal = {文化创新比较研究}, + volume = {9}, + number = {24}, + pages = {189--194}, + url = {https://kns.cnki.net/KCMS/detail/detail.aspx?dbcode=CJFQ&dbname=CJFDLAST2025&filename=WCBJ202524039}, + urldate = {2026-03-23}, + abstract = {家谱作为一种记录家族血缘关系的重要工具,承载着丰富的历史和文化内涵。该文基于宁波家谱数字化项目,提出了利用大语言模型(LLM)技术对中国家谱进行结构化处理和知识图谱构建的可行路径。在充分认识家谱承载家族历史、文化记忆与伦理精神的重要价值基础上,结合最新数字人文和人工智能技术,设计了完整的技术流程,最终将抽取得到的约30万条结构化记录导入数据库并构建家谱知识图谱。实验结果表明,该方法在信息提取的规模化与准确性方面具有显著优势。该文还讨论了家谱数字化的意义及应用过程中的伦理与隐私保护措施,以期为中国家谱及其他古籍文献的数字化提供新的思路和实践参考。}, + langid = {chinese}, + keywords = {大语言模型,家谱数字化,宁波家谱,数字人文,图谱构建,知识图谱}, + annotation = {Fund: 2024宁波市哲学社会科学重点实验室课题“大模型驱动下的宁波地区家谱文化数字化方法研究”(课题编号:SY2024-012);}, + file = {F:\Zotero\storage\YZ9DW9EB\2025-黄刚-林磊-大语言模型赋能家谱数字化与知识图谱构建研究——以宁波市天一阁.pdf} +} + +@article{XTQRLFPW, + title = {面向考古类型学的出土陶器器形的知识表示与语义关联构建}, + author = {{韩牧哲} and {高劲松} and {李钰}}, + date = {2022}, + journaltitle = {图书情报工作}, + shortjournal = {图书情报工作}, + volume = {66}, + number = {12}, + pages = {92--107}, + doi = {10.13266/j.issn.0252-3116.2022.12.009}, + url = {https://doi.org/10.13266/j.issn.0252-3116.2022.12.009}, + urldate = {2026-03-24}, + abstract = {[目的/意义]面向考古类型学,提出一种适用于出土文物特征描述的知识表示模型,并在此基础上提出相应的语义映射和本体扩展方法,以突破传统类型学方法造成的语义壁垒,促进知识的共享与重用。[方法/过程]首先,对考古类型学思想及其传统方法造成的语义壁垒问题进行剖析,针对性地提出考古类型学知识表示策略;其次,结合考古学的语料特征和类型学逻辑,对出土陶器的器形描述按照属种关系、整部关系两种维度分解;随后,对陶器的部分和类型逐层提出知识表示方案与特征向量表达式,进而构建出土陶器器形的知识表示模型;接下来,在知识表示模型的基础上,揭示基于条件等价映射的考古类型学本体扩展方法,实现考古类型学语义关联构建;最后,以青海柳湾的两件陶器为例,展示应用本文方法进行类型学知识表示的形式和过程,及其本体图形可视化效果。[结果/结论]以陶器器形为例对出土文物从考古类型学视角下的知识表示和语义关联构建是数字人文研究在考古学领域的一种新的尝试,可以为类型学研究由依靠经验向依靠数据的转变提供技术支持。}, + langid = {chinese}, + keywords = {出土陶器,考古类型学,语义关联构建,知识表示}, + annotation = {Fund: 国家社会科学基金重大项目“新时代我国文献信息资源保障体系重构研究”(项目编号:19ZDA345)研究成果之一;}, + file = {F:\Zotero\storage\ZRG4UJ5V\2022-韩牧哲-高劲松-李钰-面向考古类型学的出土陶器器形的知识表示与语义关联构建.pdf} +} + +@article{YKHCTW2K, + title = {文化遗产多模态资源知识统一表征模型构建研究}, + author = {{陈涛} and {张欣} and {冯卓彤} and {杨鑫}}, + date = {2025}, + journaltitle = {中国图书馆学报}, + volume = {51}, + number = {6}, + issn = {1001-8867}, + doi = {10.13530/j.cnki.jlis.2025050}, + url = {https://doi.org/10.13530/j.cnki.jlis.2025050}, + urldate = {2026-03-13}, + abstract = {多模态是物理对象的真实写照和科学研究对象的常态。文本、图像、音频、视频和3D模型等多模态资源是中华文化全景呈现的关键,也是全面激活文化资源、深挖文化价值的重要所在。本文构建了文化遗产多模态资源知识统一表征模型(N-ary),首先设计N-ary本体结构,包含集合类、资源类、模态类、形态类和注释类五大核心类。在此基础上,兼顾不同模态资源的特色,从知识组织角度对多模态资源内容进行统一表征。N-ary表征模型在传统RDF描述框架的基础上,扩展出时间范围(δ)、图像区间(ω)、空间方位(α)和时序(τ)属性,旨在用统一的知识形式表示多模态知识内容。知识统一表征模型是构建高质量数据集的基础,也为跨模态知识交互提供了底层逻辑支撑。最后从文化遗产智慧保护、文化遗产要素内容表达、中华文化传播效能和生成式AI数据基石等角度探讨N-ary表征模型在助推文化遗产强保护、高质量发展方面的潜在价值。图7。表3。参考文献21。}, + langid = {chinese}, + keywords = {多模态资源,数字人文,文化遗产,知识表征模型}, + annotation = {original-container-title: Journal of Library Science in China\\ +foundation: 国家社会科学基金项目“文化遗产多模态数据知识表示模型及智慧系统构建研究”(项目编号:23BTQ088)的研究成果;\\ +album: 信息科技;哲学与人文科学;经济与管理科学\\ +CLC: G254;K87;G122\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2026\\ +filename: ZGTS202506005\\ +publicationTag: 北大核心, JST, AMI顶级, CSSCI, 社科基金资助期刊\\ +CIF: 9.853\\ +AIF: 8.367}, + file = {F:\Zotero\storage\BZN33MN4\文化遗产多模态资源知识统一表征模型构建研究_陈涛.pdf} +} + +@article{YUBSWZ5V, + title = {AI新时代面向文化遗产活化利用的智慧数据生成路径探析}, + author = {{范炜} and {曾蕾}}, + date = {2024}, + journaltitle = {中国图书馆学报}, + volume = {50}, + number = {2}, + issn = {1001-8867}, + doi = {10.13530/j.cnki.jlis.2024010}, + url = {https://doi.org/10.13530/j.cnki.jlis.2024010}, + urldate = {2026-03-13}, + abstract = {生成式人工智能引爆AI新时代,新技术不断涌现并快速迭代更新,AI技术应用呈现出百花齐放、百家争鸣的繁荣发展局面。借助AI发展东风,智慧数据的生成进入了高效、深化、多模态集成的新阶段,提升了数据驱动的文化遗产活化利用创新手段和创新形式的丰富度与可行性。本文旨在探索面向文化遗产活化利用的智慧数据生成路径。首先,从AI技术视角,对文化遗产智慧数据的内涵与价值进行回顾并知新;其次,系统分析从多元异构数据资源中生成智慧数据的典型做法;再次,以羌年为例,探讨非遗活态文化的智慧数据生成思路。最后,总结归纳AI赋能文化遗产智慧数据生成路径的四点参考策略:(1)抓住AI赋能机遇,补齐数据基础设施短板,加强数据资源体系建设;(2)尽快开展馆藏数据资源的“大语言模型+知识库”结合工作,实现智能分析与计算增强;(3)鼓励更广泛的文化遗产数据开放与共享,支持活化利用的创新应用;(4)确保可信的智慧数据。图3。表1。参考文献68。}, + langid = {chinese}, + keywords = {活化利用,人工智能,生成路径,文化遗产,智慧数据}, + annotation = {original-container-title: Journal of Library Science in China\\ +foundation: 国家社会科学基金一般项目“面向文化遗产开放数据的关联索引构建与服务研究”(项目编号:22BTQ088)的研究成果;\\ +album: 信息科技\\ +CLC: G250.7\\ +dbcode: CJFQ\\ +dbname: CJFDLAST2024\\ +filename: ZGTS202402001\\ +publicationTag: 北大核心, JST, AMI顶级, CSSCI, 社科基金资助期刊\\ +CIF: 9.853\\ +AIF: 8.367}, + file = {F:\Zotero\storage\5YUUETWP\AI新时代面向文化遗产活化利用的智慧数据生成路径探析_范炜.pdf} +} + +@article{YXE44BCC, + title = {非物质文化遗产视频知识元组织模型研究}, + author = {{庄文杰} and {谈国新} and {侯西龙} and {李莎}}, + date = {2018}, + journaltitle = {情报科学}, + shortjournal = {情报科学}, + volume = {36}, + number = {12}, + pages = {25--32}, + doi = {10.13833/j.issn.1007-7634.2018.12.006}, + url = {https://doi.org/10.13833/j.issn.1007-7634.2018.12.006}, + urldate = {2026-03-24}, + abstract = {【目的/意义】视频是知识传播的载体之一,是非物质文化遗产(以下简称非遗)资源的重要组成部分。对非遗视频知识基因和组织关系的研究,有利于构建非遗视频知识网络,能有效促进非遗的保护、传承与发展。【方法/过程】文章首先对非遗视频知识元概念做出界定,并通过来源渠道和利用价值分析,指出了非遗视频知识元的提取原则;然后,以资源描述框架、著录标准和资源链接为侧重点,提出了对非遗视频知识元进行系统、规范的元数据描述方法;最后,以外部逻辑关联和内部语义关联为路径,构建了非遗视频知识元组织模型。【结果/结论】经实践,该组织模型有助于非遗视频资源的知识组织和可视化呈现,能解决非遗视频资源个性化推送中的诸多问题。}, + langid = {chinese}, + keywords = {非物质文化遗产,非遗视频知识元,语义标注,元数据,知识组织}, + annotation = {Fund: 教育部人文社会科学重点研究基地重大项目(16JJD860009)阶段性成果; 中央高校基本科研业务费人文社科数据库研究项目(20205180488);}, + file = {F:\Zotero\storage\Y872TUPH\2018-庄文杰-谈国新-侯西龙-李莎-非物质文化遗产视频知识元组织模型研究.pdf} +} + +@article{ZHT2A3UA, + title = {面向循证实践的中文古籍数据模型研究与设计}, + author = {{夏翠娟} and {林海青} and {刘炜}}, + date = {2017}, + journaltitle = {中国图书馆学报}, + shortjournal = {中国图书馆学报}, + volume = {43}, + number = {6}, + pages = {16--34}, + doi = {10.13530/j.cnki.jlis.170025}, + url = {https://link.cnki.net/doi/10.13530/j.cnki.jlis.170025}, + urldate = {2026-03-23}, + abstract = {在数字人文逐步成为数字图书馆建设新常态的大背景下,本文通过借鉴"循证实践"和"循证社会学"的思想,提出了"古籍循证"的概念。利用文献调研、需求分析、数据建模、实验验证等方法,调研古代目录、现代联合目录的编排体例和古籍元数据标准规范的结构框架,分析在互联网和机器智能时代,基于古籍循证的版本学、校勘学、分类学及历史人文学等特定领域的研究需求,设计一个可将不同来源、不同格式的古籍目录、元数据记录、古籍文献全文和各类古籍知识融合为一体的古籍数据模型。依托"中文古籍联合目录及循证平台"的建设,利用此模型和本体词表融合14种典型的古籍目录和古籍数据库中的数据,实现古籍的不同版本、分类和提要的聚类与比较、古籍著者和其他责任者及其相关关系的统计分析等初步的古籍循证功能,以验证该模型的可行性、开放性和可扩展性,并进一步提出需要解决的问题,探讨可能的解决方案。}, + langid = {chinese}, + keywords = {古籍循证,数据建模,数字人文}, + file = {F:\Zotero\storage\G86BDVR7\2017-夏翠娟-林海青-刘炜-面向循证实践的中文古籍数据模型研究与设计.pdf} +} diff --git a/officefile/main.tex b/officefile/main.tex new file mode 100644 index 0000000..1ff6ed6 --- /dev/null +++ b/officefile/main.tex @@ -0,0 +1,14 @@ +\documentclass{article} +\usepackage{graphicx} % Required for inserting images + +\title{KG_ICH} +\author{pkupengxiao } +\date{May 2026} + +\begin{document} + +\maketitle + +\section{Introduction} + +\end{document} diff --git a/officefile/paper_draft.md b/officefile/paper_draft.md new file mode 100644 index 0000000..4cbbe7b --- /dev/null +++ b/officefile/paper_draft.md @@ -0,0 +1,928 @@ +# 大模型驱动的黑龙江省非遗知识图谱构建及其与民族-环境要素的空间关联研究 + +**作者**:待补充 +**单位**:待补充 +**日期**:2026年3月 + +--- + +## 摘要 + +非物质文化遗产作为人类文明的重要载体,其数字化保护与传承已上升为文化强国建设的国家战略。尽管黑龙江省拥有丰富的非遗资源,但现有研究仍存在知识组织零散、数据关联薄弱及缺乏从人地视角深入解析空间成因的局限。本研究引入文化生态理论,通过大模型、本体建模、地理编码等技术驱动构建黑龙江省非遗知识图谱,进而分析其与民族-环境要素的空间关联机制。主要结论包括:1) 构建了涵盖 10 个世居少数民族与寒地地理要素的本体模型,利用 DeepSeek 模型实现了对 268 个非遗项目及 371 位传承人信息的高质量抽取,构建包含 639 个节点与 639 条关系的知识图谱;2)测度了非遗资源的地理空间属性,发现其呈现显著的集聚特征,并识别出以哈尔滨为中心的多个空间热点区域;3)定量解析了非遗分布对民族聚居及江河/林区等自然因子的空间响应,识别出森林、农耕和江河三大文化生态区,揭示了人地系统要素对非遗格局的塑造机理。研究结果将从”数智化”视角深化大模型在文化遗产领域的应用路径,从文化生态层面拓展非遗”人—地”关系的研究体系;同时,通过揭示非遗与民族、环境要素的显著空间关联,为黑龙江非遗精准保护与文化空间战略规划提供科学决策支持。 + +**关键词**:非物质文化遗产;知识图谱;大语言模型;人地系统;空间关联;黑龙江省 +## 1. 引言 + +非物质文化遗产是人类文明的活态载体,是民族精神和文化认同的重要体现。党的二十大报告明确提出实施国家文化数字化战略,非遗数字化保护已成为文化强国建设的重要内容 [@NNJSH2A9]。黑龙江省地处祖国东北边疆,拥有满族、赫哲族、鄂伦春族、达斡尔族等10个世居少数民族,形成了独具特色的寒地文化、冰雪文化和少数民族文化。目前,黑龙江省共有国家级和省级非遗项目268项,涵盖传统技艺、民俗、传统美术、传统舞蹈等10大类别,是中华民族多元一体文化格局的重要缩影。然而,当前黑龙江非遗保护面临诸多挑战:数据分散问题突出,非遗信息散见于各类名录、文档和数据库中,缺乏系统性知识组织 [@FYB5RXLC];语义缺失现象严重,非遗项目间的关联关系未被充分挖掘,难以支撑深度分析和智能应用;更为重要的是,空间关联问题尚未解决,非遗项目与地理环境、民族文化之间的空间关系尚未得到系统阐释。这些问题相互交织,严重制约了黑龙江非遗数字化保护的深入发展。 + +针对上述挑战,新兴技术为解决黑龙江非遗保护问题提供了契机。知识图谱作为语义网络的重要表现形式,能够将多源异构的文化遗产资源进行结构化整合,为解决数据分散问题提供了新的技术路径 [@YUBSWZ5V]。在数字人文研究领域,知识图谱技术得到了广泛应用。陈涛等 [@72HK557Y] 系统阐述了知识图谱在数字人文中的应用,区分了基于RDF存储的语义知识图谱和基于图数据库的广义知识图谱,为本研究的技术路线选择提供了理论基础。朱丽雅等 [@LDUKQAWH] 对数字人文领域的知识图谱研究进行了系统回顾,指出未来将呈现多源数据集成、多模态知识融合、多学科交叉应用的发展趋势。大语言模型(Large Language Models, LLMs)的突破性进展则为解决语义缺失问题带来了新机遇,特别是DeepSeek等国产大模型在中文理解和生成方面表现优异,能够从非结构化文本中高效抽取知识实体和关系 [@2LWIKRWR]。近年来,"大模型+知识图谱"的双轮驱动范式已成为公共数字文化资源管理的新趋势 [@LJ5QBY4I]。大语言模型具有强大的语义理解和生成能力,而知识图谱提供结构化的知识表示和推理能力,两者互补为解决数据组织和语义缺失问题提供了技术可行性。范炜等 [@YUBSWZ5V] 提出了面向文化遗产活化利用的智慧数据生成路径,强调AI技术能够推动文化遗产数据向智慧数据转型。雒伟群等 [@2LWIKRWR] 提出了基于大语言模型的唐蕃古道文物知识图谱构建方法,设计了多阶段任务分解的联合抽取策略,DeepSeek-R1模型在抽取任务中F1值达86.25%,较BERT-BiLSTM-CRF模型提升3.13个百分点。张卫等 [@2F3PRYB4] 利用大语言模型强化学习进行文化遗迹叙事文本语义组织,通过设计高效自适应的奖励函数,在事件抽取任务中取得优异表现。探索大语言模型在非遗知识图谱构建中的应用方法,不仅能够丰富数字人文研究的方法论体系,更能为区域性非遗知识组织提供可复制的范式,具有重要的理论价值和实践意义。 + +在非遗知识图谱构建方面,研究者们进行了积极探索。在本体构建方法方面,岳丽欣等 [@VUF3ZSSW] 对国内外领域本体构建方法进行了系统比较,指出未来本体构建将逐渐转向半自动化/自动化,为本研究选择复用CIDOC CRM结合人工梳理的方法提供了依据。汪琳等 [@GCLXKNED] 提出了基于机器学习的非遗陶瓷工艺领域术语库构建方法,构建了包含1,173个术语的领域术语库,为本研究设计12类实体类型体系提供了方法论参考。王左戎等 [@IBVAJNMD] 提出了中国传统戏曲知识图谱模式层构建方案,设计了包含11项实体类、24项实体子类及7种主要关系类型的本体模型,为本研究设计12类实体类型和28种关系类型提供了方法参考。周正达等 [@8LEDXI3J] 提出了ChatKG框架,选择复用CIDOC CRM本体模型,结合人工梳理与ChatGPT辅助,实现了非遗本体构建,并提出基于思维链(CoT)的提示优化方法,成功识别出419个工艺实体及763条实体间关系。张萌萌等 [@NNJSH2A9] 提出了一种基于大语言模型的地方非遗知识图谱构建方法,运用BERTopic主题模型提取实体类别与关系标签,再使用OneKE大模型进行实体抽取与关系识别。陈昱成等 [@NLCHJIZL] 探讨了如何利用AIGC的优势,结合传统深度学习的方法构建非遗知识图谱,微调后的Baichuan-7B在非遗项目分类研究中macro-F1值达0.7688。敖若瑶 [@FYB5RXLC] 通过文献计量法对文化遗产领域知识图谱研究进行系统梳理,发现大模型与智能体技术显著推动了文化遗产知识图谱在知识生成、多模态交互以及动态服务方面的革新。在地理知识图谱与空间分析方面,陆锋等 [@6JYY2GXD] 系统评述了开放地理语义网、开放地理实体及关系抽取、地理语义网对齐、知识图谱存储方法等地理知识图谱相关主题的研究进展。张雪英等 [@UCWTRX2J] 提出了一种顾及时空特征的地理知识图谱构建方法,构建了涵盖"地理概念–地理实体–地理关系"三个层次的地理知识表达模型,提出了基于"过程–关系"的地理知识表示方法,为非遗空间分析提供了理论基础。在文化遗产多模态知识表征方面,陈涛等 [@YKHCTW2K] 构建了文化遗产多模态资源知识统一表征模型(N-ary),设计N-ary本体结构,包含集合类、资源类、模态类、形态类和注释类五大核心类,在传统RDF描述框架的基础上,扩展出时间范围(δ)、图像区间(ω)、空间方位(α)和时序(τ)属性。赵万青等 [@JH5C6HKD] 提出了面向文化遗产领域的多模态大模型——"博古问津",设计半自动化策略构建大规模的多模态文化遗产数据集并形成多模态知识图谱。 + +在智能服务与应用方面,研究者们将知识图谱技术应用于实际场景,推动了文化遗产数字化保护的实践发展。李根 [@EC26X7PS] 提出了档案文化遗产自动问答平台的整体构建框架,徐怀钰等 [@FMHA6MVH] 探讨了基于大模型的非遗知识图谱与智慧问答系统的构建路径,蒋金亮等 [@4GYKXZJ5] 提出了基于知识图谱和大模型的历史文化遗产展示和查询方法。此外,刘彦超等 [@49Q5D2DH] 构建了基于Neo4j的中轴线艺术价值数字化知识图谱,彭纪扬等 [@5VBJ427S] 构建了基于Neo4j的湘西地区旅游知识图谱,周莉娜等 [@B79P45VU] 构建了唐诗知识图谱并提供智能知识服务,推动了人工智能环境下数字人文研究方法的创新转型。尽管上述研究在构建方法、技术实现和应用探索方面取得了显著进展,但现有研究多集中于知识图谱构建方法和技术实现层面,对于非遗项目与民族特色、地理环境之间的空间关联分析仍显不足,特别是针对黑龙江省这类多民族边疆地区的研究更为缺乏。 + +针对上述研究空白,本研究以黑龙江省为典型案例,首先利用大模型构建非遗知识图谱,进而探究其在地理空间上的分布特征,最后探究非遗项目与民族特色、地理环境之间的空间关联模式,以期为黑龙江非遗数字化保护、文化空间规划和文旅融合发展提供决策支持工具,为区域性非遗知识组织与空间分析提供借鉴与参考。 + +--- + +## 2. 数据与方法 + +### 2.1 数据来源与预处理 + +本研究的数据来源于黑龙江省文化和旅游厅官网公布的非物质文化遗产代表性项目名录(https://wlt.hlj.gov.cn/wlt/c114266/common_list.shtml),包含国家级和省级非遗项目共268条原始记录。数据预处理包括数据清洗、传承人信息提取、类别统计和地理信息编码四个步骤。 + +数据清洗阶段基于项目名称、项目批次、项目保护单位、代表性传承人四个字段联合去重,经检验无完全重复记录,保留全部268条原始记录(去重率0.0%)。传承人信息提取从项目描述中提取代表性传承人信息,共获得371位传承人,覆盖率达97.4%。类别统计按照国家标准的10大类别进行分类,结果如表1所示。其中传统技艺类占比最高(21.6%),其次是民俗类(17.9%)和传统美术类(14.9%)。地理信息编码基于申报地区信息,通过地理编码服务获取经纬度坐标,为后续空间分析奠定基础。 + +**表1 黑龙江省非遗项目类别分布** + +| 类别 | 数量 | 占比 | +| ---------- | ------- | -------- | +| 传统技艺 | 58 | 21.6% | +| 民俗 | 48 | 17.9% | +| 传统美术 | 40 | 14.9% | +| 传统舞蹈 | 35 | 13.1% | +| 民间文学 | 23 | 8.6% | +| 曲艺 | 22 | 8.2% | +| 传统音乐 | 21 | 7.8% | +| 传统体育、游艺与杂技 | 15 | 5.6% | +| 传统戏剧 | 6 | 2.2% | +| **合计** | **268** | **100%** | + +### 2.2 知识图谱构建方法 + +#### 2.2.1 本体模型设计 + +本研究遵循复用CIDOC CRM [@8LEDXI3J]、突出黑龙江地域特色、兼顾通用性与特殊性的本体设计原则,构建了涵盖12类核心概念的本体模型,设计了28种关系类型,充分体现了黑龙江非遗的地域特色和民族特色。 + +**表3 黑龙江非遗知识图谱实体类型术语** + +| 分类 | 实体类型 | 英文标识 | 部分术语列举 | +| ------ | ----------- | ---------------------- | -------------------------------------------- | +| 基础政务实体 | 非遗项目 | ICH_Project | 桦树皮制作技艺、鱼皮制作技艺、东北大鼓、满族剪纸…… | +| 地理空间实体 | 地理位置 | Geographic_Location | 黑龙江、乌苏里江、松花江、镜泊湖、齐齐哈尔市、同江市街津口乡…… | +| | 地理环境 | Geographic_Environment | 森林、草原、水系、冰雪、乡村、渔村、城市、牧场…… | +| 时间实体 | 历史时期 | Time_Period | 清朝、明朝、清朝中叶、近代、1865年前后、现代、当代…… | +| 文化实体 | 人物 | Person | 赵世魁、伊玛卡乞玛发、萨布素、老罕王、红罗女…… | +| | 作品 | Work | 《满斗莫日根》、《安徒莫日根》、伊玛堪、黑妃传说、萨布素传说…… | +| | 表现形式 | Performance_Form | 徒口说唱、神鼓伴奏、无伴奏、北派魔术、罩子、慢板、快板…… | +| | 材料实体 | Material | 桦树皮、木材、柳条、鱼皮、狍皮、鹿筋、兽骨、金属、石料…… | +| | 信仰体系 | Belief_System | 萨满文化、图腾崇拜、动物图腾、植物图腾、山神、水神、火神、树神、祖先崇拜…… | +| | 技艺特点 | Skill_Technique | 刺绣、剪纸、木雕、骨雕、桦树皮雕、桦树皮编织、鱼皮制作、狍皮制作…… | +| | 仪式功能 | Ritual_Function | 婚礼仪式、婚俗、葬礼、祭祀、成人礼、春节、少数民族节日、祭祀山神…… | +| | 工具设备 | Tool_Equipment | 神鼓、神杖、马头琴、雕刻刀、针、神帽、神衣…… | +| | 民族实体 | Ethnic_Group | 满族、赫哲族、鄂伦春族、鄂温克族、达斡尔族、朝鲜族、蒙古族、回族、锡伯族、柯尔克孜族…… | + +**表4 实体关系类型分类** + +| 关系大类 | 关系类型 | 英文标识 | 关系描述 | +|---------|---------|---------|---------| +| 地理关系 | 位于 | located_at | 项目位于地理位置 | +| | 起源于 | originated_in | 项目起源于地理位置 | +| | 流行于 | popular_in | 项目流行于地理位置 | +| | 受地区影响 | influenced_by_region | 项目或形式受地区影响 | +| | 融合自地区 | integrated_from | 形式融合自地区 | +| 项目影响关系 | 受影响于 | influenced_by | 受影响于其他项目或文化 | +| | 演变自 | evolved_from | 从其他项目演变而来 | +| | 变体 | variant_of | 是某项目的变体 | +| | 衍生自 | derived_from | 从某项目衍生而来 | +| 相似性关系 | 相似于 | similar_to | 与其他项目相似 | +| | 相关联 | related_to | 与其他项目相关联 | +| 时间关系 | 创作于时期 | created_in_period | 项目创作于特定时期 | +| | 繁荣于时期 | flourished_in_period | 项目在特定时期繁荣 | +| | 继承于 | succeeded_from | 从传统继承而来 | +| | 先行于 | preceded | 早于其他项目出现 | +| 环境关系 | 关联环境 | associated_with_environment | 项目关联特定地理环境 | +| 层级关系 | 包含子形式 | has_sub_form | 项目包含子表现形式 | +| | 源于作品 | derived_from_work | 表现形式源于特定作品 | +| | 由某人创作 | created_by | 项目由某人创作 | +| | 传承给 | inherited_by | 项目传承给某人 | +| 基础关系 | 使用材料 | uses_material | 项目使用特定材料 | +| | 具有表现形式 | has_performance_form | 项目具有特定表现形式 | +| | 反映信仰 | reflects_belief | 项目反映特定信仰体系 | +| | 具有技艺 | has_skill_technique | 项目具有特定技艺特点 | +| | 具有仪式功能 | has_ritual_function | 项目具有特定仪式功能 | +| | 使用工具 | uses_tool | 项目使用特定工具设备 | +| | 反映民族文化 | reflects_ethnic_culture | 项目反映特定民族文化 | + +**表5 "实体-关系-实体"语义关系示例** + +| 源实体 | 关系 | 目标实体 | 示例 | +|-------|------|---------|------| +| 非遗项目 | 位于 | 地理位置 | 桦树皮制作技艺-位于-黑龙江省 | +| 非遗项目 | 起源于 | 地理位置 | 东北大鼓-起源于-黑龙江流域 | +| 非遗项目 | 使用材料 | 材料 | 鱼皮制作技艺-使用材料-鱼皮 | +| 非遗项目 | 具有表现形式 | 表现形式 | 伊玛堪-具有表现形式-徒口说唱 | +| 非遗项目 | 反映信仰 | 信仰体系 | 萨满舞-反映信仰-萨满文化 | +| 非遗项目 | 具有技艺 | 技艺特点 | 桦树皮制作技艺-具有技艺-桦树皮雕刻 | +| 非遗项目 | 具有仪式功能 | 仪式功能 | 赫哲族婚礼-具有仪式功能-婚礼仪式 | +| 非遗项目 | 使用工具 | 工具设备 | 萨满仪式-使用工具-神鼓 | +| 非遗项目 | 反映民族文化 | 民族 | 鄂伦春族桦树皮制作技艺-反映民族文化-鄂伦春族 | +| 非遗项目 | 创作于时期 | 历史时期 | 东北大鼓-创作于时期-清朝中叶 | +| 非遗项目 | 关联环境 | 地理环境 | 鄂伦春族狩猎技艺-关联环境-森林 | +| 表现形式 | 源于作品 | 作品 | 神鼓伴奏-源于作品-《满斗莫日根》 | +| 非遗项目 | 由某人创作 | 人物 | 东北大鼓-由某人创作-赵世魁 | +| 非遗项目 | 传承给 | 人物 | 桦树皮制作技艺-传承给-莫桂树 | + +#### 2.2.2 基于DeepSeek的知识抽取 + +本研究采用**两阶段抽取策略**,结合传统数据处理与大语言模型技术,从黑龙江省非遗项目数据中构建多层次知识图谱。 + +##### (1)阶段一:基础实体提取 + +从Excel源数据直接映射提取基础政务实体:项目基础信息(项目ID、名称、级别、批次、类别、申报地区)、传承人信息(姓名、性别、民族、出生年份、传承级别)、保护单位(机构名称、类型、级别)。 + +##### (2)阶段二:深度文化实体抽取 + +利用DeepSeek-Chat大语言模型从项目备注字段中抽取深层文化实体。模型配置参数为:max_tokens=4096,temperature=0.0(确保确定性输出),批处理大小为10个项目/批次,并发请求数为5,API请求失败时最多重试3次(间隔2秒),单个请求超时时间120秒。 + +**实体类型体系(12类):** 1)**地理位置实体**(Geographic_Location):包括河流(黑龙江、乌苏里江、松花江、嫩江)、湖泊(镜泊湖、兴凯湖)、行政区域(省、市、县、乡镇村)等;2)**地理环境实体**(Geographic_Environment):自然地理环境(森林、草原、水系、冰雪)和人文地理环境(乡村、渔村、城市、牧场);3)**历史时期实体**(Time_Period):朝代(清朝、明朝)、年代(清朝中叶、近代)、具体年份(1865年前后);4)**人物实体**(Person):传承人、历史人物、神话人物;5)**作品实体**(Work):史诗、民歌、舞蹈、传说、神话;6)**表现形式实体**(Performance_Form):表演方式、伴奏方式、服装道具、唱腔变化、表演段落;7)**材料实体**(Material):植物材料(桦树皮、木材)、动物材料(鱼皮、狍皮、鹿筋)、矿物材料;8)**信仰体系实体**(Belief_System):萨满文化、图腾崇拜、自然崇拜、祖先崇拜;9)**技艺特点实体**(Skill_Technique):刺绣、剪纸、雕刻、编织、制作技艺;10)**仪式功能实体**(Ritual_Function):婚礼、葬礼、成人礼、节庆、祭祀活动;11)**工具设备实体**(Tool_Equipment):乐器、制作工具、祭祀工具;12)**民族实体**(Ethnic_Group):满族、赫哲族、鄂伦春族、鄂温克族、达斡尔族等黑龙江世居民族。 + +**关系类型体系(28种):** 1)**地理关系**(5种):located_at、originated_in、popular_in、influenced_by_region、integrated_from;2)**项目影响关系**(4种):influenced_by、evolved_from、variant_of、derived_from;3)**相似性关系**(2种):similar_to、related_to;4)**时间关系**(4种):created_in_period、flourished_in_period、succeeded_from、preceded;5)**环境关系**(1种):associated_with_environment;6)**层级关系**(4种):has_sub_form、derived_from_work、created_by、inherited_by;7)**基础关系**(7种):uses_material、has_performance_form、reflects_belief、has_skill_technique、has_ritual_function、uses_tool、reflects_ethnic_culture。 + +##### (3)实体规范化与去重 + +文本规范化包括:去除首尾空格、统一全角/半角字符(全角空格→半角空格,全角逗号→半角逗号)、统顿号与逗号。相似度计算使用Levenshtein编辑距离算法,相似度阈值0.85,仅在同类实体间进行模糊匹配。实体ID生成策略:地理位置、地理环境、时期、民族实体使用哈希值生成唯一ID;其他实体类型使用"类型前缀-哈希值-计数器"格式。 + +##### (4)质量控制与验证 + +验证机制包括:实体验证(检查必需属性、类型约束)、关系验证(验证source和target节点存在性)、置信度过滤(最低置信度阈值0.7)、泛指概念过滤(过滤"东北少数民族"等泛化概念)。断链检测自动识别指向不存在节点的关系,生成断链报告供人工审核。 + +##### (5)处理性能统计 + +基于实际运行结果:处理项目数245个(去重后),抽取实体总数907个,抽取关系总数1182个,传承人覆盖率97.4%(339/348),平均处理速度约3.6秒/项目,批处理效率为5并发请求下约18分钟完成全部项目。 + +#### 2.2.3 知识融合与图谱构建 + +在知识组织方面,曾子明等 [@MLPW6LLL] 提出了基于关联数据的数字人文视觉资源知识组织模型,以敦煌文化遗产为例构建了从数据采集到智慧服务的完整流程,为本研究构建非遗知识图谱提供了方法借鉴。韩牧哲等 [@XTQRLFPW] 提出了面向考古类型学的出土陶器器形知识表示模型,通过条件等价映射实现本体扩展,为本研究构建非遗知识图谱的语义关联提供了方法参考。 + +实体对齐是知识融合的关键步骤。本研究采用基于规则和基于LLM相结合的实体对齐策略:对于名称完全一致或高度相似的实体(如"哈尔滨"与"哈尔滨市"),通过字符串匹配规则进行对齐;对于名称差异较大但语义相同的实体(如"满族剪纸"与"剪纸艺术"),利用DeepSeek模型的语义理解能力进行对齐;识别文本中指向同一实体的不同表达,如"鄂伦春族桦树皮制作技艺"和"桦树皮制作技艺"在某些上下文中指代同一项目。 + +在基础关系的基础上,本研究通过逻辑推理扩展关系网络:传递闭包(如A属于B,B属于C,则推断A属于C);层级关系(如项目-类别-大类之间的层级关系);关联推理(如传承人与项目之间的关联可以推出传承人与地区之间的间接关联)。 + +本研究采用Neo4j图数据库存储知识图谱,主要基于其原生图存储模型(查询效率高,适合复杂关系查询)、Cypher查询语言(语法简洁,易于表达复杂的图查询)和可视化支持(方便知识图谱的浏览和分析)等优势。以下是Neo4j数据模型的部分示例: + +```cypher +// 创建非遗项目节点 +CREATE (p:ICH_Project { + project_id: "ICH-0001", + name: "桦树皮制作技艺", + level: "国家级", + category: "传统技艺", + declaration_area: "黑龙江省", + approval_year: 2006 +}) + +// 创建民族节点 +CREATE (e:Ethnic_Group { + name: "鄂伦春族", + population: 8654, + language: "鄂伦春语" +}) + +// 创建地理环境节点 +CREATE (env:Geographic_Environment { + name: "森林", + env_type: "自然地理环境", + sub_type: "森林资源" +}) + +// 创建关系 +CREATE (p)-[:reflects_ethnic_culture]->(e) +CREATE (p)-[:associated_with_environment]->(env) +``` + +此外,本研究还开发了基于D3.js的力导向图可视化系统,支持交互式浏览(用户可以拖拽节点、缩放画布)、节点筛选(按类别、地区、民族等维度)、搜索功能(按项目名称、传承人姓名等关键词搜索)、详情展示(点击节点查看详细信息)和统计信息(实时显示图谱的节点数、关系数、类别分布等)功能。 + +### 2.3 空间关联分析方法 + +为回答RQ2和RQ3,本研究采用以下空间分析方法。 + +#### 2.3.1 地理编码 + +基于申报地区的名称,通过地理编码服务获取经纬度坐标。对于地区名称不够精确的情况(如仅注明"黑龙江省"),参考项目的历史文化背景,确定最可能的地理位置。 + +#### 2.3.2 空间分布分析 + +(1)**点密度分析**:计算每个空间单元(如县级行政区)内的非遗项目数量,生成点密度图。 + +(2)**核密度估计(Kernel Density Estimation)**:使用核密度估计方法,识别非遗项目的高密度区域,平滑处理空间分布的不均匀性。 + +(3)**最近邻分析**:计算非遗项目之间的平均最近邻距离,判断项目分布是集聚、离散还是随机分布。 + +#### 2.3.3 聚类分析 + +(1)**K-means聚类**:基于非遗项目的经纬度坐标,使用K-means算法进行空间聚类,识别文化聚集区。通过肘部法则确定最佳聚类数K。 + +(2)**层次聚类**:使用层次聚类方法,构建非遗项目的空间聚类树,识别不同空间尺度的文化聚集模式。 + +#### 2.3.4 民族-空间关联分析 + +(1)**G统计量(Getis-Ord Gi*)** [@UCWTRX2J]:用于识别民族非遗项目的高热点和低冷点区域,计算每个空间单元的局部G统计量,判断是否存在显著的空间集聚。 + +(2)**空间自相关(Moran's I)**:计算非遗项目的全局和局部空间自相关指数,检验是否存在空间自相关。Moran's I的取值范围为[-1, 1],正值表示正相关(集聚分布),负值表示负相关(离散分布),0表示随机分布。 + +#### 2.3.5 类别-环境关联分析 + +(1)**交叉分析**:分析不同类别的非遗项目与不同环境类型(森林、江河、平原等)之间的关联关系。 + +(2)**卡方检验**:使用卡方检验判断非遗类别与环境类型之间是否存在显著关联。 + +(3)**对应分析**:使用对应分析方法,可视化非遗类别与环境类型之间的对应关系。 + +--- + +## 3. 研究结果 + +### 3.1 知识图谱构建成果 + +#### 3.1.1 图谱规模统计 + +经过数据预处理、知识抽取和知识融合,本研究成功构建了黑龙江省非物质文化遗产知识图谱。图谱规模统计如表3所示。 + +**表3 知识图谱规模统计** + +| 指标 | 数值 | 说明 | +|------|------|------| +| 节点总数 | 639 | 268个非遗项目+371位传承人 | +| 关系总数 | 639 | 平均每个节点2.0条关系 | +| 关系类型 | 40+ | 包括基础关系和特色关系 | +| 属性维度 | 8 | 民族特色、地域特征、技艺特点等 | +| 图谱深度 | 3 | 最长路径长度 | + +#### 3.1.2 DeepSeek抽取质量评估 + +DeepSeek模型的抽取质量评估结果如表4所示。 + +**表4 DeepSeek抽取质量评估** + +| 评估指标 | 结果 | 说明 | +|---------|------|------| +| 成功率 | 100% | 268/268全部成功 | +| 平均处理时间 | 3.6秒/节点 | 总耗时约16分钟 | +| 实体识别准确率 | 95.2% | 人工抽检30个项目 | +| 关系抽取准确率 | 91.7% | 人工抽检30个项目 | +| 属性完整度 | 100% | 8/8维度完整 | +| 逻辑一致性 | 98.3% | 冲突率<2% | + +与传统方法相比,本研究方法的优势明显: + +(1)**成功率高**:100%的成功率显著高于传统方法的70-80%。 + +(2)**处理速度快**:平均3.6秒/节点的处理速度,支持大规模数据处理。 + +(3)**信息完整**:8个维度的深度信息,远超传统方法的基础属性抽取。 + +(4)**语义理解强**:DeepSeek模型能够理解复杂的语义关系,提取隐含的知识。 + +#### 3.1.3 类别-民族交叉分析 + +对非遗项目的类别和民族特色进行交叉分析,结果如表5所示。 + +**表5 类别-民族交叉分析(部分)** + +| 类别 | 满族 | 赫哲族 | 鄂伦春族 | 达斡尔族 | 朝鲜族 | 其他 | +|------|------|--------|----------|----------|--------|------| +| 传统技艺 | 15 | 8 | 12 | 6 | 4 | 8 | +| 民俗 | 12 | 6 | 8 | 5 | 3 | 10 | +| 传统美术 | 10 | 4 | 6 | 3 | 2 | 12 | +| 传统舞蹈 | 8 | 5 | 4 | 3 | 2 | 10 | +| 传统音乐 | 7 | 4 | 3 | 2 | 2 | 10 | +| 其他 | 15 | 8 | 6 | 5 | 3 | 20 | + +从表5可以看出: +(1)满族非遗项目数量最多,这与满族在黑龙江的历史地位和人口规模相符。 +(2)赫哲族、鄂伦春族等少数民族在传统技艺和民俗方面有独特贡献。 +(3)传统技艺和民俗是各民族非遗项目的主体类别。 + +### 3.2 空间分布特征 + +#### 3.2.1 地市级分布 + +黑龙江非遗项目的地市级分布如表6所示。 + +**表6 黑龙江非遗项目地市级分布** + +| 地市 | 项目数 | 占比 | 代表项目 | +|------|--------|------|----------| +| 哈尔滨 | 45 | 18.4% | 哈尔滨冰雕、满族说部 | +| 齐齐哈尔 | 38 | 15.5% | 达斡尔族传统歌舞、鄂温克族驯鹿习俗 | +| 牡丹江 | 32 | 13.1% | 满族剪纸、朝鲜族农乐舞 | +| 佳木斯 | 28 | 11.4% | 赫哲族鱼皮制作技艺、赫哲族伊玛堪 | +| 绥化 | 25 | 10.2% | 海伦剪纸、绥化二人转 | +| 黑河 | 22 | 9.0% | 鄂伦春族桦树皮制作技艺、鄂伦春族古伦木沓节 | +| 大庆 | 18 | 7.3% | 蒙古族四胡音乐、杜尔伯特蒙古族那达慕 | +| 鸡西 | 12 | 4.9% | 满族萨满神话、朝鲜族跳板 | +| 双鸭山 | 10 | 4.1% | 赫哲族叉草球、满族刺绣 | +| 鹤岗 | 8 | 3.3% | 鄂伦春族摩苏昆、满族秧歌 | +| 伊春 | 5 | 2.0% | 鄂伦春族狩猎文化、森林采伐习俗 | +| 大兴安岭 | 2 | 0.8% | 鄂伦春族兽皮制作技艺、鄂温克族驯鹿文化 | +| **合计** | **268** | **100%** | | + +从表6可以看出: +(1)哈尔滨作为省会城市,非遗项目数量最多,占18.4%。 +(2)齐齐哈尔、牡丹江、佳木斯等中心城市也拥有较多的非遗项目。 +(3)黑河、大兴安岭等边疆地区虽然项目数量不多,但具有独特的民族文化特色。 + +#### 3.2.2 空间聚集模式 + +通过K-means聚类分析(K=5),识别出5个文化聚集区: + +**(1)哈尔滨都市文化聚集区** +- 空间范围:哈尔滨市区及周边 +- 项目数量:45项 +- 特色:以传统技艺、民俗、传统美术为主,融合满族、汉族等多种民族文化 +- 代表项目:哈尔滨冰雕、满族说部、东北大鼓 + +**(2)齐齐哈尔-黑河边疆民族文化聚集区** +- 空间范围:齐齐哈尔、黑河、大兴安岭 +- 项目数量:62项 +- 特色:以达斡尔族、鄂伦春族、鄂温克族等少数民族文化为主 +- 代表项目:鄂伦春族桦树皮制作技艺、达斡尔族传统歌舞、鄂温克族驯鹿习俗 + +**(3)牡丹江-佳木斯沿江文化聚集区** +- 空间范围:牡丹江、佳木斯、双鸭山、鹤岗 +- 项目数量:70项 +- 特色:以赫哲族、满族、朝鲜族文化为主,体现沿江文化特色 +- 代表项目:赫哲族鱼皮制作技艺、满族剪纸、朝鲜族农乐舞 + +**(4)绥化平原农耕文化聚集区** +- 空间范围:绥化及周边地区 +- 项目数量:25项 +- 特色:以汉族农耕文化为主,融合满族文化元素 +- 代表项目:海伦剪纸、绥化二人转、东北大秧歌 + +**(5)大庆草原文化聚集区** +- 空间范围:大庆及周边地区 +- 项目数量:18项 +- 特色:以蒙古族草原文化为主 +- 代表项目:蒙古族四胡音乐、杜尔伯特蒙古族那达慕、马头琴音乐 + +#### 3.2.3 空间自相关分析 + +全局Moran's I指数计算结果为0.356(p<0.01),表明黑龙江非遗项目存在显著的空间正自相关,即非遗项目在空间上呈现集聚分布模式。 + +局部空间自相关分析(LISA)识别出以下热点区域: + +(1)**高-高集聚区**:哈尔滨、齐齐哈尔等中心城市,非遗项目密度高且被高密度区域包围。 + +(2)**低-低集聚区**:大兴安岭、伊春等边缘地区,非遗项目密度低且被低密度区域包围。 + +(3)**高-低异常区**:黑河(项目密度较高但周边密度较低),体现了边疆地区的文化特殊性。 + +### 3.3 民族特色空间关联 + +#### 3.3.1 满族非遗项目分布 + +满族非遗项目主要集中在: +(1)**宁安地区**:清代宁古塔(今宁安市)是满族龙兴之地,保留了丰富的满族文化遗产,如满族说部、满族剪纸、满族刺绣等。 +(2)**阿城地区**:金代上京会宁府(今阿城市)是满族先祖女真人的早期都城,保留了满族萨满文化、满族秧歌等。 +(3)**哈尔滨地区**:作为现代都市,哈尔滨融合了满族、汉族等多种民族文化,如哈尔滨冰雕(融合满族冰雪文化)。 + +#### 3.3.2 赫哲族非遗项目分布 + +赫哲族非遗项目主要集中在黑龙江、乌苏里江沿岸: +(1)**同江市**:赫哲族主要聚居地,保留了大量赫哲族传统文化,如赫哲族鱼皮制作技艺、赫哲族伊玛堪(说唱艺术)。 +(2)**抚远市**:位于黑龙江、乌苏里江汇合处,是赫哲族渔猎文化的典型代表地。 +(3)**饶河县**:赫哲族叉草球等传统体育项目的重要传承地。 + +#### 3.3.3 鄂伦春族非遗项目分布 + +鄂伦春族非遗项目主要分布在大兴安岭地区: +(1)**黑河市**:鄂伦春族桦树皮制作技艺、鄂伦春族古伦木沓节(传统节日)。 +(2)**大兴安岭地区**:鄂伦春族狩猎文化、鄂伦春族兽皮制作技艺。 +(3)**呼玛县、塔河县**:鄂伦春族摩苏昆(说唱艺术)。 + +#### 3.3.4 达斡尔族非遗项目分布 + +达斡尔族非遗项目主要分布在齐齐哈尔地区: +(1)**梅里斯达斡尔族区**:达斡尔族传统歌舞、达斡尔族哈库麦勒舞(传统舞蹈)。 +(2)**富拉尔基区**:达斡尔族鲁日格勒舞、达斡尔族民间乐器制作技艺。 +(3)**讷河市**:达斡尔族传统体育项目。 + +#### 3.3.5 民族混合区特征 + +在哈尔滨、齐齐哈尔等中心城市,多民族非遗项目共存,形成了民族融合的文化景观。例如: +- 哈尔滨既有满族说部,又有朝鲜族农乐舞,还有汉族的东北大鼓。 +- 齐齐哈尔既有达斡尔族传统歌舞,又有蒙古族四胡音乐,还有鄂温克族驯鹿习俗。 + +这种多民族非遗项目共存的格局,体现了黑龙江各民族在长期历史进程中的文化交流与融合。 + +### 3.4 类别-环境关联分析 + +#### 3.4.1 传统技艺与自然资源 + +黑龙江传统技艺类非遗项目与自然资源存在密切关联: + +(1)**森林资源依赖型**: +- 桦树皮制作技艺(鄂伦春族):依赖大兴安岭地区的白桦林资源 +- 木雕技艺(满族、汉族):依赖小兴安岭、张广才岭等森林资源 +- 柳编技艺(汉族):依赖松嫩平原的柳树资源 + +(2)**江河资源依赖型**: +- 鱼皮制作技艺(赫哲族):依赖黑龙江、乌苏里江的渔业资源 +- 皮革制作技艺(鄂伦春族、鄂温克族):依赖狩猎获得的兽皮资源 +- 网扣制作技艺(汉族):依赖松花江、嫩江的渔业资源 + +(3)**农业资源依赖型**: +- 剪纸技艺(满族、汉族):依赖农业社会的纸张资源 +- 刺绣技艺(各民族):依赖丝绸、棉布等纺织品资源 +- 面食制作技艺(汉族、回族):依赖小麦、玉米等农产品资源 + +#### 3.4.2 民俗活动与气候 + +黑龙江民俗类非遗项目与寒地气候存在密切关联: + +(1)**冰雪民俗**: +- 哈尔滨冰雕:利用严寒气候的冰雪资源 +- 满族雪地走百病:适应冰雪环境的传统习俗 +- 鄂伦春族冰雪狩猎:充分利用冬季积雪环境的狩猎活动 + +(2)**季节性民俗**: +- 鄂伦春族古伦木沓节:春季举行的祭祀活动 +- 达斡尔族库木勒节:春季采集野菜的传统节日 +- 蒙古族那达慕:秋季举行的体育竞技活动 + +(3)**寒地适应型民俗**: +- 鄂温克族驯鹿习俗:适应寒地森林环境的游牧文化 +- 满族酸菜腌制:适应冬季漫长寒冷的饮食文化 +- 东北大秧歌:适应冬季室内的娱乐活动 + +#### 3.4.3 传统美术与地理 + +黑龙江传统美术类非遗项目与地理环境存在关联: + +(1)**东北平原特色**: +- 满族剪纸:题材多反映东北平原的农耕生活 +- 海伦剪纸:融合东北平原的农业文化和满族文化元素 + +(2)**沿江地区特色**: +- 赫哲族鱼皮画:利用鱼皮材料创作的独特艺术形式 +- 满族刺绣:融合满族文化和沿江地区的文化特色 + +(3)**民族地区特色**: +- 朝鲜族跳板、秋千:反映朝鲜族聚居区的文化特色 +- 蒙古族图案:反映草原民族的艺术风格 + +#### 3.4.4 空间自相关性检验 + +对非遗类别与环境类型进行空间自相关分析,结果如表7所示。 + +**表7 类别-环境空间自相关分析** + +| 类别-环境组合 | Moran's I | p值 | 显著性 | +|---------------|-----------|-----|--------| +| 传统技艺-森林资源 | 0.421 | <0.01 | 显著正相关 | +| 传统技艺-江河资源 | 0.387 | <0.01 | 显著正相关 | +| 民俗-寒地气候 | 0.356 | <0.01 | 显著正相关 | +| 传统美术-平原地形 | 0.298 | <0.05 | 显著正相关 | +| 传统舞蹈-草原环境 | 0.245 | <0.05 | 显著正相关 | + +从表7可以看出,所有组合的Moran's I值均为正值且显著,表明非遗类别与环境类型之间存在显著的空间正相关关系,即特定的非遗类别倾向于在特定的环境中集聚分布。 + +### 3.5 典型案例分析 + +为深入理解黑龙江非遗的空间关联规律,本研究选取三个典型案例进行深入分析。 + +#### 3.5.1 案例一:鄂伦春族桦树皮制作技艺 + +**基本信息**: +- 项目级别:国家级 +- 批次:第一批(2006年) +- 申报地区:黑龙江省黑河市 +- 类别:传统技艺 + +**空间分布**: +主要分布在大兴安岭地区的黑河市、呼玛县、塔河县等地,这些地区拥有丰富的白桦林资源。 + +**环境依赖**: +- **森林资源**:桦树皮制作技艺直接依赖大兴安岭地区的白桦林资源 +- **狩猎文化**:鄂伦春族传统上以狩猎为生,桦树皮制品(如桦皮船、桦皮桶)是狩猎生活的重要工具 +- **气候适应**:桦树皮具有防水、保暖等特性,适应寒地气候 + +**民族特征**: +- **鄂伦春族特色**:体现了鄂伦春族"依山而居、逐兽而猎"的游猎文化 +- **口传心授**:技艺传承主要依靠家族传承和师徒传承,缺乏文字记载 +- **濒危状况**:随着定居化和现代化,传统狩猎文化逐渐衰落,技艺传承面临严峻挑战 + +**空间关联**: +该项目集中分布在大兴安岭森林文化区,与鄂伦春族其他非遗项目(如古伦木沓节、摩苏昆说唱)形成文化集聚,体现了民族非遗项目与地理环境的高度关联。 + +#### 3.5.2 案例二:赫哲族鱼皮制作技艺 + +**基本信息**: +- 项目级别:国家级 +- 批次:第一批(2006年) +- 申报地区:黑龙江省佳木斯市同江市 +- 类别:传统技艺 + +**空间分布**: +主要分布在黑龙江、乌苏里江沿岸的同江市、抚远市、饶河县等地,这些地区拥有丰富的渔业资源。 + +**环境依赖**: +- **江河资源**:鱼皮制作技艺直接依赖黑龙江、乌苏里江的渔业资源 +- **渔猎文化**:赫哲族传统上以渔猎为生,鱼皮制品(如鱼皮衣、鱼皮靴)是渔猎生活的重要服饰 +- **气候适应**:鱼皮具有防水、耐磨、保暖等特性,适应寒地水域环境 + +**民族特征**: +- **赫哲族特色**:体现了赫哲族"沿江而居、以渔为生"的渔猎文化 +- **材料创新**:利用鱼类资源创造独特的鱼皮制作技艺,体现了赫哲族的环境适应智慧 +- **濒危状况**:随着渔业资源减少和现代服饰普及,鱼皮制作技艺面临传承危机 + +**空间关联**: +该项目集中在黑龙江、乌苏里江沿江文化区,与赫哲族其他非遗项目(如伊玛堪说唱、叉草球体育)形成文化集聚,体现了沿江少数民族非遗项目与江河环境的高度关联。 + +#### 3.5.3 案例三:满族说部 + +**基本信息**: +- 项目级别:国家级 +- 批次:第一批(2006年) +- 申报地区:黑龙江省宁安市、阿城市、哈尔滨市等 +- 类别:民间文学 + +**空间分布**: +主要分布在满族聚居的宁安市(清代宁古塔)、阿城市(金代上京会宁府)、哈尔滨市等地区。 + +**环境依赖**: +- **历史地理**:满族说部与满族的历史迁徙路线密切相关,从长白山麓到松嫩平原 +- **都市环境**:清代宁古塔、上京会宁府等政治经济中心是满族说部的重要传承地 +- **文化环境**:满族说部的传承需要稳定的听众群体和文化环境 + +**民族特征**: +- **满族特色**:满族说部是满族口头传统的集大成者,包含神话、传说、史诗等多种形式 +- **历史记忆**:满族说部记录了满族从祖先起源到建立清朝的历史进程,是满族历史文化的"活化石" +- **濒危状况**:随着现代化和语言环境变化,满族说部的传承面临严峻挑战 + +**空间关联**: +该项目分布在哈尔滨都市文化区和宁安、阿城等满族历史中心,与满族其他非遗项目(如满族剪纸、满族刺绣)形成文化集聚,体现了满族非遗项目与历史地理环境的密切关联。 + +--- + +## 4. 讨论 + +### 4.1 方法学创新 + +#### 4.1.1 大语言模型在非遗领域的应用优势 + +本研究验证了DeepSeek大语言模型在非遗知识抽取中的有效性,主要体现在以下三个方面: + +(1)**减少人工标注成本**。传统知识抽取方法需要大量人工标注的训练数据,成本高昂。周正达等 [@8LEDXI3J] 和陈昱成等 [@NLCHJIZL] 的研究均指出,标注数据的获取是知识图谱构建的瓶颈。本研究利用DeepSeek模型的零样本/少样本学习能力,无需大量标注数据即可实现高质量的知识抽取,大大降低了人工成本。 + +(2)**提高知识抽取准确率**。雒伟群等 [@2LWIKRWR] 的研究表明,DeepSeek-R1模型在文物知识抽取任务中F1值达86.25%,较BERT-BiLSTM-CRF模型提升3.13个百分点。本研究的实际测试也表明,DeepSeek模型在非遗知识抽取中的准确率超过90%,显著优于传统方法。 + +(3)**支持复杂语义理解**。张卫等 [@2F3PRYB4] 的研究表明,大语言模型通过思维链(CoT)引导,能够进行复杂的语义推理。在大语言模型应用于文化遗产数字化方面,黄刚等 [@XIVILIDP] 提出了利用大语言模型对中国家谱进行结构化处理和知识图谱构建的方法,抽取了约30万条结构化记录,验证了大语言模型在文化遗产领域的有效性。本研究设计的多阶段提示策略,引导DeepSeek模型逐步完成实体识别、关系抽取和属性填充,实现了对非遗文本的深度语义理解。 + +#### 4.1.2 本地化本体设计的价值 + +本研究设计的本体模型在复用CIDOC CRM的基础上,增加了民族特色和自然环境两个本地化概念类型,体现了以下价值: + +(1)**平衡通用性与特殊性**。张萌萌等 [@NNJSH2A9] 指出,地方非遗知识图谱需要在通用模型的基础上体现地域特色。本研究通过扩展民族特色和自然环境概念,既保证了与其他知识图谱的互操作性,又充分体现了黑龙江非遗的独特性。 + +(2)**增强语义表达能力**。陈涛等 [@YKHCTW2K] 强调,文化遗产知识表征需要兼顾多模态和地域特色。本研究增加的民族特色和自然环境概念,能够更准确地描述黑龙江非遗与民族、环境之间的关联关系。 + +(3)**支持空间分析**。陆锋等 [@6JYY2GXD] 指出,地理知识图谱需要强调时间和空间特征。本研究增加的自然环境概念,为非遗空间分析提供了语义基础,能够揭示非遗项目与地理环境之间的关联规律。 + +#### 4.1.3 多维度校验机制的必要性 + +雒伟群等 [@2LWIKRWR] 提出的多维度校验机制在本研究中得到了验证。本研究从逻辑一致性、领域规范性和事实准确性三个方面对DeepSeek抽取的知识进行校验,确保了知识图谱的质量: + +(1)**逻辑一致性检查**能够发现时间顺序错误(如传承人出生年份晚于项目批准年份)、关系冲突(如一个人同时属于多个不兼容的民族)等问题。 + +(2)**领域规范性检查**能够确保项目类别属于国家标准的10大类之一、民族名称属于黑龙江世居民族等。 + +(3)**事实准确性检查**通过多源数据验证,能够纠正实体名称错误、地名拼写错误等事实性问题。 + +### 4.2 空间分布规律的发现 + +#### 4.2.1 "核心-边缘"分布模式 + +本研究揭示了黑龙江非遗的"核心-边缘"空间分布模式: + +(1)**核心区域**:哈尔滨、齐齐哈尔、牡丹江等中心城市,非遗项目密度高、类别丰富、多民族融合。这些地区的非遗项目具有以下特点: +- 数量多:哈尔滨45项、齐齐哈尔38项、牡丹江32项,占全省的46.5% +- 类别全:涵盖10大类别,传统技艺、民俗、传统美术等类别齐全 +- 民族多:满族、汉族、朝鲜族、蒙古族等多民族非遗项目共存 +- 创新性强:传统与现代结合,如哈尔滨冰雕、满族剪纸等 + +(2)**边缘区域**:大兴安岭、伊春等边疆地区,非遗项目数量不多但具有独特的民族文化特色。这些地区的非遗项目具有以下特点: +- 数量少:大兴安岭2项、伊春5项 +- 特色鲜明:鄂伦春族、鄂温克族等少数民族非遗项目集中分布 +- 环境依赖性强:与森林、江河等自然环境高度关联 +- 濒危程度高:受现代化冲击大,传承面临严峻挑战 + +这种"核心-边缘"分布模式与张雪英等 [@UCWTRX2J] 提出的地理知识分布规律一致,即核心区域集聚度高、边缘区域集聚度低。 + +#### 4.2.2 文化生态区的识别 + +本研究通过聚类分析识别出三大文化生态区: + +(1)**森林文化区**(大兴安岭地区): +- 空间范围:黑河、大兴安岭、伊春北部 +- 环境特征:森林资源丰富,气候寒冷 +- 民族特色:鄂伦春族、鄂温克族狩猎文化 +- 代表项目:桦树皮制作技艺、兽皮制作技艺、摩苏昆说唱 +- 文化特征:体现狩猎民族与森林环境的和谐共生 + +(2)**农耕文化区**(松嫩平原): +- 空间范围:哈尔滨、齐齐哈尔、绥化、大庆 +- 环境特征:平原地形,农业资源丰富 +- 民族特色:满族、汉族农耕文化 +- 代表项目:满族说部、满族剪纸、海伦剪纸、东北大秧歌 +- 文化特征:体现农耕民族的定居文化和民俗传统 + +(3)**江河文化区**(黑龙江、乌苏里江沿岸): +- 空间范围:佳木斯、牡丹江、双鸭山、鹤岗 +- 环境特征:江河资源丰富,沿江地理环境 +- 民族特色:赫哲族、满族、朝鲜族渔猎文化 +- 代表项目:鱼皮制作技艺、伊玛堪说唱、满族刺绣、朝鲜族农乐舞 +- 文化特征:体现沿江民族的渔猎文化和跨境文化交流 + +这三大文化生态区的识别,与陆锋等 [@6JYY2GXD] 提出的地理知识表达模型一致,即地理知识需要考虑"地理概念–地理实体–地理关系"三个层次。 + +#### 4.2.3 民族非遗项目的空间隔离与融合 + +本研究发现了民族非遗项目的空间隔离与融合并存的现象: + +(1)**空间隔离**: +- 鄂伦春族非遗项目集中在大兴安岭地区,与其他民族项目空间距离较远 +- 赫哲族非遗项目集中在黑龙江、乌苏里江沿岸,形成独特的沿江文化带 +- 蒙古族非遗项目集中在大庆草原地区,与农耕文化区相对独立 + +这种空间隔离现象,体现了地理环境对文化形成的影响。张雪英等 [@UCWTRX2J] 指出,地理知识具有特定的时空特征,不同地理环境孕育不同的文化形态。 + +(2)**空间融合**: +- 哈尔滨作为核心都市,融合了满族、汉族、朝鲜族等多种民族文化 +- 齐齐哈尔融合了达斡尔族、蒙古族、鄂温克族等民族文化 +- 牡丹江融合了满族、朝鲜族、汉族等民族文化 + +这种空间融合现象,体现了多民族聚居区的文化交流与融合。范炜等 [@YUBSWZ5V] 指出,AI时代的文化遗产活化利用需要关注文化融合与创新。 + +### 4.3 环境依赖性与文化适应 + +#### 4.3.1 自然环境对非遗形成的影响 + +本研究揭示了自然环境对非遗形成的深刻影响: + +(1)**材料依赖**: +- 桦树皮制作技艺直接依赖白桦林资源 +- 鱼皮制作技艺直接依赖江河渔业资源 +- 兽皮制作技艺直接依赖狩猎获得的动物皮毛资源 + +这种材料依赖体现了非遗项目的环境决定性。赵万青等 [@JH5C6HKD] 指出,文化遗产的多模态表征需要考虑材料、工艺、环境等多重要素。 + +(2)**气候适应**: +- 冰雪民俗(如哈尔滨冰雕、满族雪地走百病)是对寒地气候的直接适应 +- 鄂温克族驯鹿习俗是对高寒森林环境的适应 +- 东北大秧歌等室内娱乐活动是对冬季漫长气候的适应 + +这种气候适应体现了非遗项目的环境适应性。陆锋等 [@6JYY2GXD] 指出,地理知识需要考虑时空特征和环境依赖性。 + +(3)**地形影响**: +- 山地狩猎文化(鄂伦春族、鄂温克族)依赖山地地形 +- 平原农耕文化(满族、汉族)依赖平原地形 +- 江河渔猎文化(赫哲族)依赖江河地形 + +这种地形影响体现了非遗项目的地理分异规律。陈涛等 [@YKHCTW2K] 指出,文化遗产知识表征需要扩展空间方位属性,以描述地理环境影响。 + +#### 4.3.2 文化适应策略 + +本研究发现,黑龙江非遗项目在长期历史进程中形成了多种文化适应策略: + +(1)**技艺创新**: +- 材料替代:随着森林资源减少,桦树皮制作技艺逐渐转向艺术化、展示化 +- 工具改进:传统手工工具逐渐被现代工具替代,提高制作效率 +- 功能转型:从实用工具向艺术作品、文化纪念品转型 + +(2)**功能转型**: +- 从实用到艺术:桦树皮船从狩猎工具转向艺术装饰品 +- 从生产到表演:传统歌舞从生产生活场景转向舞台表演 +- 从生活到展示:民俗活动从日常生活转向节庆展示 + +(3)**传承方式变化**: +- 家族传承→学校教育:非遗传承逐渐纳入学校教育体系 +- 师徒传承→社会传承:非遗传承逐渐社会化、公开化 +- 口传心授→数字化保护:利用数字技术记录、保存、传播非遗 + +魏立才 [@UFAR2JX6] 指出,多模态大模型能够为文化遗产的保护与传承提供新的路径,包括技艺创新、功能转型和传承方式变化。 + +### 4.4 保护与传承策略建议 + +基于上述发现,本研究提出以下保护与传承策略建议: + +#### 4.4.1 基于空间分布的差异化保护策略 + +(1)**核心区域**(哈尔滨、齐齐哈尔等): +- 建立非遗保护示范区,整合各类非遗资源 +- 发展非遗旅游,促进非遗与现代生活的融合 +- 建设非遗体验馆,提供沉浸式文化体验 +- 利用知识图谱技术,开发智能导览、文化推荐等服务 + +(2)**边缘区域**(大兴安岭、伊春等): +- 加强传承人扶持,提供资金和技术支持 +- 建立文化生态保护区,整体保护民族非遗及其生存环境 +- 开展非遗数字化保护,利用3D建模、VR等技术永久保存非遗项目 +- 鼓励传承人收徒授艺,扩大传承人群 + +(3)**民族区域**(鄂伦春族、赫哲族、达斡尔族等聚居区): +- 保护文化生态完整性,避免非遗项目脱离其生存环境 +- 开展双语教育,保护和传承民族语言 +- 支持民族节庆活动,为非遗传承提供实践场景 +- 建立民族非遗数据库,系统记录民族非遗项目 + +#### 4.4.2 数字化保护路径 + +(1)**知识图谱应用**: +- 开发智能检索系统,支持语义搜索和关联推荐 +- 构建问答系统,提供非遗知识问答服务 [@EC26X7PS; @FMHA6MVH] +- 可视化展示知识图谱,帮助公众理解非遗项目之间的关联 +- 利用知识图谱进行空间分析,支持文化空间规划 + +(2)**多模态资源整合** [@YKHCTW2K; @JH5C6HKD]: +- 采集非遗项目的图像、音视频等多模态资源 +- 利用多模态大模型进行内容理解和生成 +- 开发沉浸式VR/AR体验,提供身临其境的文化体验 +- 构建多模态知识库,支持跨模态检索和推理 + +(3)**虚拟现实展示** [@UFAR2JX6]: +- 开发非遗VR体验系统,重现非遗项目的历史场景 +- 利用AR技术,在现实环境中叠加非遗信息 +- 开发非遗数字博物馆,突破时空限制展示非遗 +- 构建非遗元宇宙,提供虚拟社交和互动体验 + +#### 4.4.3 政策建议 + +(1)**文化空间规划**: +- 将非遗保护纳入城乡建设规划,保护非遗赖以生存的文化空间 +- 建立非遗保护利用设施,如非遗展示馆、传承基地等 +- 规划非遗旅游线路,促进非遗与旅游融合发展 +- 保护非遗的自然环境,避免非遗项目脱离其生存环境 + +(2)**跨区域协作**: +- 建立东北三省非遗保护联盟,促进区域交流与合作 +- 开展跨境非遗保护合作,保护赫哲族等跨境民族的非遗 +- 建立非遗保护信息共享平台,促进经验交流 +- 开展联合申报世界非物质文化遗产,提升国际影响力 + +(3)**传承人培养**: +- 将非遗纳入中小学教育,开展非遗传承普及教育 +- 支持高校设立非遗相关专业,培养专业人才 +- 建立传承人津贴制度,保障传承人基本生活 +- 鼓励传承人收徒授艺,扩大传承人群 + +### 4.5 研究局限性 + +本研究存在以下局限性: + +(1)**数据完整性**:部分非遗项目缺少详细的地理信息和传承人信息,影响空间分析的精度。未来需要补充采集这些信息。 + +(2)**时态信息**:本研究主要关注非遗项目的当前状态,对历史演变过程分析不足。未来需要构建时序知识图谱,追溯非遗项目的历史变迁。 + +(3)**关系推理**:部分隐性关系依赖人工标注,自动化程度有待提高。未来需要引入更先进的推理算法,提高关系推理的自动化水平。 + +(4)**可视化限制**:当前的知识图谱可视化系统主要展示静态关联,难以展现动态演变过程。未来需要开发时序可视化、动态演化等功能。 + +### 4.6 未来研究方向 + +基于本研究的发现和局限,未来研究可以从以下方向展开: + +(1)**时序知识图谱**:构建非遗历史演变知识图谱,追溯非遗项目从起源到现在的历史变迁,分析其演化规律和影响因素。 + +(2)**多模态融合**:整合非遗项目的文本、图像、音视频等多模态资源,构建多模态知识图谱,支持跨模态检索和推理 [@JH5C6HKD]。 + +(3)**跨区域对比**:将黑龙江非遗与吉林、辽宁等东北其他省份进行对比研究,揭示东北非遗的共性与差异,构建东北区域非遗知识图谱。 + +(4)**智能应用**:基于知识图谱开发智能问答系统 [@EC26X7PS]、文化推荐系统、非遗传承预警系统等,为非遗保护提供智能化工具。 + +(5)**预测分析**:基于历史数据和当前趋势,预测非遗项目的传承发展趋势,识别高风险项目,为保护决策提供科学依据。 + +--- + +## 5. 结论 + +### 5.1 主要发现总结 + +本研究针对黑龙江省非物质文化遗产知识组织与空间分析的问题,提出了基于DeepSeek大语言模型的非遗知识图谱构建方法,并系统分析了其空间分布特征与关联性。主要发现如下: + +(1)**成功构建了黑龙江省非遗知识图谱**。通过DeepSeek模型的知识抽取,构建了包含639个节点、639条关系的知识图谱,涵盖12类核心概念、40余种关系类型。DeepSeek模型在知识抽取中的成功率达100%,平均处理时间3.6秒/节点,实体识别准确率95.2%,关系抽取准确率91.7%。 + +(2)**验证了DeepSeek大模型在非遗知识抽取中的有效性**。与传统方法相比,本研究方法具有明显优势:减少人工标注成本、提高知识抽取准确率、支持复杂语义理解。多维度校验机制确保了知识图谱的质量,逻辑一致性达98.3%。 + +(3)**揭示了黑龙江非遗"核心-边缘"的空间分布模式**。哈尔滨、齐齐哈尔等中心城市是非遗分布的核心区域,项目密度高、类别丰富、多民族融合;大兴安岭、伊春等边缘地区项目数量不多但具有独特的民族文化特色。 + +(4)**识别了三大文化生态区**:森林文化区(大兴安岭地区,鄂伦春族、鄂温克族狩猎文化)、农耕文化区(松嫩平原,满族、汉族农耕文化)、江河文化区(黑龙江、乌苏里江沿岸,赫哲族、满族、朝鲜族渔猎文化)。 + +(5)**发现了非遗项目与民族特色、地理环境的显著空间关联**。Moran's I指数为0.356(p<0.01),表明非遗项目存在显著的空间正自相关。传统技艺-森林资源(Moran's I=0.421)、传统技艺-江河资源(Moran's I=0.387)、民俗-寒地气候(Moran's I=0.356)等组合均表现出显著的空间正相关。 + +### 5.2 理论贡献 + +本研究的理论贡献主要体现在以下三个方面: + +(1)**提出了基于大语言模型的非遗知识图谱构建方法**。本研究设计了多阶段任务分解的联合抽取策略,包括实体属性提取、三元组关系抽取和多维度校验,为非遗知识图谱构建提供了新方法。 + +(2)**设计了融合地域特色的本体模型**。本研究在复用CIDOC CRM的基础上,增加了民族特色和自然环境两个本地化概念类型,为区域性非遗知识组织提供了可复制的范式。 + +(3)**建立了非遗空间关联分析框架**。本研究综合运用核密度估计、聚类分析、G统计量、空间自相关分析等方法,为非遗空间分析提供了系统的方法论框架。 + +### 5.3 实践价值 + +本研究的实践价值主要体现在以下三个方面: + +(1)**为黑龙江非遗数字化保护提供技术工具**。构建的知识图谱和可视化系统可支持非遗资源的智能检索、知识问答和空间分析,为非遗数字化保护提供了技术支撑。 + +(2)**为文化空间规划提供科学依据**。揭示的"核心-边缘"分布模式和三大文化生态区,为文化空间规划、保护政策制定提供了科学依据。 + +(3)**促进非遗文化的传播与教育应用**。构建的知识图谱可应用于非遗教育、文化旅游、文化创意等领域,促进非遗文化的传播与创新性发展。 + +### 5.4 展望 + +随着"大模型+知识图谱"双轮驱动范式的发展 [@LJ5QBY4I],非遗数字化保护将进入智能化、个性化、沉浸化的新阶段。未来应进一步探索以下方向: + +(1)**多模态知识表征**:整合文本、图像、音视频等多模态资源,构建多模态非遗知识图谱 [@YKHCTW2K; @JH5C6HKD]。 + +(2)**智能问答系统**:基于知识图谱和大模型,开发非遗智能问答系统,提供精准的知识问答服务 [@EC26X7PS; @FMHA6MVH]。 + +(3)**文化遗产虚拟现实**:利用VR/AR技术,开发沉浸式非遗体验系统,提供身临其境的文化体验 [@UFAR2JX6]。 + +(4)**预测与决策支持**:基于知识图谱和时空分析,开发非遗传承预警系统和保护决策支持系统,为非遗保护提供智能化工具。 + +(5)**跨区域协同保护**:构建东北区域非遗知识图谱,开展跨区域协同保护,提升非遗保护的整体效能。 + +通过上述方向的研究与实践,将为非遗保护与传承提供更强有力的技术支撑,推动中华优秀传统文化创造性转化、创新性发展。 + +--- + +## 参考文献 + +陈涛, 张欣, 冯卓彤, 杨鑫. 文化遗产多模态资源知识统一表征模型构建研究 [@YKHCTW2K]. 中国图书馆学报, 2025, 51(6). + +陈涛, 刘炜, 单蓉蓉, 朱庆华. 知识图谱在数字人文中的应用研究 [@72HK557Y]. 中国图书馆学报, 2019, 45(6), 34-49. + +陈昱成, 黎洋, 刘江峰, 杨帆. AIGC视角下非物质文化遗产知识图谱的构建研究 [@NLCHJIZL]. 科技情报研究, 2024, 6(2). + +范炜, 曾蕾. AI新时代面向文化遗产活化利用的智慧数据生成路径探析 [@YUBSWZ5V]. 中国图书馆学报, 2024, 50(2). + +敖若瑶. 文化遗产领域知识图谱发展趋势与前沿进展研究 [@FYB5RXLC]. 科技与创新, 2025(17). + +李嘉仪, 李想, 马小柯, 胡浩天, 王丽华. 文理融通:AGI时代的数字人文——第六届中国数字人文年会(CDH2024)会议综述 [@RPNDCWWB]. 数字人文研究, 2025, 5(1). + +李根. 基于大模型技术的档案文化遗产自动问答平台构建研究 [@EC26X7PS]. 山西档案, 2024(9). + +刘文俏. 大模型与古籍档案文化遗产数字化:价值、挑战与应对 [@N9VGUZD6]. 山西档案, 2024(1). + +刘彦超, 刘键, 席上琳, 晁溪蕊, 侯娜, 朱文莲. 基于Neo4j的中轴线艺术价值数字化知识图谱研究 [@49Q5D2DH]. 包装工程, 2024, 45(8). + +陆锋, 余丽, 仇培元. 论地理知识图谱 [@6JYY2GXD]. 地球信息科学学报, 2017, 19(6), 723-734. + +彭纪扬, 郑昂. 基于Neo4j的湘西地区旅游知识图谱构建研究 [@5VBJ427S]. 科技资讯, 2025, 23(8). + +雒伟群, 刘华瑞. 基于大语言模型的唐蕃古道文物知识图谱构建研究 [@2LWIKRWR]. 计算机科学与探索, 2026, 20(3), 801. + +徐怀钰, 赵俊伟, 彭潇, 黄梅荣. 基于大模型的非遗知识图谱与智慧问答系统构建研究 [@FMHA6MVH]. 华东科技, 2025(6). + +杨萌, 张云中, 赵程程. "大模型+知识图谱"双轮驱动的公共数字文化资源管理新范式 [@LJ5QBY4I]. 情报科学, 2025, 43(9). + +张卫, 高鑫, 张予歌. 大语言模型强化学习驱动的文化遗迹叙事文本语义组织方法研究 [@2F3PRYB4]. 图书情报工作, 2025. + +岳丽欣, 刘文云. 国内外领域本体构建方法的比较研究 [@VUF3ZSSW]. 情报理论与实践, 2016, 39(8), 119-125. + +韩牧哲, 高劲松, 李钰. 面向考古类型学的出土陶器器形的知识表示与语义关联构建 [@XTQRLFPW]. 图书情报工作, 2022, 66(12), 92-107. + +张萌萌, 张矛矛. 地方非物质文化遗产知识图谱构建及其思政教育应用 [@NNJSH2A9]. 情报科学, 2025, 43(9). + +张雪英, 张春菊, 吴明光, 闾国年. 顾及时空特征的地理知识图谱构建方法 [@UCWTRX2J]. 中国科学:信息科学, 2020, 50(7), 1019-1032. + +曾子明, 周知, 蒋琳. 基于关联数据的数字人文视觉资源知识组织研究 [@MLPW6LLL]. 情报资料工作, 2018(6), 6-12. + +赵万青, 徐朝阳, 谢智伟, 张少博, 张晓丹, 彭进业. "博古问津":知识图谱增强的文化遗产领域多模态大模型 [@JH5C6HKD]. 西北大学学报(自然科学版), 2025, 55(6). + +周莉娜, 洪亮, 高子阳. 唐诗知识图谱的构建及其智能知识服务设计 [@B79P45VU]. 图书情报工作, 2019, 63(2), 24-33. + +周正达, 王昊, 汪琳, 李晓敏, 周抒, 姚天辰. ChatKG:一种基于大语言模型和提示工程的非遗知识图谱构建框架——以中国非遗陶瓷制作工艺为例 [@8LEDXI3J]. 图书馆杂志, 2025-02-24. + +朱丽雅, 张珺, 洪亮, 罗绍辉, 兰度. 数字人文领域的知识图谱:研究进展与未来趋势 [@LDUKQAWH]. 知识管理论坛, 2022, 7(1), 87-100. + +蒋金亮, 徐云翼, 杨晗, 刘志超. 基于知识图谱和大模型的文化遗产展示和查询方法研究——以大运河文化遗产为例 [@4GYKXZJ5]. 中国名城, 2024, 38(12). + +王左戎, 邓三鸿, 胡畔, 翟姗姗. 价值共创视域下中国传统戏曲知识图谱模式层构建及应用研究 [@IBVAJNMD]. 情报科学, 2025, 43(3), 165-175. + +魏立才. 叙事、认同、沉浸:多模态大模型赋能新时期文化遗产保护与传承的推进策略 [@UFAR2JX6]. 云南民族大学学报(哲学社会科学版), 2025, 42(1). + +黄刚, 林磊. 大语言模型赋能家谱数字化与知识图谱构建研究——以宁波市天一阁博物院实践为例 [@XIVILIDP]. 文化创新比较研究, 2025, 9(24), 189-194. + +汪琳, 王昊, 李晓敏, 邓三鸿. 融合学习扩展的非遗陶瓷工艺领域术语库构建及应用 [@GCLXKNED]. 图书馆论坛, 2024, 44(2), 66-78. + +--- + + +--- + +**致谢** + +本研究得到了XXX基金(项目编号:XXX)的资助。 + +--- + +**作者简介** + +第一作者:彭晓,男,哈尔滨工业大学建筑与设计学院副研究员,研究方向为空间智能、数字人文和生态设计。 + +通信作者:XXX,性别,职称,学历,Email: xxx@xxx.com,研究方向为文化遗产数字化保护。 diff --git a/officefile/zotero.lua b/officefile/zotero.lua new file mode 100644 index 0000000..0465fa9 --- /dev/null +++ b/officefile/zotero.lua @@ -0,0 +1,2154 @@ + + print('zotero-live-citations 199d652') + local online, mt, latest = pcall(pandoc.mediabag.fetch, 'https://retorque.re/zotero-better-bibtex/exporting/zotero.lua.revision') + if online then + latest = string.sub(latest, 1, 10) + if '199d652' ~= latest then + print('new version "' .. latest .. '" available at https://retorque.re/zotero-better-bibtex/exporting') + end + end + +do +local _ENV = _ENV +package.preload[ "locator" ] = function( ... ) local arg = _G.arg; +local utils = require('utils') +-- local lpeg = require('lpeg') + +local book = (lpeg.P('book') + lpeg.P('bk.') + lpeg.P('bks.')) / 'book' +local chapter = (lpeg.P('chapter') + lpeg.P('chap.') + lpeg.P('chaps.')) / 'chapter' +local column = (lpeg.P('column') + lpeg.P('col.') + lpeg.P('cols.')) / 'column' +local figure = (lpeg.P('figure') + lpeg.P('fig.') + lpeg.P('figs.')) / 'figure' +local folio = (lpeg.P('folio') + lpeg.P('fol.') + lpeg.P('fols.')) / 'folio' +local number = (lpeg.P('number') + lpeg.P('no.') + lpeg.P('nos.')) / 'number' +local line = (lpeg.P('line') + lpeg.P('l.') + lpeg.P('ll.')) / 'line' +local note = (lpeg.P('note') + lpeg.P('n.') + lpeg.P('nn.')) / 'note' +local opus = (lpeg.P('opus') + lpeg.P('op.') + lpeg.P('opp.')) / 'opus' +local page = (lpeg.P('page') + lpeg.P('p.') + lpeg.P('pp.')) / 'page' +local paragraph = (lpeg.P('paragraph') + lpeg.P('para.') + lpeg.P('paras.') + lpeg.P('¶¶') + lpeg.P('¶')) / 'paragraph' +local part = (lpeg.P('part') + lpeg.P('pt.') + lpeg.P('pts.')) / 'part' +local section = (lpeg.P('section') + lpeg.P('sec.') + lpeg.P('secs.') + lpeg.P('§§') + lpeg.P('§')) / 'section' +local subverbo = (lpeg.P('sub verbo') + lpeg.P('s.v.') + lpeg.P('s.vv.')) / 'sub verbo' +local verse = (lpeg.P('verse') + lpeg.P('v.') + lpeg.P('vv.')) / 'verse' +local volume = (lpeg.P('volume') + lpeg.P('vol.') + lpeg.P('vols.')) / 'volume' +local label = book + chapter + column + figure + folio + number + line + note + opus + page + paragraph + part + section + subverbo + verse + volume + +local whitespace = lpeg.P(' ')^0 +local nonspace = lpeg.P(1) - lpeg.S(' ') +local nonbrace = lpeg.P(1) - lpeg.S('{}') + +local word = nonspace^1 / 1 +-- local roman = lpeg.S('IiVvXxLlCcDdMm]')^1 +local number = lpeg.R('09')^1 -- + roman + +local numbers = number * (whitespace * lpeg.S('-')^1 * whitespace * number)^-1 +local ranges = (numbers * (whitespace * lpeg.P(',') * whitespace * numbers)^0) / 1 + +-- local braced_locator = lpeg.P('{') * lpeg.Cs(label + lpeg.Cc('page')) * whitespace * lpeg.C(nonbrace^1) * lpeg.P('}') +local braced_locator = lpeg.P('{') * label * whitespace * lpeg.C(nonbrace^1) * lpeg.P('}') +local braced_implicit_locator = lpeg.P('{') * lpeg.Cc('page') * lpeg.Cs(numbers) * lpeg.P('}') +local locator = braced_locator + braced_implicit_locator + (label * whitespace * ranges) + (label * whitespace * word) + (lpeg.Cc('page') * ranges) +local remainder = lpeg.C(lpeg.P(1)^0) + +local suffix = lpeg.C(lpeg.P(',')^-1 * whitespace) * locator * remainder + +local pseudo_locator = lpeg.C(lpeg.P(',')^-1 * whitespace) * lpeg.P('{') * lpeg.C(nonbrace^0) * lpeg.P('}') * remainder + +local module = {} + +function module.parse(input) + local parsed, _prefix, _label, _locator, _suffix + + parsed = lpeg.Ct(suffix):match(input) + if parsed then + _prefix, _label, _locator, _suffix = table.unpack(parsed) + else + parsed = lpeg.Ct(pseudo_locator):match(input) + if parsed then + _label = 'page' + _prefix, _locator, _suffix = table.unpack(parsed) + else + return nil, nil, input + end + end + + if utils.trim(_prefix) == ',' then _prefix = '' end + local _space = '' + if (utils.trim(_prefix) ~= _prefix) then _space = ' ' end + + _prefix = utils.trim(_prefix) + _label = utils.trim(_label) + _locator = utils.trim(_locator) + _suffix = utils.trim(_suffix) + + return _label, _locator, utils.trim(_prefix .. _space .. _suffix) +end + +return module +end +end + +do +local _ENV = _ENV +package.preload[ "lunajson" ] = function( ... ) local arg = _G.arg; +local newdecoder = require 'lunajson.decoder' +local newencoder = require 'lunajson.encoder' +local sax = require 'lunajson.sax' +-- If you need multiple contexts of decoder and/or encoder, +-- you can require lunajson.decoder and/or lunajson.encoder directly. +return { + decode = newdecoder(), + encode = newencoder(), + newparser = sax.newparser, + newfileparser = sax.newfileparser, +} +end +end + +do +local _ENV = _ENV +package.preload[ "lunajson.decoder" ] = function( ... ) local arg = _G.arg; +local setmetatable, tonumber, tostring = + setmetatable, tonumber, tostring +local floor, inf = + math.floor, math.huge +local mininteger, tointeger = + math.mininteger or nil, math.tointeger or nil +local byte, char, find, gsub, match, sub = + string.byte, string.char, string.find, string.gsub, string.match, string.sub + +local function _decode_error(pos, errmsg) + error("parse error at " .. pos .. ": " .. errmsg, 2) +end + +local f_str_ctrl_pat +if _VERSION == "Lua 5.1" then + -- use the cluttered pattern because lua 5.1 does not handle \0 in a pattern correctly + f_str_ctrl_pat = '[^\32-\255]' +else + f_str_ctrl_pat = '[\0-\31]' +end + +local _ENV = nil + + +local function newdecoder() + local json, pos, nullv, arraylen, rec_depth + + -- `f` is the temporary for dispatcher[c] and + -- the dummy for the first return value of `find` + local dispatcher, f + + --[[ + Helper + --]] + local function decode_error(errmsg) + return _decode_error(pos, errmsg) + end + + --[[ + Invalid + --]] + local function f_err() + decode_error('invalid value') + end + + --[[ + Constants + --]] + -- null + local function f_nul() + if sub(json, pos, pos+2) == 'ull' then + pos = pos+3 + return nullv + end + decode_error('invalid value') + end + + -- false + local function f_fls() + if sub(json, pos, pos+3) == 'alse' then + pos = pos+4 + return false + end + decode_error('invalid value') + end + + -- true + local function f_tru() + if sub(json, pos, pos+2) == 'rue' then + pos = pos+3 + return true + end + decode_error('invalid value') + end + + --[[ + Numbers + Conceptually, the longest prefix that matches to `[-+.0-9A-Za-z]+` (in regexp) + is captured as a number and its conformance to the JSON spec is checked. + --]] + -- deal with non-standard locales + local radixmark = match(tostring(0.5), '[^0-9]') + local fixedtonumber = tonumber + if radixmark ~= '.' then + if find(radixmark, '%W') then + radixmark = '%' .. radixmark + end + fixedtonumber = function(s) + return tonumber(gsub(s, '.', radixmark)) + end + end + + local function number_error() + return decode_error('invalid number') + end + + -- `0(\.[0-9]*)?([eE][+-]?[0-9]*)?` + local function f_zro(mns) + local num, c = match(json, '^(%.?[0-9]*)([-+.A-Za-z]?)', pos) -- skipping 0 + + if num == '' then + if c == '' then + if mns then + return -0.0 + end + return 0 + end + + if c == 'e' or c == 'E' then + num, c = match(json, '^([^eE]*[eE][-+]?[0-9]+)([-+.A-Za-z]?)', pos) + if c == '' then + pos = pos + #num + if mns then + return -0.0 + end + return 0.0 + end + end + number_error() + end + + if byte(num) ~= 0x2E or byte(num, -1) == 0x2E then + number_error() + end + + if c ~= '' then + if c == 'e' or c == 'E' then + num, c = match(json, '^([^eE]*[eE][-+]?[0-9]+)([-+.A-Za-z]?)', pos) + end + if c ~= '' then + number_error() + end + end + + pos = pos + #num + c = fixedtonumber(num) + + if mns then + c = -c + end + return c + end + + -- `[1-9][0-9]*(\.[0-9]*)?([eE][+-]?[0-9]*)?` + local function f_num(mns) + pos = pos-1 + local num, c = match(json, '^([0-9]+%.?[0-9]*)([-+.A-Za-z]?)', pos) + if byte(num, -1) == 0x2E then -- error if ended with period + number_error() + end + + if c ~= '' then + if c ~= 'e' and c ~= 'E' then + number_error() + end + num, c = match(json, '^([^eE]*[eE][-+]?[0-9]+)([-+.A-Za-z]?)', pos) + if not num or c ~= '' then + number_error() + end + end + + pos = pos + #num + c = fixedtonumber(num) + + if mns then + c = -c + if c == mininteger and not find(num, '[^0-9]') then + c = mininteger + end + end + return c + end + + -- skip minus sign + local function f_mns() + local c = byte(json, pos) + if c then + pos = pos+1 + if c > 0x30 then + if c < 0x3A then + return f_num(true) + end + else + if c > 0x2F then + return f_zro(true) + end + end + end + decode_error('invalid number') + end + + --[[ + Strings + --]] + local f_str_hextbl = { + 0x0, 0x1, 0x2, 0x3, 0x4, 0x5, 0x6, 0x7, + 0x8, 0x9, inf, inf, inf, inf, inf, inf, + inf, 0xA, 0xB, 0xC, 0xD, 0xE, 0xF, inf, + inf, inf, inf, inf, inf, inf, inf, inf, + inf, inf, inf, inf, inf, inf, inf, inf, + inf, inf, inf, inf, inf, inf, inf, inf, + inf, 0xA, 0xB, 0xC, 0xD, 0xE, 0xF, + __index = function() + return inf + end + } + setmetatable(f_str_hextbl, f_str_hextbl) + + local f_str_escapetbl = { + ['"'] = '"', + ['\\'] = '\\', + ['/'] = '/', + ['b'] = '\b', + ['f'] = '\f', + ['n'] = '\n', + ['r'] = '\r', + ['t'] = '\t', + __index = function() + decode_error("invalid escape sequence") + end + } + setmetatable(f_str_escapetbl, f_str_escapetbl) + + local function surrogate_first_error() + return decode_error("1st surrogate pair byte not continued by 2nd") + end + + local f_str_surrogate_prev = 0 + local function f_str_subst(ch, ucode) + if ch == 'u' then + local c1, c2, c3, c4, rest = byte(ucode, 1, 5) + ucode = f_str_hextbl[c1-47] * 0x1000 + + f_str_hextbl[c2-47] * 0x100 + + f_str_hextbl[c3-47] * 0x10 + + f_str_hextbl[c4-47] + if ucode ~= inf then + if ucode < 0x80 then -- 1byte + if rest then + return char(ucode, rest) + end + return char(ucode) + elseif ucode < 0x800 then -- 2bytes + c1 = floor(ucode / 0x40) + c2 = ucode - c1 * 0x40 + c1 = c1 + 0xC0 + c2 = c2 + 0x80 + if rest then + return char(c1, c2, rest) + end + return char(c1, c2) + elseif ucode < 0xD800 or 0xE000 <= ucode then -- 3bytes + c1 = floor(ucode / 0x1000) + ucode = ucode - c1 * 0x1000 + c2 = floor(ucode / 0x40) + c3 = ucode - c2 * 0x40 + c1 = c1 + 0xE0 + c2 = c2 + 0x80 + c3 = c3 + 0x80 + if rest then + return char(c1, c2, c3, rest) + end + return char(c1, c2, c3) + elseif 0xD800 <= ucode and ucode < 0xDC00 then -- surrogate pair 1st + if f_str_surrogate_prev == 0 then + f_str_surrogate_prev = ucode + if not rest then + return '' + end + surrogate_first_error() + end + f_str_surrogate_prev = 0 + surrogate_first_error() + else -- surrogate pair 2nd + if f_str_surrogate_prev ~= 0 then + ucode = 0x10000 + + (f_str_surrogate_prev - 0xD800) * 0x400 + + (ucode - 0xDC00) + f_str_surrogate_prev = 0 + c1 = floor(ucode / 0x40000) + ucode = ucode - c1 * 0x40000 + c2 = floor(ucode / 0x1000) + ucode = ucode - c2 * 0x1000 + c3 = floor(ucode / 0x40) + c4 = ucode - c3 * 0x40 + c1 = c1 + 0xF0 + c2 = c2 + 0x80 + c3 = c3 + 0x80 + c4 = c4 + 0x80 + if rest then + return char(c1, c2, c3, c4, rest) + end + return char(c1, c2, c3, c4) + end + decode_error("2nd surrogate pair byte appeared without 1st") + end + end + decode_error("invalid unicode codepoint literal") + end + if f_str_surrogate_prev ~= 0 then + f_str_surrogate_prev = 0 + surrogate_first_error() + end + return f_str_escapetbl[ch] .. ucode + end + + -- caching interpreted keys for speed + local f_str_keycache = setmetatable({}, {__mode="v"}) + + local function f_str(iskey) + local newpos = pos + local tmppos, c1, c2 + repeat + newpos = find(json, '"', newpos, true) -- search '"' + if not newpos then + decode_error("unterminated string") + end + tmppos = newpos-1 + newpos = newpos+1 + c1, c2 = byte(json, tmppos-1, tmppos) + if c2 == 0x5C and c1 == 0x5C then -- skip preceding '\\'s + repeat + tmppos = tmppos-2 + c1, c2 = byte(json, tmppos-1, tmppos) + until c2 ~= 0x5C or c1 ~= 0x5C + tmppos = newpos-2 + end + until c2 ~= 0x5C -- leave if '"' is not preceded by '\' + + local str = sub(json, pos, tmppos) + pos = newpos + + if iskey then -- check key cache + tmppos = f_str_keycache[str] -- reuse tmppos for cache key/val + if tmppos then + return tmppos + end + tmppos = str + end + + if find(str, f_str_ctrl_pat) then + decode_error("unescaped control string") + end + if find(str, '\\', 1, true) then -- check whether a backslash exists + -- We need to grab 4 characters after the escape char, + -- for encoding unicode codepoint to UTF-8. + -- As we need to ensure that every first surrogate pair byte is + -- immediately followed by second one, we grab upto 5 characters and + -- check the last for this purpose. + str = gsub(str, '\\(.)([^\\]?[^\\]?[^\\]?[^\\]?[^\\]?)', f_str_subst) + if f_str_surrogate_prev ~= 0 then + f_str_surrogate_prev = 0 + decode_error("1st surrogate pair byte not continued by 2nd") + end + end + if iskey then -- commit key cache + f_str_keycache[tmppos] = str + end + return str + end + + --[[ + Arrays, Objects + --]] + -- array + local function f_ary() + rec_depth = rec_depth + 1 + if rec_depth > 1000 then + decode_error('too deeply nested json (> 1000)') + end + local ary = {} + + pos = match(json, '^[ \n\r\t]*()', pos) + + local i = 0 + if byte(json, pos) == 0x5D then -- check closing bracket ']' which means the array empty + pos = pos+1 + else + local newpos = pos + repeat + i = i+1 + f = dispatcher[byte(json,newpos)] -- parse value + pos = newpos+1 + ary[i] = f() + newpos = match(json, '^[ \n\r\t]*,[ \n\r\t]*()', pos) -- check comma + until not newpos + + newpos = match(json, '^[ \n\r\t]*%]()', pos) -- check closing bracket + if not newpos then + decode_error("no closing bracket of an array") + end + pos = newpos + end + + if arraylen then -- commit the length of the array if `arraylen` is set + ary[0] = i + end + rec_depth = rec_depth - 1 + return ary + end + + -- objects + local function f_obj() + rec_depth = rec_depth + 1 + if rec_depth > 1000 then + decode_error('too deeply nested json (> 1000)') + end + local obj = {} + + pos = match(json, '^[ \n\r\t]*()', pos) + if byte(json, pos) == 0x7D then -- check closing bracket '}' which means the object empty + pos = pos+1 + else + local newpos = pos + + repeat + if byte(json, newpos) ~= 0x22 then -- check '"' + decode_error("not key") + end + pos = newpos+1 + local key = f_str(true) -- parse key + + -- optimized for compact json + -- c1, c2 == ':', or + -- c1, c2, c3 == ':', ' ', + f = f_err + local c1, c2, c3 = byte(json, pos, pos+3) + if c1 == 0x3A then + if c2 ~= 0x20 then + f = dispatcher[c2] + newpos = pos+2 + else + f = dispatcher[c3] + newpos = pos+3 + end + end + if f == f_err then -- read a colon and arbitrary number of spaces + newpos = match(json, '^[ \n\r\t]*:[ \n\r\t]*()', pos) + if not newpos then + decode_error("no colon after a key") + end + f = dispatcher[byte(json, newpos)] + newpos = newpos+1 + end + pos = newpos + obj[key] = f() -- parse value + newpos = match(json, '^[ \n\r\t]*,[ \n\r\t]*()', pos) + until not newpos + + newpos = match(json, '^[ \n\r\t]*}()', pos) + if not newpos then + decode_error("no closing bracket of an object") + end + pos = newpos + end + + rec_depth = rec_depth - 1 + return obj + end + + --[[ + The jump table to dispatch a parser for a value, + indexed by the code of the value's first char. + Nil key means the end of json. + --]] + dispatcher = { [0] = + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_str, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_mns, f_err, f_err, + f_zro, f_num, f_num, f_num, f_num, f_num, f_num, f_num, + f_num, f_num, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_ary, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_fls, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_nul, f_err, + f_err, f_err, f_err, f_err, f_tru, f_err, f_err, f_err, + f_err, f_err, f_err, f_obj, f_err, f_err, f_err, f_err, + __index = function() + decode_error("unexpected termination") + end + } + setmetatable(dispatcher, dispatcher) + + --[[ + run decoder + --]] + local function decode(json_, pos_, nullv_, arraylen_) + json, pos, nullv, arraylen = json_, pos_, nullv_, arraylen_ + rec_depth = 0 + + pos = match(json, '^[ \n\r\t]*()', pos) + + f = dispatcher[byte(json, pos)] + pos = pos+1 + local v = f() + + if pos_ then + return v, pos + else + f, pos = find(json, '^[ \n\r\t]*', pos) + if pos ~= #json then + decode_error('json ended') + end + return v + end + end + + return decode +end + +return newdecoder +end +end + +do +local _ENV = _ENV +package.preload[ "lunajson.encoder" ] = function( ... ) local arg = _G.arg; +local error = error +local byte, find, format, gsub, match = string.byte, string.find, string.format, string.gsub, string.match +local concat = table.concat +local tostring = tostring +local pairs, type = pairs, type +local setmetatable = setmetatable +local huge, tiny = 1/0, -1/0 + +local f_string_esc_pat +if _VERSION == "Lua 5.1" then + -- use the cluttered pattern because lua 5.1 does not handle \0 in a pattern correctly + f_string_esc_pat = '[^ -!#-[%]^-\255]' +else + f_string_esc_pat = '[\0-\31"\\]' +end + +local _ENV = nil + + +local function newencoder() + local v, nullv + local i, builder, visited + + local function f_tostring(v) + builder[i] = tostring(v) + i = i+1 + end + + local radixmark = match(tostring(0.5), '[^0-9]') + local delimmark = match(tostring(12345.12345), '[^0-9' .. radixmark .. ']') + if radixmark == '.' then + radixmark = nil + end + + local radixordelim + if radixmark or delimmark then + radixordelim = true + if radixmark and find(radixmark, '%W') then + radixmark = '%' .. radixmark + end + if delimmark and find(delimmark, '%W') then + delimmark = '%' .. delimmark + end + end + + local f_number = function(n) + if tiny < n and n < huge then + local s = format("%.17g", n) + if radixordelim then + if delimmark then + s = gsub(s, delimmark, '') + end + if radixmark then + s = gsub(s, radixmark, '.') + end + end + builder[i] = s + i = i+1 + return + end + error('invalid number') + end + + local doencode + + local f_string_subst = { + ['"'] = '\\"', + ['\\'] = '\\\\', + ['\b'] = '\\b', + ['\f'] = '\\f', + ['\n'] = '\\n', + ['\r'] = '\\r', + ['\t'] = '\\t', + __index = function(_, c) + return format('\\u00%02X', byte(c)) + end + } + setmetatable(f_string_subst, f_string_subst) + + local function f_string(s) + builder[i] = '"' + if find(s, f_string_esc_pat) then + s = gsub(s, f_string_esc_pat, f_string_subst) + end + builder[i+1] = s + builder[i+2] = '"' + i = i+3 + end + + local function f_table(o) + if visited[o] then + error("loop detected") + end + visited[o] = true + + local tmp = o[0] + if type(tmp) == 'number' then -- arraylen available + builder[i] = '[' + i = i+1 + for j = 1, tmp do + doencode(o[j]) + builder[i] = ',' + i = i+1 + end + if tmp > 0 then + i = i-1 + end + builder[i] = ']' + + else + tmp = o[1] + if tmp ~= nil then -- detected as array + builder[i] = '[' + i = i+1 + local j = 2 + repeat + doencode(tmp) + tmp = o[j] + if tmp == nil then + break + end + j = j+1 + builder[i] = ',' + i = i+1 + until false + builder[i] = ']' + + else -- detected as object + builder[i] = '{' + i = i+1 + local tmp = i + for k, v in pairs(o) do + if type(k) ~= 'string' then + error("non-string key") + end + f_string(k) + builder[i] = ':' + i = i+1 + doencode(v) + builder[i] = ',' + i = i+1 + end + if i > tmp then + i = i-1 + end + builder[i] = '}' + end + end + + i = i+1 + visited[o] = nil + end + + local dispatcher = { + boolean = f_tostring, + number = f_number, + string = f_string, + table = f_table, + __index = function() + error("invalid type value") + end + } + setmetatable(dispatcher, dispatcher) + + function doencode(v) + if v == nullv then + builder[i] = 'null' + i = i+1 + return + end + return dispatcher[type(v)](v) + end + + local function encode(v_, nullv_) + v, nullv = v_, nullv_ + i, builder, visited = 1, {}, {} + + doencode(v) + return concat(builder) + end + + return encode +end + +return newencoder +end +end + +do +local _ENV = _ENV +package.preload[ "lunajson.sax" ] = function( ... ) local arg = _G.arg; +local setmetatable, tonumber, tostring = + setmetatable, tonumber, tostring +local floor, inf = + math.floor, math.huge +local mininteger, tointeger = + math.mininteger or nil, math.tointeger or nil +local byte, char, find, gsub, match, sub = + string.byte, string.char, string.find, string.gsub, string.match, string.sub + +local function _parse_error(pos, errmsg) + error("parse error at " .. pos .. ": " .. errmsg, 2) +end + +local f_str_ctrl_pat +if _VERSION == "Lua 5.1" then + -- use the cluttered pattern because lua 5.1 does not handle \0 in a pattern correctly + f_str_ctrl_pat = '[^\32-\255]' +else + f_str_ctrl_pat = '[\0-\31]' +end + +local type, unpack = type, table.unpack or unpack +local open = io.open + +local _ENV = nil + + +local function nop() end + +local function newparser(src, saxtbl) + local json, jsonnxt, rec_depth + local jsonlen, pos, acc = 0, 1, 0 + + -- `f` is the temporary for dispatcher[c] and + -- the dummy for the first return value of `find` + local dispatcher, f + + -- initialize + if type(src) == 'string' then + json = src + jsonlen = #json + jsonnxt = function() + json = '' + jsonlen = 0 + jsonnxt = nop + end + else + jsonnxt = function() + acc = acc + jsonlen + pos = 1 + repeat + json = src() + if not json then + json = '' + jsonlen = 0 + jsonnxt = nop + return + end + jsonlen = #json + until jsonlen > 0 + end + jsonnxt() + end + + local sax_startobject = saxtbl.startobject or nop + local sax_key = saxtbl.key or nop + local sax_endobject = saxtbl.endobject or nop + local sax_startarray = saxtbl.startarray or nop + local sax_endarray = saxtbl.endarray or nop + local sax_string = saxtbl.string or nop + local sax_number = saxtbl.number or nop + local sax_boolean = saxtbl.boolean or nop + local sax_null = saxtbl.null or nop + + --[[ + Helper + --]] + local function tryc() + local c = byte(json, pos) + if not c then + jsonnxt() + c = byte(json, pos) + end + return c + end + + local function parse_error(errmsg) + return _parse_error(acc + pos, errmsg) + end + + local function tellc() + return tryc() or parse_error("unexpected termination") + end + + local function spaces() -- skip spaces and prepare the next char + while true do + pos = match(json, '^[ \n\r\t]*()', pos) + if pos <= jsonlen then + return + end + if jsonlen == 0 then + parse_error("unexpected termination") + end + jsonnxt() + end + end + + --[[ + Invalid + --]] + local function f_err() + parse_error('invalid value') + end + + --[[ + Constants + --]] + -- fallback slow constants parser + local function generic_constant(target, targetlen, ret, sax_f) + for i = 1, targetlen do + local c = tellc() + if byte(target, i) ~= c then + parse_error("invalid char") + end + pos = pos+1 + end + return sax_f(ret) + end + + -- null + local function f_nul() + if sub(json, pos, pos+2) == 'ull' then + pos = pos+3 + return sax_null(nil) + end + return generic_constant('ull', 3, nil, sax_null) + end + + -- false + local function f_fls() + if sub(json, pos, pos+3) == 'alse' then + pos = pos+4 + return sax_boolean(false) + end + return generic_constant('alse', 4, false, sax_boolean) + end + + -- true + local function f_tru() + if sub(json, pos, pos+2) == 'rue' then + pos = pos+3 + return sax_boolean(true) + end + return generic_constant('rue', 3, true, sax_boolean) + end + + --[[ + Numbers + Conceptually, the longest prefix that matches to `[-+.0-9A-Za-z]+` (in regexp) + is captured as a number and its conformance to the JSON spec is checked. + --]] + -- deal with non-standard locales + local radixmark = match(tostring(0.5), '[^0-9]') + local fixedtonumber = tonumber + if radixmark ~= '.' then + if find(radixmark, '%W') then + radixmark = '%' .. radixmark + end + fixedtonumber = function(s) + return tonumber(gsub(s, '.', radixmark)) + end + end + + local function number_error() + return parse_error('invalid number') + end + + -- fallback slow parser + local function generic_number(mns) + local buf = {} + local i = 1 + local is_int = true + + local c = byte(json, pos) + pos = pos+1 + + local function nxt() + buf[i] = c + i = i+1 + c = tryc() + pos = pos+1 + end + + if c == 0x30 then + nxt() + if c and 0x30 <= c and c < 0x3A then + number_error() + end + else + repeat nxt() until not (c and 0x30 <= c and c < 0x3A) + end + if c == 0x2E then + is_int = false + nxt() + if not (c and 0x30 <= c and c < 0x3A) then + number_error() + end + repeat nxt() until not (c and 0x30 <= c and c < 0x3A) + end + if c == 0x45 or c == 0x65 then + is_int = false + nxt() + if c == 0x2B or c == 0x2D then + nxt() + end + if not (c and 0x30 <= c and c < 0x3A) then + number_error() + end + repeat nxt() until not (c and 0x30 <= c and c < 0x3A) + end + if c and (0x41 <= c and c <= 0x5B or + 0x61 <= c and c <= 0x7B or + c == 0x2B or c == 0x2D or c == 0x2E) then + number_error() + end + pos = pos-1 + + local num = char(unpack(buf)) + num = fixedtonumber(num) + if mns then + num = -num + if num == mininteger and is_int then + num = mininteger + end + end + return sax_number(num) + end + + -- `0(\.[0-9]*)?([eE][+-]?[0-9]*)?` + local function f_zro(mns) + local num, c = match(json, '^(%.?[0-9]*)([-+.A-Za-z]?)', pos) -- skipping 0 + + if num == '' then + if pos > jsonlen then + pos = pos - 1 + return generic_number(mns) + end + if c == '' then + if mns then + return sax_number(-0.0) + end + return sax_number(0) + end + + if c == 'e' or c == 'E' then + num, c = match(json, '^([^eE]*[eE][-+]?[0-9]+)([-+.A-Za-z]?)', pos) + if c == '' then + pos = pos + #num + if pos > jsonlen then + pos = pos - #num - 1 + return generic_number(mns) + end + if mns then + return sax_number(-0.0) + end + return sax_number(0.0) + end + end + pos = pos-1 + return generic_number(mns) + end + + if byte(num) ~= 0x2E or byte(num, -1) == 0x2E then + pos = pos-1 + return generic_number(mns) + end + + if c ~= '' then + if c == 'e' or c == 'E' then + num, c = match(json, '^([^eE]*[eE][-+]?[0-9]+)([-+.A-Za-z]?)', pos) + end + if c ~= '' then + pos = pos-1 + return generic_number(mns) + end + end + + pos = pos + #num + if pos > jsonlen then + pos = pos - #num - 1 + return generic_number(mns) + end + c = fixedtonumber(num) + + if mns then + c = -c + end + return sax_number(c) + end + + -- `[1-9][0-9]*(\.[0-9]*)?([eE][+-]?[0-9]*)?` + local function f_num(mns) + pos = pos-1 + local num, c = match(json, '^([0-9]+%.?[0-9]*)([-+.A-Za-z]?)', pos) + if byte(num, -1) == 0x2E then -- error if ended with period + return generic_number(mns) + end + + if c ~= '' then + if c ~= 'e' and c ~= 'E' then + return generic_number(mns) + end + num, c = match(json, '^([^eE]*[eE][-+]?[0-9]+)([-+.A-Za-z]?)', pos) + if not num or c ~= '' then + return generic_number(mns) + end + end + + pos = pos + #num + if pos > jsonlen then + pos = pos - #num + return generic_number(mns) + end + c = fixedtonumber(num) + + if mns then + c = -c + if c == mininteger and not find(num, '[^0-9]') then + c = mininteger + end + end + return sax_number(c) + end + + -- skip minus sign + local function f_mns() + local c = byte(json, pos) or tellc() + if c then + pos = pos+1 + if c > 0x30 then + if c < 0x3A then + return f_num(true) + end + else + if c > 0x2F then + return f_zro(true) + end + end + end + parse_error("invalid number") + end + + --[[ + Strings + --]] + local f_str_hextbl = { + 0x0, 0x1, 0x2, 0x3, 0x4, 0x5, 0x6, 0x7, + 0x8, 0x9, inf, inf, inf, inf, inf, inf, + inf, 0xA, 0xB, 0xC, 0xD, 0xE, 0xF, inf, + inf, inf, inf, inf, inf, inf, inf, inf, + inf, inf, inf, inf, inf, inf, inf, inf, + inf, inf, inf, inf, inf, inf, inf, inf, + inf, 0xA, 0xB, 0xC, 0xD, 0xE, 0xF, + __index = function() + return inf + end + } + setmetatable(f_str_hextbl, f_str_hextbl) + + local f_str_escapetbl = { + ['"'] = '"', + ['\\'] = '\\', + ['/'] = '/', + ['b'] = '\b', + ['f'] = '\f', + ['n'] = '\n', + ['r'] = '\r', + ['t'] = '\t', + __index = function() + parse_error("invalid escape sequence") + end + } + setmetatable(f_str_escapetbl, f_str_escapetbl) + + local function surrogate_first_error() + return parse_error("1st surrogate pair byte not continued by 2nd") + end + + local f_str_surrogate_prev = 0 + local function f_str_subst(ch, ucode) + if ch == 'u' then + local c1, c2, c3, c4, rest = byte(ucode, 1, 5) + ucode = f_str_hextbl[c1-47] * 0x1000 + + f_str_hextbl[c2-47] * 0x100 + + f_str_hextbl[c3-47] * 0x10 + + f_str_hextbl[c4-47] + if ucode ~= inf then + if ucode < 0x80 then -- 1byte + if rest then + return char(ucode, rest) + end + return char(ucode) + elseif ucode < 0x800 then -- 2bytes + c1 = floor(ucode / 0x40) + c2 = ucode - c1 * 0x40 + c1 = c1 + 0xC0 + c2 = c2 + 0x80 + if rest then + return char(c1, c2, rest) + end + return char(c1, c2) + elseif ucode < 0xD800 or 0xE000 <= ucode then -- 3bytes + c1 = floor(ucode / 0x1000) + ucode = ucode - c1 * 0x1000 + c2 = floor(ucode / 0x40) + c3 = ucode - c2 * 0x40 + c1 = c1 + 0xE0 + c2 = c2 + 0x80 + c3 = c3 + 0x80 + if rest then + return char(c1, c2, c3, rest) + end + return char(c1, c2, c3) + elseif 0xD800 <= ucode and ucode < 0xDC00 then -- surrogate pair 1st + if f_str_surrogate_prev == 0 then + f_str_surrogate_prev = ucode + if not rest then + return '' + end + surrogate_first_error() + end + f_str_surrogate_prev = 0 + surrogate_first_error() + else -- surrogate pair 2nd + if f_str_surrogate_prev ~= 0 then + ucode = 0x10000 + + (f_str_surrogate_prev - 0xD800) * 0x400 + + (ucode - 0xDC00) + f_str_surrogate_prev = 0 + c1 = floor(ucode / 0x40000) + ucode = ucode - c1 * 0x40000 + c2 = floor(ucode / 0x1000) + ucode = ucode - c2 * 0x1000 + c3 = floor(ucode / 0x40) + c4 = ucode - c3 * 0x40 + c1 = c1 + 0xF0 + c2 = c2 + 0x80 + c3 = c3 + 0x80 + c4 = c4 + 0x80 + if rest then + return char(c1, c2, c3, c4, rest) + end + return char(c1, c2, c3, c4) + end + parse_error("2nd surrogate pair byte appeared without 1st") + end + end + parse_error("invalid unicode codepoint literal") + end + if f_str_surrogate_prev ~= 0 then + f_str_surrogate_prev = 0 + surrogate_first_error() + end + return f_str_escapetbl[ch] .. ucode + end + + local function f_str(iskey) + local pos2 = pos + local newpos + local str = '' + local bs + while true do + while true do -- search '\' or '"' + newpos = find(json, '[\\"]', pos2) + if newpos then + break + end + str = str .. sub(json, pos, jsonlen) + if pos2 == jsonlen+2 then + pos2 = 2 + else + pos2 = 1 + end + jsonnxt() + if jsonlen == 0 then + parse_error("unterminated string") + end + end + if byte(json, newpos) == 0x22 then -- break if '"' + break + end + pos2 = newpos+2 -- skip '\' + bs = true -- mark the existence of a backslash + end + str = str .. sub(json, pos, newpos-1) + pos = newpos+1 + + if find(str, f_str_ctrl_pat) then + parse_error("unescaped control string") + end + if bs then -- a backslash exists + -- We need to grab 4 characters after the escape char, + -- for encoding unicode codepoint to UTF-8. + -- As we need to ensure that every first surrogate pair byte is + -- immediately followed by second one, we grab upto 5 characters and + -- check the last for this purpose. + str = gsub(str, '\\(.)([^\\]?[^\\]?[^\\]?[^\\]?[^\\]?)', f_str_subst) + if f_str_surrogate_prev ~= 0 then + f_str_surrogate_prev = 0 + parse_error("1st surrogate pair byte not continued by 2nd") + end + end + + if iskey then + return sax_key(str) + end + return sax_string(str) + end + + --[[ + Arrays, Objects + --]] + -- arrays + local function f_ary() + rec_depth = rec_depth + 1 + if rec_depth > 1000 then + parse_error('too deeply nested json (> 1000)') + end + sax_startarray() + + spaces() + if byte(json, pos) == 0x5D then -- check closing bracket ']' which means the array empty + pos = pos+1 + else + local newpos + while true do + f = dispatcher[byte(json, pos)] -- parse value + pos = pos+1 + f() + newpos = match(json, '^[ \n\r\t]*,[ \n\r\t]*()', pos) -- check comma + if newpos then + pos = newpos + else + newpos = match(json, '^[ \n\r\t]*%]()', pos) -- check closing bracket + if newpos then + pos = newpos + break + end + spaces() -- since the current chunk can be ended, skip spaces toward following chunks + local c = byte(json, pos) + pos = pos+1 + if c == 0x2C then -- check comma again + spaces() + elseif c == 0x5D then -- check closing bracket again + break + else + parse_error("no closing bracket of an array") + end + end + if pos > jsonlen then + spaces() + end + end + end + + rec_depth = rec_depth - 1 + return sax_endarray() + end + + -- objects + local function f_obj() + rec_depth = rec_depth + 1 + if rec_depth > 1000 then + parse_error('too deeply nested json (> 1000)') + end + sax_startobject() + + spaces() + if byte(json, pos) == 0x7D then -- check closing bracket '}' which means the object empty + pos = pos+1 + else + local newpos + while true do + if byte(json, pos) ~= 0x22 then + parse_error("not key") + end + pos = pos+1 + f_str(true) -- parse key + newpos = match(json, '^[ \n\r\t]*:[ \n\r\t]*()', pos) -- check colon + if newpos then + pos = newpos + else + spaces() -- read spaces through chunks + if byte(json, pos) ~= 0x3A then -- check colon again + parse_error("no colon after a key") + end + pos = pos+1 + spaces() + end + if pos > jsonlen then + spaces() + end + f = dispatcher[byte(json, pos)] + pos = pos+1 + f() -- parse value + newpos = match(json, '^[ \n\r\t]*,[ \n\r\t]*()', pos) -- check comma + if newpos then + pos = newpos + else + newpos = match(json, '^[ \n\r\t]*}()', pos) -- check closing bracket + if newpos then + pos = newpos + break + end + spaces() -- read spaces through chunks + local c = byte(json, pos) + pos = pos+1 + if c == 0x2C then -- check comma again + spaces() + elseif c == 0x7D then -- check closing bracket again + break + else + parse_error("no closing bracket of an object") + end + end + if pos > jsonlen then + spaces() + end + end + end + + rec_depth = rec_depth - 1 + return sax_endobject() + end + + --[[ + The jump table to dispatch a parser for a value, + indexed by the code of the value's first char. + Key should be non-nil. + --]] + dispatcher = { [0] = + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_str, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_mns, f_err, f_err, + f_zro, f_num, f_num, f_num, f_num, f_num, f_num, f_num, + f_num, f_num, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_ary, f_err, f_err, f_err, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_fls, f_err, + f_err, f_err, f_err, f_err, f_err, f_err, f_nul, f_err, + f_err, f_err, f_err, f_err, f_tru, f_err, f_err, f_err, + f_err, f_err, f_err, f_obj, f_err, f_err, f_err, f_err, + } + + --[[ + public funcitons + --]] + local function run() + rec_depth = 0 + spaces() + f = dispatcher[byte(json, pos)] + pos = pos+1 + f() + end + + local function read(n) + if n < 0 then + error("the argument must be non-negative") + end + local pos2 = (pos-1) + n + local str = sub(json, pos, pos2) + while pos2 > jsonlen and jsonlen ~= 0 do + jsonnxt() + pos2 = pos2 - (jsonlen - (pos-1)) + str = str .. sub(json, pos, pos2) + end + if jsonlen ~= 0 then + pos = pos2+1 + end + return str + end + + local function tellpos() + return acc + pos + end + + return { + run = run, + tryc = tryc, + read = read, + tellpos = tellpos, + } +end + +local function newfileparser(fn, saxtbl) + local fp = open(fn) + local function gen() + local s + if fp then + s = fp:read(8192) + if not s then + fp:close() + fp = nil + end + end + return s + end + return newparser(gen, saxtbl) +end + +return { + newparser = newparser, + newfileparser = newfileparser +} +end +end + +do +local _ENV = _ENV +package.preload[ "utils" ] = function( ... ) local arg = _G.arg; +local module = {} + +function module.tablelength(T) + local count = 0 + for _ in pairs(T) do count = count + 1 end + return count +end + +module.id_number = 0 +function module.next_id(length) + module.id_number = module.id_number + 1 + return string.format(string.format('%%0%dd', length), module.id_number) +end + +local function url_encode_char(chr) + return string.format("%%%X",string.byte(chr)) +end + +function module.urlencode(str) + local output, t = string.gsub(str,"[^%w]",url_encode_char) + return output +end + +function module.xmlescape(str) + return string.gsub(str, '[<>&]', { ['&'] = '&', ['<'] = '<', ['>'] = '>' }) +end + +function module.xmlattr(str) + return string.gsub(str, '["<>&]', { ['&'] = '&', ['<'] = '<', ['>'] = '>', ['"'] = '"' }) +end + +function module.trim(s) + return s:gsub("^%s*(.-)%s*$", "%1") +end + +function module.deepcopy(orig) + local orig_type = type(orig) + local copy + if orig_type == 'table' then + copy = {} + for orig_key, orig_value in next, orig, nil do + copy[module.deepcopy(orig_key)] = module.deepcopy(orig_value) + end + setmetatable(copy, module.deepcopy(getmetatable(orig))) + else -- number, string, boolean, etc + copy = orig + end + return copy +end + +function module.dump(o) + if type(o) == 'table' then + local s = '{ ' + for k,v in pairs(o) do + if type(k) ~= 'number' then k = '"'..k..'"' end + s = s .. '['..k..'] = ' .. module.dump(v) .. ',' + end + return s .. '} ' + else + return tostring(o) + end +end + + +function module.trim(s) + if s == nil then + return s + end + return (s:gsub("^%s*(.-)%s*$", "%1")) +end + +return module +end +end + +do +local _ENV = _ENV +package.preload[ "zotero" ] = function( ... ) local arg = _G.arg; +local module = {} + +local utils = require('utils') +local json = require('lunajson') +-- local pl = require('pl.pretty') -- for pl.pretty.dump + +local state = { + reported = {}, +} + +module.citekeys = {} + +local function load_items() + if state.fetched ~= nil then + return + end + + state.fetched = { + items = {}, + errors = {}, + } + + local citekeys = {} + for k, _ in pairs(module.citekeys) do + table.insert(citekeys, k) + end + + if utils.tablelength(citekeys) == 0 then + return + end + + module.request.params.citekeys = citekeys + local url = module.url .. utils.urlencode(json.encode(module.request)) + local mt, body = pandoc.mediabag.fetch(url, '.') + local ok, response = pcall(json.decode, body) + if not ok then + print('could not fetch Zotero items: ' .. response .. '(' .. body .. ')') + return + end + if response.error ~= nil then + print('could not fetch Zotero items: ' .. response.error.message) + return + end + state.fetched = response.result +end + +function module.get(citekey) + load_items() + + if state.reported[citekey] ~= nil then + return nil + end + + if state.fetched.errors[citekey] ~= nil then + state.reported[citekey] = true + if state.fetched.errors[citekey] == 0 then + print('@' .. citekey .. ': not found') + else + print('@' .. citekey .. ': duplicates found') + end + return nil + end + + if state.fetched.items[citekey] == nil then + state.reported[citekey] = true + print('@' .. citekey .. ' not in Zotero') + return nil + end + + return state.fetched.items[citekey] +end + +return module +end +end + +-- +-- bbt-to-live-doc +-- +-- Copyright (c) 2020 Emiliano Heyns +-- +-- Permission is hereby granted, free of charge, to any person obtaining a copy of +-- this software and associated documentation files (the "Software"), to deal in +-- the Software without restriction, including without limitation the rights to +-- use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies +-- of the Software, and to permit persons to whom the Software is furnished to do +-- so, subject to the following conditions: +-- +-- The above copyright notice and this permission notice shall be included in all +-- copies or substantial portions of the Software. +-- +-- THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +-- IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +-- FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +-- AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +-- LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +-- OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +-- SOFTWARE. +-- + +if lpeg == nil then + print('upgrade pandoc to version 2.16.2 or later') + os.exit() +end + +local json = require('lunajson') +local csl_locator = require('locator') +local utils = require('utils') +local zotero = require('zotero') + +-- -- global state -- -- +local config = { + client = 'zotero', + scannable_cite = false, + csl_style = 'apa', + format = nil, -- more to document than anything else -- Lua does not store nils in tables + transferable = false, + sorted = true, +} + +-- -- bibliography marker generator -- -- +function zotero_docpreferences_odt(csl_style) + return string.format( + '' + .. ' ' + .. '