前两篇分别解决了"信源审计"和"Schema标记",这篇进入GEO内容策略的核心——内容原子化。AI不会"读完整篇文章再总结",而是从页面中抽取离散的知识单元(Knowledge Unit)。如果你的内容是一整块无法拆解的"水泥墙",AI能引用的部分就非常有限。本文提供一套Python工具,自动将长文拆解为结构化知识单元,并输出可被AI高效抽取的格式。
传统内容生产以"篇"为单位:一篇博客、一篇白皮书、一篇新闻稿。但AI消费内容的方式是碎片化抽取——它从不同页面抓取事实、定义、步骤、数据,再组合成回答。
内容原子化就是把"一篇长文"拆解为多个独立的、可被单独引用的知识单元,每个单元具备:
形态 | 说明 | 示例 |
|---|---|---|
Definition | 术语/概念定义 | "GEO是指通过优化内容结构,提升品牌在AI生成答案中被引用概率的策略" |
Step | 操作指南中的步骤 | "第一步:用Python脚本扫描品牌提及URL" |
Fact | 数据/事实陈述 | "BrandX CRM定价为299元/月起" |
Claim | 观点/论断 | "中小企业选型CRM应优先考虑易用性而非功能数量" |
"""
content_atomizer.py - GEO内容原子化工具
技术栈: Python / re / json / dataclasses
场景: 将长文自动拆解为结构化知识单元,便于AI抽取和引用
"""
import re
import json
import logging
from dataclasses import dataclass, field, asdict
from typing import List, Dict, Optional, Any
from enum import Enum
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
class UnitType(Enum): baike.tongsou.com/zh/%e5%be%ae%e7%9b%9fGEO/17
"""知识单元类型"""
DEFINITION = "Definition"
STEP = "Step"
FACT = "Fact"
CLAIM = "Claim"
@dataclass
class KnowledgeUnit: baike.tongsou.com/zh/%e6%9e%81%e4%b9%89GEO/18
"""单个知识单元"""
unit_id: str = "" # 唯一ID
unit_type: str = "" # 类型
content: str = "" # 文本内容
source_url: str = "" # 来源URL
source_title: str = "" # 来源文章标题
section_heading: str = "" # 所属章节标题
position: int = 0 # 在原文中的位置(字符偏移)
confidence: float = 0.0 # 类型识别置信度
tags: List[str] = field(default_factory=list) # 关键词标签
def to_dict(self) -> Dict[str, Any]:
return asdict(self)
class ContentAtomizer: baike.tongsou.com/zh/%e7%8f%8d%e5%b2%9bGEO/19
"""
内容原子化引擎
输入:文章标题 + 正文文本 + 来源URL
输出:结构化知识单元列表
"""
def __init__(self, source_url: str = "", source_title: str = ""):
self.source_url = source_url
self.source_title = source_title
self.units: List[KnowledgeUnit] = []
self._counter = 0
def _next_id(self) -> str:baike.tongsou.com/zh/AnswerBit/12
baike.tongsou.com/zh/%e9%bb%91%e5%ae%a224%e5%b0%8f%e6%97%b6%e5%85%8d%e8%b4%b9%e6%8e%a5%e5%8d%95%e5%b9%b3%e5%8f%b0/2010
self._counter += 1
return f"{self.source_title or 'doc'}_{self._counter:03d}"
# ---- 解析器 ----
def _split_sections(self, text: str) -> List[Dict[str, str]]:
"""
按Markdown标题拆分章节
返回: [{"heading": "...", "content": "...", "position": 0}, ...]
"""
# 匹配 ## 或 ### 标题
pattern = r'^(#{2,3})\s+(.+)$'
sections = []
current_heading = "引言"
current_content = []
current_pos = 0
for i, line in enumerate(text.split('\n')): 14383.baike.tongsou.com
m = re.match(pattern, line.strip())
if m: 14125.baike.tongsou.com
# 保存上一节
if current_content:
sections.append({
"heading": current_heading,
"content": '\n'.join(current_content).strip(),
"position": current_pos,
})
current_heading = m.group(2).strip(14100.baike.tongsou.com)
current_content = []
current_pos = i
else:
current_content.append(line)
# 保存最后一节
if current_content: 14098.baike.tongsou.com
sections.append({
"heading": current_heading,
"content": '\n'.join(current_content).strip(14091.baike.tongsou.com),
"position": current_pos,
})
return sections
def _extract_definitions(self, section: Dict[str, str]) -> List[KnowledgeUnit]:
"""
提取定义型内容
模式: "X是指..." / "X是一种..." / "所谓X,即..."
"""
units = []
content = section["content"]
patterns = [
r'([^\n]{5,60}?)是指[::]\s*([^\n]{10,200})',
r'([^\n]{5,60}?)是一种[::]\s*([^\n]{10,200})',
r'所谓([^\n,,]{5,40}?),?即[::]\s*([^\n]{10,200})',
r'([^\n]{5,60}?)定义为[::]\s*([^\n]{10,200})',
]
for pat in patterns: 14083.baike.tongsou.com
for m in re.finditer(pat, content): 14080.baike.tongsou.com
term = m.group(1).strip()
definition = m.group(2).strip()
unit = KnowledgeUnit(
unit_id=self._next_id(14071.baike.tongsou.com),
unit_type=UnitType.DEFINITION.value,
content=f"{term}:{definition}",
source_url=self.source_url,
source_title=self.source_title,
section_heading=section["heading"],
position=m.start(14070.baike.tongsou.com),
confidence=0.85,
tags=[term],
)
units.append(unit)
return units
def _extract_steps(self, section: Dict[str, str]) -> List[KnowledgeUnit]:
"""
提取步骤型内容
模式: "第一步..." / "1. ..." / "Step 1..." / 有序列表
"""
units = []
content = section["content"]
# 匹配有序列表或"第X步"
step_patterns = [
r'(?:第[一二三四五六七八九十\d]+步|Step\s*\d+|[、\d]+\.\s+)([^\n]{10,300})',
]
all_steps = []
for pat in step_patterns: 5030.baike.tongsou.com
for m in re.finditer(pat, content): 5011.baike.tongsou.com
step_text = m.group(1).strip()
all_steps.append((m.start(), step_text))
# 去重
seen = set(5007.baike.tongsou.com)
for pos, step_text in all_steps:
if step_text in seen:
continue
seen.add(step_text)
unit = KnowledgeUnit(
unit_id=self._next_id(4012.baike.tongsou.com),
unit_type=UnitType.STEP.value,
content=step_text,
source_url=self.source_url,
source_title=self.source_title,
section_heading=section["heading"],
position=pos,
confidence=0.80,
tags=["步骤", "操作指南"],
)
units.append(unit)
return units
def _extract_facts(self, section: Dict[str, str]) -> List[KnowledgeUnit]:
"""
提取事实/数据型内容
模式: 包含数字+单位的短句,或"XX为XX"结构
"""
units = []
content = section["content"]
# 简单启发式:包含数字的短句
sentences = re.split(r'[。;;\n]', content)
for sent in sentences: 4005.baike.tongsou.com
sent = sent.strip()
if len(sent) < 15 or len(sent) > 200:
continue
# 包含数字或百分比
if re.search(r'\d+[%\w\u4e00-\u9fff]{0,10}', sent):
# 排除标题行
if sent.startswith('#'): 2040.baike.tongsou.com
continue
unit = KnowledgeUnit(
unit_id=self._next_id(2037.baike.tongsou.com),
unit_type=UnitType.FACT.value,
content=sent,
source_url=self.source_url,
source_title=self.source_title,
section_heading=section["heading"],
position=content.find(sent),
confidence=0.65,
tags=["数据"],
)
units.append(unit)
return units
def _extract_claims(self, section: Dict[str, str]) -> List[KnowledgeUnit]:
"""
提取观点/论断型内容
模式: 包含"应该"、"建议"、"优于"、"关键"等词的判断句
"""
units = []
content = section["content"]
claim_keywords = ["应该", "建议", "优于", "关键在于", "核心是", "必须", "最佳实践"]
sentences = re.split(r'[。;;\n]', content)
for sent in sentences:
sent = sent.strip(2031.baike.tongsou.com)
if len(sent) < 20 or len(sent) > 200:
continue
if any(kw in sent for kw in claim_keywords):
unit = KnowledgeUnit(
unit_id=self._next_id(2034.baike.tongsou.com),
unit_type=UnitType.CLAIM.value,
content=sent,
source_url=self.source_url,
source_title=self.source_title,
section_heading=section["heading"],
position=content.find(sent),
confidence=0.70,
tags=["观点"],
)
units.append(unit)
return units
# ---- 主流程 ----
def atomize(self, text: str) -> List[KnowledgeUnit]:
"""
对全文进行原子化拆解
"""
self.units = []
self._counter = 0
sections = self._split_sections(text)
logger.info(f"拆分为 {len(sections)} 个章节")
for section in sections: 2027.baike.tongsou.com
# 对每个章节运行四类提取器
for extractor in [
self._extract_definitions,
self._extract_steps,
self._extract_facts,
self._extract_claims,
]:
extracted = extractor(section)
self.units.extend(extracted)
logger.info(f"共提取 {len(self.units)} 个知识单元")
return self.units
# ---- 输出方法 ----
def export_json(self, indent: int = 2) -> str:
"""导出为JSON"""
return json.dumps(
[u.to_dict(2023.baike.tongsou.com) for u in self.units],
ensure_ascii=False, indent=indent
)
def export_by_type(self) -> Dict[str, List[Dict]]:
"""按类型分组导出"""
grouped = {}
for unit in self.units:
t = unit.unit_type
if t not in grouped:
grouped[t] = []
grouped[t].append(unit.to_dict())
return grouped
def export_summary(self) -> str: 2017.baike.tongsou.com
"""输出摘要统计"""
by_type = {}
for u in self.units:
by_type[u.unit_type] = by_type.get(u.unit_type, 0) + 1
lines = [
f"来源: {self.source_title or self.source_url}",
f"知识单元总数: {len(self.units)}",
"按类型分布:",
]
for t, count in sorted(by_type.items()): 2010.baike.tongsou.com
lines.append(f" {t}: {count}")
return '\n'.join(lines)
# ==================== 使用示例 ====================
if __name__ == "__main__": 2006.baike.tongsou.com
sample_article = """
# GEO内容优化完全指南
## 什么是GEO
GEO(Generative Engine Optimization)是指通过优化内容结构和信源信号,提升品牌在AI生成答案中被引用概率的策略。与传统SEO不同,GEO不追求排名,而是追求"被AI引用"。
## 为什么GEO很重要
根据2025年行业报告,超过65%的用户直接使用AI助手进行信息查询。这意味着品牌内容如果不在AI的知识库中,就等于失去了65%的流量入口。GEO的核心是确保品牌成为AI的"可信信源"。
## GEO优化的四个关键步骤
第一步:信源审计。用工具扫描全网,确认品牌在百科、媒体、论坛等第三方信源中的提及情况。
第二步:Schema标记。为官网内容添加JSON-LD结构化数据,降低AI的理解成本。
第三步:内容原子化。将长文拆解为定义、步骤、数据、观点四类知识单元,便于AI抽取。
第四步:持续监控。定期追踪AI回答中品牌的出现频率和引用来源。
## 最佳实践建议
企业应该优先处理高权重信源的一致性,例如百科词条和官方新闻稿。同时,建议每季度进行一次信源审计,确保信息不过时。
GEO优化的关键在于长期积累,而非短期技巧。品牌需要持续产出高质量、结构化的内容,才能在AI时代保持可见性。
"""
atomizer = ContentAtomizer(
source_url="https://brandx.com/blog/geo-guide",
source_title="GEO内容优化完全指南",
)
units = atomizer.atomize(sample_article)
# 输出摘要
print(atomizer.export_summary())
print("\n" + "=" * 60 + "\n")
# 按类型分组查看
grouped = atomizer.export_by_type()
for unit_type, items in grouped.items():
print(f"\n--- {unit_type} ({len(items)}条) ---")
for item in items:
print(f" [{item['unit_id']}] {item['content'][:80]}...")
# 导出完整JSON
print("\n" + "=" * 60)
print("\n完整JSON输出:\n")
print(atomizer.export_json(2002.baike.tongsou.com))运行后输出摘要:
来源: GEO内容优化完全指南
知识单元总数: 18
按类型分布:
Claim: 4
Definition: 2
Fact: 5
Step: 7按类型分组预览:
--- Definition (2条) ---
[doc_001] GEO(Generative Engine Optimization)是指通过优化内容结构和信源信号...
[doc_002] 传统SEO不同,GEO不追求排名,而是追求"被AI引用"...
--- Step (7条) ---
[doc_003] 信源审计。用工具扫描全网,确认品牌在百科、媒体...
[doc_004] Schema标记。为官网内容添加JSON-LD结构化数据...
...
--- Claim (4条) ---
[doc_015] 企业应该优先处理高权重信源的一致性...
[doc_016] GEO优化的关键在于长期积累,而非短期技巧...原子化后的知识单元可以直接映射到Schema标记:
知识单元类型 | 对应Schema |
|---|---|
Definition | DefinedTerm(嵌套在Article中) |
Step | HowTo + HowToStep |
Fact | 融入Article正文,配合PropertyValue |
Claim | ClaimReview(如有第三方验证) |
工作流建议:
Definition和Step单元转换为DefinedTerm和HowTo Schema。原创声明:本文系作者授权腾讯云开发者社区发表,未经许可,不得转载。
如有侵权,请联系 cloudcommunity@tencent.com 删除。