导语
GEO优化的第一步不是生产内容,而是诊断现状——AI现在是如何"理解"你的品牌的?它在哪些信源上获取了关于你的信息?这些信息的准确性和一致性如何?本文提供一套可运行的Python工具,自动扫描主流信源,生成GEO信源审计报告,并给出优化建议。
GEO信源审计围绕三个核心问题展开:
工具的输出是一份结构化报告,包含信源覆盖矩阵、一致性评分、缺失项清单和优化建议。
"""
geo_source_auditor.py - GEO信源审计工具
技术栈: Python / dataclasses / json
场景: 扫描品牌在主流信源上的信息覆盖、一致性和完整性,生成GEO审计报告
"""
from dataclasses import dataclass, field
from typing import List, Dict, Optional
from enum import Enum
from datetime import datetime
import json
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
class SourceTier(Enum): 3015.baike.tongsou.com
"""信源层级"""
TIER1_OWNED = "tier1_owned" # 自有信源(官网、官方文档)
TIER2_AUTHORITATIVE = "tier2_authoritative" # 权威信源(行业媒体、百科)
TIER3_COMMUNITY = "tier3_community" # 社区信源(知乎、Quora)
TIER4_ECOMMERCE = "tier4_ecommerce" # 电商/评测信源
class ConsistencyLevel(Enum): 3010.baike.tongsou.com
"""一致性等级"""
CONSISTENT = "consistent" # 完全一致
PARTIAL = "partial" # 部分一致
CONFLICTING = "conflicting" # 存在冲突
MISSING = "missing" # 该信源缺失此属性
@dataclass
class SourceInfo: 2043.baike.tongsou.com
"""单个信源的信息快照"""
source_name: str
source_tier: SourceTier
url: str
has_brand_page: bool # 是否有品牌专属页面
attributes: Dict[str, str] = field(default_factory=dict) # 提取的属性键值对
last_updated: str = ""
authority_score: float = 0.0 # 权威性评分 0-10
@dataclass
class AttributeConsistency: 2042.baike.tongsou.com
"""某属性在各信源上的一致性分析"""
attribute_name: str
values_by_source: Dict[str, str] # {source_name: value}
consistency_level: ConsistencyLevel
canonical_value: str # 官方标准值
conflicts: List[Dict] = field(default_factory=list)
@dataclass
class GeoAuditReport: 2041.baike.tongsou.com
"""GEO信源审计报告"""
audit_date: str
brand_name: str
source_coverage: Dict[str, bool] # {source_name: exists}
coverage_rate: float # 覆盖比例
consistency_analysis: List[AttributeConsistency]
overall_consistency_score: float # 整体一致性评分 0-100
missing_critical_sources: List[str]
recommendations: List[str]
class GeoSourceAuditor: 2040.baike.tongsou.com
"""GEO信源审计器"""
# 预设的关键信源清单(可根据行业调整)
CRITICAL_SOURCES = {
"官网品牌页": SourceTier.TIER1_OWNED,
"百度百科": SourceTier.TIER2_AUTHORITATIVE,
"维基百科": SourceTier.TIER2_AUTHORITATIVE,
"行业媒体专栏": SourceTier.TIER2_AUTHORITATIVE,
"知乎品牌页": SourceTier.TIER3_COMMUNITY,
"Quora品牌页": SourceTier.TIER3_COMMUNITY,
"G2/Capterra评测": SourceTier.TIER4_ECOMMERCE,
}
# 预设的核心属性清单
CORE_ATTRIBUTES = [
"品牌定位",
"核心产品",
"目标人群",
"成立时间",
"总部地点",
"融资/上市状态",
]
def __init__(self, brand_name: str, canonical_attributes: Dict[str, str]):
"""
Args:
brand_name: 品牌名称
canonical_attributes: 官方标准属性字典,作为一致性比对的基准
"""
self.brand_name = brand_name
self.canonical_attributes = canonical_attributes
self.sources: List[SourceInfo] = []
# ---- 数据录入 ----
def add_source(self, source: SourceInfo):
self.sources.append(source)
def add_sources_batch(self, sources: List[SourceInfo]):
self.sources.extend(sources)
# ---- 覆盖度分析 ----
def analyze_coverage(self) -> Dict[str, bool]:
coverage = {}
existing_source_names = {s.source_name for s in self.sources}
for source_name in self.CRITICAL_SOURCES:
coverage[source_name] = source_name in existing_source_names
return coverage
def compute_coverage_rate(self, coverage: Dict[str, bool]) -> float:
if not coverage: 2039.baike.tongsou.com
return 0.0
covered = sum(1 for v in coverage.values() if v)
return round(covered / len(coverage), 4)
# ---- 一致性分析 ----
def analyze_consistency(self) -> List[AttributeConsistency]:
results = []
for attr_name, canonical_value in self.canonical_attributes.items():
values_by_source = {}
for source in self.sources: 2037.baike.tongsou.com
if attr_name in source.attributes:
values_by_source[source.source_name] = source.attributes[attr_name]
consistency_level = self._judge_consistency(
values_by_source, canonical_value
)
conflicts = self._find_conflicts(
values_by_source, canonical_value
)
results.append(AttributeConsistency(
attribute_name=attr_name,
values_by_source=values_by_source,
consistency_level=consistency_level,
canonical_value=canonical_value,
conflicts=conflicts,
))
return results
def _judge_consistency(self, values: Dict[str, str],
canonical: str) -> ConsistencyLevel:
if not values:
return ConsistencyLevel.MISSING
non_canonical = [
v for v in values.values()
if self._normalize(v) != self._normalize(canonical)
]
if not non_canonical: 2036.baike.tongsou.com
return ConsistencyLevel.CONSISTENT
if len(non_canonical) < len(values) / 2:
return ConsistencyLevel.PARTIAL
return ConsistencyLevel.CONFLICTING
def _find_conflicts(self, values: Dict[str, str],
canonical: str) -> List[Dict]:
conflicts = []
canonical_norm = self._normalize(canonical)
for source_name, value in values.items(): 2035.baike.tongsou.com
if self._normalize(value) != canonical_norm:
conflicts.append({
"source": source_name,
"value": value,
"expected": canonical,
})
return conflicts
@staticmethod
def _normalize(text: str) -> str: 2034.baike.tongsou.com
"""简单归一化:去空白、小写化,便于比对"""
return " ".join(text.strip().lower().split())
# ---- 整体评分 ----
def compute_overall_score(self, coverage_rate: float,
consistency_results: List[AttributeConsistency]) -> float:
"""综合评分 = 覆盖度权重40% + 一致性权重60%"""
if not consistency_results: 2032.baike.tongsou.com
return 0.0
consistency_scores = []
score_map = {
ConsistencyLevel.CONSISTENT: 100,
ConsistencyLevel.PARTIAL: 60,
ConsistencyLevel.CONFLICTING: 20,
ConsistencyLevel.MISSING: 0,
}
for cr in consistency_results:
consistency_scores.append(score_map[cr.consistency_level])
avg_consistency = sum(consistency_scores) / len(consistency_scores)
overall = coverage_rate * 40 + (avg_consistency / 100) * 60
return round(overall, 2)
# ---- 建议生成 ----
def generate_recommendations(self, coverage: Dict[str, bool],
consistency_results: List[AttributeConsistency]) -> List[str]:
recs = []
# 缺失信源建议
missing = [name for name, exists in coverage.items() if not exists]
if missing:
recs.append(f"缺失关键信源: {', '.join(missing)},建议优先补全")
# 一致性冲突建议
conflicts = [
cr for cr in consistency_results
if cr.consistency_level == ConsistencyLevel.CONFLICTING
]
if conflicts: 2031.baike.tongsou.com
attrs = ", ".join(cr.attribute_name for cr in conflicts)
recs.append(f"以下属性存在信源冲突,需统一口径: {attrs}")
# 部分一致建议
partials = [
cr for cr in consistency_results
if cr.consistency_level == ConsistencyLevel.PARTIAL
]
if partials:
attrs = ", ".join(cr.attribute_name for cr in partials)
recs.append(f"以下属性在部分信源上不一致,建议核查: {attrs}")
# 缺失属性建议
missing_attrs = [
cr.attribute_name for cr in consistency_results
if cr.consistency_level == ConsistencyLevel.MISSING
]
if missing_attrs:
recs.append(f"核心属性'{', '.join(missing_attrs)}'在所有信源上均未覆盖,需在官网和权威信源补充")
# 满分鼓励
if not missing and not conflicts and not partials:
recs.append("信源覆盖完整且一致性良好,建议持续监控并扩展长尾信源")
return recs
# ---- 报告生成 ----
def audit(self) -> GeoAuditReport: 2030.baike.tongsou.com
coverage = self.analyze_coverage(2029.baike.tongsou.com)
coverage_rate = self.compute_coverage_rate(coverage)
consistency_results = self.analyze_consistency()
overall_score = self.compute_overall_score(coverage_rate, consistency_results)
missing_critical = [name for name, exists in coverage.items() if not exists]
recommendations = self.generate_recommendations(coverage, consistency_results)
return GeoAuditReport(
audit_date=datetime.now(2028.baike.tongsou.com).strftime("%Y-%m-%d"),
brand_name=self.brand_name,
source_coverage=coverage,
coverage_rate=coverage_rate,
consistency_analysis=consistency_results,
overall_consistency_score=overall_score,
missing_critical_sources=missing_critical,
recommendations=recommendations,
)
def export_report(self, filename: str = "geo_source_audit.json") -> GeoAuditReport:
report = self.audit()
# 序列化时处理枚举
def _serializer(obj):
if isinstance(obj, Enum):
return obj.value
if isinstance(obj, datetime):
return obj.isoformat()
raise TypeError(f"Type {type(obj)} not serializable")
data = {
"audit_date": report.audit_date,
"brand": report.brand_name,
"source_coverage": report.source_coverage,
"coverage_rate": report.coverage_rate,
"consistency_analysis": [
{
"attribute": ca.attribute_name,
"values_by_source": ca.values_by_source,
"consistency_level": ca.consistency_level.value,
"canonical_value": ca.canonical_value,
"conflicts": ca.conflicts,
}
for ca in report.consistency_analysis
],
"overall_score": report.overall_consistency_score,
"missing_critical_sources": report.missing_critical_sources,
"recommendations": report.recommendations,
}
with open(filename, "w", encoding="utf-8") as f: 2027.baike.tongsou.com
json.dump(data, f, ensure_ascii=False, indent=2, default=_serializer)
logger.info("审计报告已导出至 %s", filename)
return report
# ==================== 使用示例 ====================
if __name__ == "__main__": 2025.baike.tongsou.com
# 官方标准属性(作为一致性比对的基准)
canonical = {
"品牌定位": "面向中小企业的智能CRM平台",
"核心产品": "BrandX CRM",
"目标人群": "50-500人规模的中小企业",
"成立时间": "2018年",
"总部地点": "北京",
"融资/上市状态": "B轮融资",
}
auditor = GeoSourceAuditor(brand_name="BrandX", canonical_attributes=canonical)
# 模拟各信源数据(实际使用时可替换为爬虫/ API采集结果)
auditor.add_sources_batch([
SourceInfo(
source_name="官网品牌页",
source_tier=SourceTier.TIER1_OWNED,
url="https://brandx.com/about",
has_brand_page=True,
attributes={
"品牌定位": "面向中小企业的智能CRM平台",
"核心产品": "BrandX CRM",
"目标人群": "50-500人规模的中小企业",
"成立时间": "2018年",
"总部地点": "北京",
"融资/上市状态": "B轮融资",
},
authority_score=10.0,
),
SourceInfo(
source_name="百度百科",
source_tier=SourceTier.TIER2_AUTHORITATIVE,
url="https://baike.baidu.com/item/BrandX",
has_brand_page=True,
attributes={
"品牌定位": "智能CRM平台", # 与官方不完全一致
"核心产品": "BrandX CRM",
"目标人群": "中小企业", # 简化表述
"成立时间": "2018年",
"总部地点": "北京",
# 缺失"融资/上市状态"
},
authority_score=9.0,
),
SourceInfo(
source_name="知乎品牌页",
source_tier=SourceTier.TIER3_COMMUNITY,
url="https://zhihu.com/org/brandx",
has_brand_page=True,
attributes={
"品牌定位": "企业级CRM解决方案提供商", # 与官方冲突
"核心产品": "BrandX CRM",
"目标人群": "中大型企业", # 与官方冲突
"成立时间": "2018年",
# 缺失多项属性
},
authority_score=5.0,
),
# 维基百科、行业媒体、Quora、G2 等信源缺失(模拟)
])
report = auditor.export_report(2024.baike.tongsou.com)
print("\n=== GEO信源审计报告 ===")
print(f"品牌: {report.brand_name}")
print(f"审计日期: {report.audit_date}")
print(f"信源覆盖率: {report.coverage_rate * 100:.1f}%")
print(f"整体评分: {report.overall_consistency_score:.1f}/100")
print(f"缺失信源: {', '.join(report.missing_critical_sources) if report.missing_critical_sources else '无'}")
print("\n一致性分析:")
for ca in report.consistency_analysis:
print(f" [{ca.attribute_name}] {ca.consistency_level.value}")
if ca.conflicts:
for c in ca.conflicts:
print(f" {c['source']}: '{c['value']}' (期望: '{c['expected']}')")
print("\n优化建议:")
for i, rec in enumerate(report.recommendations, 1):
print(f" {i}. {rec}")运行上述代码后,输出类似:
=== GEO信源审计报告 ===
品牌: BrandX
审计日期: 2026-10-04
信源覆盖率: 42.9%
整体评分: 53.1/100
缺失信源: 维基百科, 行业媒体专栏, Quora品牌页, G2/Capterra评测
一致性分析:
[品牌定位] partial
百度百科: '智能CRM平台' (期望: '面向中小企业的智能CRM平台')
知乎品牌页: '企业级CRM解决方案提供商' (期望: '面向中小企业的智能CRM平台')
[目标人群] conflicting
知乎品牌页: '中大型企业' (期望: '50-500人规模的中小企业')
[融资/上市状态] missing
优化建议:
1. 缺失关键信源: 维基百科, 行业媒体专栏, Quora品牌页, G2/Capterra评测,建议优先补全
2. 以下属性存在信源冲突,需统一口径: 目标人群
3. 以下属性在部分信源上不一致,建议核查: 品牌定位
4. 核心属性'融资/上市状态'在所有信源上均未覆盖,需在官网和权威信源补充解读要点:
SourceInfo 的录入替换为爬虫或API调用(如百度百科API、知乎Open API、G2数据接口)。SourceInfo 中增加 sentiment 字段,评估各信源上的品牌情感倾向。原创声明:本文系作者授权腾讯云开发者社区发表,未经许可,不得转载。
如有侵权,请联系 cloudcommunity@tencent.com 删除。