lactic-analyze.js 15 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354
  1. // ==============================================================================
  2. // 江中乳酸菌素片儿童版 · VOC 数据分析模块(参考 liver-analyze.js 架构)
  3. // ==============================================================================
  4. // 数据源:docs/乳酸菌/raw/_merged.json(已在 lactic-collect.js mergeAll 阶段预展平 items[])
  5. // 所以本模块不需要 flattenMerged(直接读取 items 即可)
  6. // ==============================================================================
  7. const fs = require('fs');
  8. const path = require('path');
  9. const ROOT = path.resolve(__dirname, '..', '..');
  10. const RAW_DIR = path.join(ROOT, 'docs', '乳酸菌', 'raw');
  11. const MERGED_PATH = path.join(RAW_DIR, '_merged.json');
  12. // ------------------------------------------------------------
  13. // 8 条核心假设定义(对应 docs/乳酸菌/2.VOC深度思路.md)
  14. // ------------------------------------------------------------
  15. const HYPOTHESES = {
  16. H1: { title: '药品益生菌痛点', desc: '妈咪爱/亿活/金双岐 家长吐槽冷链、活菌失活、处方门槛、依赖焦虑' },
  17. H2: { title: '食品益生菌信任缺口', desc: '合生元/万益蓝/拜奥/inne 家长"怕没效果"纠结' },
  18. H3: { title: '儿童肠道刚需', desc: '便秘/腹泻/积食/挑食/消化不良 真实家长诉求' },
  19. H4: { title: '场景触发', desc: '入园/换季/抗生素后/开学 集中发生的肠道问题' },
  20. H5: { title: '喂药依从性', desc: '孩子抗拒吃药/口感差/独立包装/喂药像打仗' },
  21. H6: { title: '包装 × 颜值决策', desc: '新生代妈妈专属感需求/货架辨识度/分龄/IP' },
  22. H7: { title: '营养吸收焦虑', desc: '瓶瓶罐罐太多/补了不吸收/内源调理替代' },
  23. H8: { title: '糖分 × 添加剂焦虑', desc: '新生代家长抵触糖/防腐剂/香精/色素' },
  24. };
  25. // 章节 → 假设映射
  26. const CHAPTER_HYPOTHESIS_MAP = {
  27. challenge: ['H1', 'H2', 'H3'], // 诘问:现状三大痛点
  28. drug_voc: ['H1'], // 药品益生菌 VOC
  29. food_voc: ['H2', 'H3'], // 食品益生菌 VOC
  30. kano: ['H3', 'H5', 'H8'], // KANO×JTBD
  31. scene: ['H4', 'H3'], // 场景四元素
  32. competitor: ['H1', 'H2'], // 三轴 × 竞品
  33. opportunity: ['H7', 'H5', 'H6'], // 新机会
  34. blueprint: ['H3', 'H7'], // 4P 蓝图
  35. };
  36. // ------------------------------------------------------------
  37. // 平台标签
  38. // ------------------------------------------------------------
  39. const PLATFORM_LABELS = {
  40. xhs: { name: '小红书', color: '#FF2442', short: '红' },
  41. douyin: { name: '抖音', color: '#1A1A1A', short: '抖' },
  42. taobao: { name: '淘宝', color: '#FF5000', short: '淘' },
  43. jd: { name: '京东', color: '#E1251B', short: '京' },
  44. tmall: { name: '天猫', color: '#FF0036', short: '猫' },
  45. unknown: { name: '其他', color: '#888', short: '—' },
  46. };
  47. // ------------------------------------------------------------
  48. // 关键词 → 产品 / 竞品 归一化
  49. // ------------------------------------------------------------
  50. const KEYWORD_TO_PRODUCT = {
  51. '乳酸菌素片': '江中乳酸菌素片',
  52. '乳酸菌素片儿童': '江中乳酸菌素片儿童',
  53. '江中乳酸菌素片': '江中乳酸菌素片',
  54. // 药品益生菌竞品
  55. '妈咪爱': '妈咪爱',
  56. '妈咪爱 宝宝': '妈咪爱',
  57. '亿活 儿童': '亿活',
  58. '金双岐': '金双岐',
  59. '宝乐安': '宝乐安',
  60. // 食品益生菌竞品
  61. '合生元益生菌': '合生元 儿童益生菌',
  62. '万益蓝益生菌': '万益蓝 儿童益生菌',
  63. '拜奥益生菌': '拜奥益生菌',
  64. 'inne 噗噗宝': 'inne 噗噗宝',
  65. '小胖瓶益生菌': '小胖瓶益生菌',
  66. '康萃乐益生菌': '康萃乐益生菌',
  67. 'lifespace 儿童': 'lifespace 儿童益生菌',
  68. '儿童益生菌推荐': '儿童益生菌(品类)',
  69. '儿童益生菌': '儿童益生菌(品类)',
  70. // 儿童肠道痛点话题
  71. '宝宝便秘': '便秘话题',
  72. '宝宝积食': '积食话题',
  73. '宝宝腹泻': '腹泻话题',
  74. '儿童挑食': '挑食话题',
  75. '抗生素 益生菌': '抗生素后肠道话题',
  76. // 场景
  77. '宝宝入园生病': '入园场景',
  78. '宝宝换季腹泻': '换季场景',
  79. // 喂药依从 × 包装
  80. '喂药难': '喂药难话题',
  81. '儿童咀嚼片': '咀嚼片形态',
  82. '宝宝专用': '儿童专属话题',
  83. // 营养吸收赛道
  84. '儿童营养软糖': '儿童营养软糖(竞品)',
  85. '儿童DHA': 'DHA 补充话题',
  86. '瓶瓶罐罐': '瓶瓶罐罐焦虑',
  87. '儿童补钙': '补钙话题',
  88. // 糖分 / 安全焦虑
  89. '儿童无糖': '无糖需求',
  90. };
  91. // ------------------------------------------------------------
  92. // 启发式再判(兜底 · lactic-collect.js 已做主要打标)
  93. // ------------------------------------------------------------
  94. const NEG_KEYS = ['智商税', '没用', '没效', '没有效果', '骗', '差评', '退货', '拉肚子', '副作用', '难喝', '难吃', '难喂', '抗拒', '吐出', '失望', '浪费钱', '坑'];
  95. const POS_KEYS = ['有效', '真香', '好用', '推荐', '回购', '复购', '神器', '爱吃', '管用', '对症', '好转', '有改善', '主动要'];
  96. const CONF_KEYS = ['纠结', '矛盾', '不知道.*好', '又.*又', '害怕', '担心', '想停', '是不是'];
  97. function inferSentiment(item) {
  98. if (item.sentiment && ['positive', 'negative', 'neutral', 'conflicted'].includes(item.sentiment)) return item.sentiment;
  99. const c = String(item.content || '').toLowerCase();
  100. for (const k of NEG_KEYS) if (c.includes(k.toLowerCase())) return 'negative';
  101. for (const k of POS_KEYS) if (c.includes(k.toLowerCase())) return 'positive';
  102. for (const k of CONF_KEYS) if (new RegExp(k).test(c)) return 'conflicted';
  103. return 'neutral';
  104. }
  105. // ------------------------------------------------------------
  106. // 数据加载
  107. // ------------------------------------------------------------
  108. const SKELETON = {
  109. meta: {
  110. collectedAt: '待采集',
  111. platforms: {},
  112. hypotheses: {},
  113. products: {},
  114. stage: 'skeleton',
  115. sourceNote: '暂无采集数据',
  116. },
  117. items: [],
  118. raw: null,
  119. };
  120. function loadMerged() {
  121. if (fs.existsSync(MERGED_PATH)) {
  122. try {
  123. const raw = JSON.parse(fs.readFileSync(MERGED_PATH, 'utf8'));
  124. const items = (raw.items || []).map((it) => ({
  125. ...it,
  126. product: KEYWORD_TO_PRODUCT[it.keyword] || it.product || it.keyword,
  127. sentiment: inferSentiment(it),
  128. }));
  129. return {
  130. meta: {
  131. sourceTier: raw.meta?.sourceTier || 'real-collected',
  132. stage: raw.meta?.stage || 'batch-real',
  133. collectedAt: raw.meta?.collectedAt || new Date().toISOString().slice(0, 10),
  134. product: '江中乳酸菌素片儿童版',
  135. stats: raw.meta || {},
  136. sourceNote: 'docs/乳酸菌/raw/_merged.json · 真实多平台采集',
  137. },
  138. items,
  139. raw,
  140. };
  141. } catch (err) {
  142. console.warn(`⚠ _merged.json 解析失败:${err.message}`);
  143. }
  144. }
  145. return SKELETON;
  146. }
  147. // ------------------------------------------------------------
  148. // 元数据汇总(给 cover / agenda 用)
  149. // ------------------------------------------------------------
  150. function getMeta(data) {
  151. const items = (data && data.items) || [];
  152. const platforms = {};
  153. const products = {};
  154. const hypotheses = {};
  155. const sources = {};
  156. const sentiments = {};
  157. const keywords = new Set();
  158. const tags = new Set();
  159. for (const it of items) {
  160. const pf = it.platform || 'unknown';
  161. platforms[pf] = (platforms[pf] || 0) + 1;
  162. const prod = it.product || 'unknown';
  163. products[prod] = (products[prod] || 0) + 1;
  164. const hs = Array.isArray(it.hypothesis) ? it.hypothesis : (it.hypothesis ? [it.hypothesis] : []);
  165. for (const h of hs) hypotheses[h] = (hypotheses[h] || 0) + 1;
  166. const src = it.source || 'unknown';
  167. sources[src] = (sources[src] || 0) + 1;
  168. sentiments[it.sentiment || 'unknown'] = (sentiments[it.sentiment || 'unknown'] || 0) + 1;
  169. if (it.keyword) keywords.add(it.keyword);
  170. if (Array.isArray(it.tags)) it.tags.forEach((t) => tags.add(t));
  171. }
  172. return {
  173. comments: items.length,
  174. keywords: keywords.size,
  175. tagsTotal: tags.size,
  176. platforms,
  177. products,
  178. productsCount: Object.keys(products).length,
  179. hypotheses,
  180. sources,
  181. sentiments,
  182. stage: data?.meta?.stage || 'unknown',
  183. sourceTier: data?.meta?.sourceTier || 'unknown',
  184. collectedAt: data?.meta?.collectedAt || '待采集',
  185. sourceNote: data?.meta?.sourceNote || '',
  186. };
  187. }
  188. // ------------------------------------------------------------
  189. // 筛选接口(与 liver-analyze 一致)
  190. // ------------------------------------------------------------
  191. function filterByHypothesis(items, h) {
  192. return items.filter((it) => {
  193. const hs = Array.isArray(it.hypothesis) ? it.hypothesis : (it.hypothesis ? [it.hypothesis] : []);
  194. return hs.includes(h);
  195. });
  196. }
  197. function filterByProduct(items, product) {
  198. return items.filter((it) => (it.product || '').includes(product) || (it.keyword || '').includes(product));
  199. }
  200. function filterByKeyword(items, kw) {
  201. return items.filter((it) => it.keyword === kw);
  202. }
  203. function filterByPlatform(items, platform) {
  204. return items.filter((it) => it.platform === platform);
  205. }
  206. function filterBySentiment(items, sentiment) {
  207. return items.filter((it) => it.sentiment === sentiment);
  208. }
  209. function filterByTag(items, tag) {
  210. return items.filter((it) => Array.isArray(it.tags) && it.tags.some((t) => t.includes(tag)));
  211. }
  212. function filterByContent(items, re) {
  213. const rx = re instanceof RegExp ? re : new RegExp(String(re), 'i');
  214. return items.filter((it) => rx.test(String(it.content || '')));
  215. }
  216. function filterByMinLikes(items, min = 1) {
  217. return items.filter((it) => (it.likes || 0) >= min);
  218. }
  219. // ------------------------------------------------------------
  220. // 排序 / 抽样
  221. // ------------------------------------------------------------
  222. function topByLikes(items, n = 10) {
  223. return items.slice().sort((a, b) => (b.likes || 0) - (a.likes || 0)).slice(0, n);
  224. }
  225. function sample(items, n = 6, seed = 1) {
  226. const arr = items.slice();
  227. const result = [];
  228. let s = seed;
  229. while (result.length < n && arr.length) {
  230. s = (s * 9301 + 49297) % 233280;
  231. const idx = Math.floor((s / 233280) * arr.length);
  232. result.push(arr.splice(idx, 1)[0]);
  233. }
  234. return result;
  235. }
  236. function groupByTag(items) {
  237. const map = new Map();
  238. for (const it of items) {
  239. if (!Array.isArray(it.tags)) continue;
  240. for (const t of it.tags) {
  241. if (!map.has(t)) map.set(t, { tag: t, count: 0, items: [] });
  242. const g = map.get(t);
  243. g.count++;
  244. g.items.push(it);
  245. }
  246. }
  247. return Array.from(map.values()).sort((a, b) => b.count - a.count);
  248. }
  249. // ------------------------------------------------------------
  250. // 主查询接口:getEvidence
  251. // ------------------------------------------------------------
  252. function isSubstantive(content, minChars) {
  253. const s = String(content || '').trim();
  254. if (s.length < minChars) return false;
  255. const stripped = s.replace(/\[[^\]]+\]/g, '').replace(/[\s\p{P}\p{Emoji_Presentation}\p{Extended_Pictographic}]/gu, '');
  256. return stripped.length >= Math.max(4, Math.floor(minChars / 2));
  257. }
  258. function getEvidence(items, opts = {}) {
  259. const {
  260. hypothesis, product, keyword, platform, sentiment, tag,
  261. minLikes = 0, minChars = 10, contentMatch, requireContentHit = false,
  262. dedupByContent = true, dedupByNickname = false,
  263. top = 6, seed = 7, sortBy = 'likes',
  264. } = opts;
  265. let filtered = items.slice();
  266. if (hypothesis) filtered = filterByHypothesis(filtered, hypothesis);
  267. if (product) filtered = filterByProduct(filtered, product);
  268. if (keyword) filtered = filterByKeyword(filtered, keyword);
  269. if (platform) filtered = filterByPlatform(filtered, platform);
  270. if (sentiment) filtered = filterBySentiment(filtered, sentiment);
  271. if (tag) filtered = filterByTag(filtered, tag);
  272. if (minLikes) filtered = filterByMinLikes(filtered, minLikes);
  273. filtered = filtered.filter((it) => isSubstantive(it.content, minChars));
  274. if (contentMatch) filtered = filterByContent(filtered, contentMatch);
  275. if (requireContentHit) {
  276. filtered = filtered.filter((it) => {
  277. const c = String(it.content || '').toLowerCase();
  278. const kw = String(it.keyword || '').toLowerCase();
  279. if (!kw) return true;
  280. const fragments = kw.split(/\s+|儿童|宝宝|益生菌/).filter((x) => x.length >= 2);
  281. if (fragments.length === 0) return c.includes(kw);
  282. return fragments.some((f) => c.includes(f));
  283. });
  284. }
  285. if (sortBy === 'likes') {
  286. filtered = filtered.sort((a, b) => (b.likes || 0) - (a.likes || 0));
  287. }
  288. if (dedupByContent) {
  289. const seen = new Set();
  290. filtered = filtered.filter((it) => {
  291. const key = dedupByNickname
  292. ? `${(it.content || '').slice(0, 40)}|${it.nickname || ''}`
  293. : (it.content || '').slice(0, 40);
  294. if (seen.has(key)) return false;
  295. seen.add(key);
  296. return true;
  297. });
  298. }
  299. const pool = filtered.slice(0, Math.max(top * 2, top + 3));
  300. return sample(pool, Math.min(top, pool.length), seed);
  301. }
  302. // ------------------------------------------------------------
  303. // 源标识辅助
  304. // ------------------------------------------------------------
  305. function isSeed(item) { return (item?.source || '').includes('pattern') || (item?.source || '').includes('seed'); }
  306. function isReal(item) { return (item?.source || '') === 'real-collected'; }
  307. function getGlobalSourceLabel(meta) {
  308. if (!meta) return '未加载';
  309. const tier = meta.sourceTier;
  310. if (tier === 'real-collected') return '真实采集';
  311. if (tier === 'pattern-curated') return '公开模式归纳 · 种子样本';
  312. return '骨架占位';
  313. }
  314. module.exports = {
  315. HYPOTHESES,
  316. CHAPTER_HYPOTHESIS_MAP,
  317. PLATFORM_LABELS,
  318. loadMerged,
  319. getMeta,
  320. // filters
  321. filterByHypothesis, filterByProduct, filterByKeyword, filterByPlatform,
  322. filterBySentiment, filterByTag, filterByContent, filterByMinLikes,
  323. // sorting/sampling
  324. topByLikes, sample, groupByTag,
  325. // high-level
  326. getEvidence,
  327. isSeed, isReal, getGlobalSourceLabel,
  328. SKELETON,
  329. };