HTTP与HTTPS加载同词典文件时搜索结果不一致的问题求助
汉字词典查询脚本本地与CDN加载结果不一致问题
我修改了一款汉字词典查询脚本,遇到一个问题:在loadDictData()方法中,从localhost加载词典数据文件时,search方法返回结果正常;但切换为从CDN加载词典数据文件(如cedict_ts.u8)时,search方法返回的结果和本地情况不一致。
本地测试步骤
git clone https://github.com/cschiller/zhongwen.git cd zhongwen python -m http.server
在浏览器中粘贴下方测试代码,可得到正确结果(本地测试结果截图);若将代码中的host切换为CDN地址,打开新标签页运行脚本,得到的结果与本地测试不同(CDN测试结果截图)。
测试代码
let host = "http://127.0.0.1:8000"; async function loadDictData() { let wordIndex = fetch(`${host}/data/cedict.idx`).then(r => r.text()); let grammarKeywords = fetch(`${host}/data/grammarKeywordsMin.json`).then(r => r.json()); let vocabKeywords = fetch(`${host}/data/vocabularyKeywordsMin.json`).then(r => r.json()); // 注释或取消注释此行来测试结果差异 host = "https://cdn.jsdelivr.net/gh/cschiller/zhongwen@latest"; let wordDict = fetch(`${host}/data/cedict_ts.u8`).then(r => r.text()); return Promise.all([wordDict, wordIndex, grammarKeywords, vocabKeywords]); } class ZhongwenDictionary { constructor(wordDict, wordIndex, grammarKeywords, vocabKeywords) { this.wordDict = wordDict; this.wordIndex = wordIndex; this.grammarKeywords = grammarKeywords; this.vocabKeywords = vocabKeywords; this.cache = {}; } static find(needle, haystack) { let beg = 0; let end = haystack.length - 1; while (beg < end) { let mi = Math.floor((beg + end) / 2); let i = haystack.lastIndexOf('\n', mi) + 1; let mis = haystack.substr(i, needle.length); if (needle < mis) { end = i - 1; } else if (needle > mis) { beg = haystack.indexOf('\n', mi + 1) + 1; } else { return haystack.substring(i, haystack.indexOf('\n', mi + 1)); } } return null; } hasGrammarKeyword(keyword) { return this.grammarKeywords[keyword]; } hasVocabKeyword(keyword) { return this.vocabKeywords[keyword]; } wordSearch(word, max) { let entry = { data: [] }; let dict = this.wordDict; let index = this.wordIndex; let maxTrim = max || 7; let count = 0; let maxLen = 0; WHILE: while (word.length > 0) { let ix = this.cache[word]; if (!ix) { ix = ZhongwenDictionary.find(word + ',', index); if (!ix) { this.cache[word] = []; continue; } ix = ix.split(','); this.cache[word] = ix; } for (let j = 1; j < ix.length; ++j) { let offset = ix[j]; let dentry = dict.substring(offset, dict.indexOf('\n', offset)); if (count >= maxTrim) { entry.more = 1; break WHILE; } ++count; if (maxLen === 0) { maxLen = word.length; } entry.data.push([dentry, word]); } word = word.substr(0, word.length - 1); } if (entry.data.length === 0) { return null; } entry.matchLen = maxLen; return entry; } } async function loadDictionary() { const [wordDict, wordIndex, grammarKeywords, vocabKeywords] = await loadDictData(); return new ZhongwenDictionary(wordDict, wordIndex, grammarKeywords, vocabKeywords); } let dict; await loadDictionary().then(r => dict = r); function search(text) { if (!dict) { return; } let entry = dict.wordSearch(text); console.log("entry", entry); if (entry) { for (let i = 0; i < entry.data.length; i++) { let word = entry.data[i][1]; if (dict.hasGrammarKeyword(word) && (entry.matchLen === word.length)) { // 最终索引应为最大长度匹配的最后一项 entry.grammar = { keyword: word, index: i }; } if (dict.hasVocabKeyword(word) && (entry.matchLen === word.length)) { // 最终索引应为最大长度匹配的最后一项 entry.vocab = { keyword: word, index: i }; } } } return entry; } let res = search("你好"); console.log(res);
内容的提问来源于stack exchange,提问作者krmani
相关产品推荐
相关产品推荐

