Java假新闻检测项目:StringToWordVector向量化报错及入模问题
我正在开发一个Java假新闻检测应用,数据集包含Text(新闻文章)和Label(0:假新闻/1:真实新闻)两列,已转换为JSON文件。用Regex预处理文本(转小写、去URL/HTML标签、特殊字符)后,尝试用Weka的StringToWordVector做文本向量化时遇到错误,注释掉instance.setValue(textAttr, vectorizedInstance.instance(0).stringValue(0));后程序能运行,但不知道怎么正确完成向量化并把数据喂给模型。
错误信息
java.lang.IllegalArgumentException: Attribute isn't nominal, string or date!
at weka.core.AbstractInstance.stringValue(AbstractInstance.java:674)
at weka.core.AbstractInstance.stringValue(AbstractInstance.java:644)
at fnd.DataProcessor.main(DataProcessor.java:60)
相关代码
DataProcessor.java
package fnd; import com.fasterxml.jackson.databind.ObjectMapper; import weka.core.Attribute; import weka.core.DenseInstance; import weka.core.Instance; import weka.core.Instances; import weka.core.converters.ArffSaver; import java.io.File; import java.io.IOException; import java.util.ArrayList; public class DataProcessor { public static void main(String[] args) { try { // Specify the path to your JSON file containing news data String jsonFilePath = "src/main/resources/fnd_output.json"; // Create ObjectMapper instance to read JSON ObjectMapper objectMapper = new ObjectMapper(); // Deserialize JSON array into an array of News objects News[] newsArray = objectMapper.readValue(new File(jsonFilePath), News[].class); // Prepare attributes for the Instances ArrayList<Attribute> attributes = new ArrayList<>(); attributes.add(new Attribute("text", (ArrayList<String>) null)); // Text attribute as string // Define nominal values for the label attribute ArrayList<String> labelValues = new ArrayList<>(); labelValues.add("positive"); labelValues.add("negative"); Attribute labelAttribute = new Attribute("label", labelValues); // Label attribute as nominal attributes.add(labelAttribute); // Create an empty Instances object Instances instances = new Instances("TextInstances", attributes, 0); // Set the index of the class attribute (label attribute) instances.setClassIndex(attributes.size() - 1); // Process each News object and add to Instances for (News news : newsArray) { String processedText = TextPreprocessor.preprocessText(news.getText()); // Vectorize the processed text Instances vectorizedInstance = TextVectorization.vectorizeText(processedText); // Create a new Instance Instance instance = new DenseInstance(attributes.size()); // Set the dataset for the instance instance.setDataset(instances); // Handle text attribute (assuming it's a string attribute) Attribute textAttr = attributes.get(0); if (textAttr.isString()) { instance.setValue(textAttr, vectorizedInstance.instance(0).stringValue(0)); } else { System.err.println("Text attribute is not a string attribute."); } // Handle label attribute (assuming it's a nominal attribute) Attribute labelAttr = labelAttribute; if (labelAttr.isNominal()) { instance.setValue(labelAttr, news.getLabel()); } else { System.err.println("Label attribute is not a nominal attribute."); } // Add the instance to Instances instances.add(instance); } // Output instances to ARFF file ArffSaver arffSaver = new ArffSaver(); arffSaver.setInstances(instances); arffSaver.setFile(new File("vectorized_text_with_labels.arff")); arffSaver.writeBatch(); System.out.println("Text vectorization complete with labels. Saved as vectorized_text_with_labels.arff"); } catch (IOException e) { e.printStackTrace(); } catch (Exception ex) { ex.printStackTrace(); } } }
TextVectorization.java
package fnd; import java.io.File;import java.io.IOException;import java.util.ArrayList; import weka.core.Attribute;import weka.core.Instances;import weka.core.DenseInstance;import weka.core.converters.ArffSaver;import weka.filters.Filter;import weka.filters.unsupervised.attribute.StringToWordVector;import weka.core.Instance; public class TextVectorization { // Method to perform text vectorization (convert string to word vector) public static Instances vectorizeText(String text) throws Exception { // Create ArrayList to hold attributes ArrayList<Attribute> attributes = new ArrayList<>(); // Create a single attribute named "text" Attribute textAttribute = new Attribute("text", (ArrayList<String>) null); attributes.add(textAttribute); // Create Instances object with the specified attribute Instances instances = new Instances("TextInstances", attributes, 0); instances.setClass(textAttribute); // Set the class attribute to "text" // Create a new Instance with the provided text // Create a new Instance Instance instance = new DenseInstance(instances.numAttributes()); instance.setValue(textAttribute, text); instances.add(instance); // Apply StringToWordVector filter to vectorize the text StringToWordVector filter = new StringToWordVector(); filter.setInputFormat(instances); Instances vectorizedData = Filter.useFilter(instances, filter); return vectorizedData; } public static void saveInstancesToArff(Instances instances, String filename) throws IOException { ArffSaver arffSaver = new ArffSaver(); arffSaver.setInstances(instances); arffSaver.setFile(new File(filename)); arffSaver.writeBatch(); } }
TextPreprocessor.java
package fnd; import java.util.regex.Matcher;import java.util.regex.Pattern; public class TextPreprocessor { private static final Pattern URL_PATTERN = Pattern.compile("http[s]?://\\S+|www\\.\\S+"); private static final Pattern HTML_TAG_PATTERN = Pattern.compile("<[^>]+>"); public static String preprocessText(String text) { if (text == null || text.isEmpty()) { return ""; } // Convert text to lowercase text = text.toLowerCase(); // Remove URLs and HTML tags text = removeUrlsAndHtmlTags(text); // Remove non-word characters (except spaces), digits, and newline characters text = removeSpecialCharacters(text); return text; } private static String removeUrlsAndHtmlTags(String text) { Matcher urlMatcher = URL_PATTERN.matcher(text); text = urlMatcher.replaceAll(""); Matcher htmlTagMatcher = HTML_TAG_PATTERN.matcher(text); text = htmlTagMatcher.replaceAll(""); return text; } private static String removeSpecialCharacters(String text) { StringBuilder processedText = new StringBuilder(text.length()); for (char ch : text.toCharArray()) { if (Character.isLetter(ch) || Character.isWhitespace(ch)) { processedText.append(ch); } } return processedText.toString(); } }
问题原因与解决方法
错误原因
调用vectorizedInstance.instance(0).stringValue(0)时出错,因为经过StringToWordVector处理后,原字符串属性已经被转换成数值型的词向量属性,不再是字符串类型,所以无法用stringValue()方法获取值。同时,单独对每个文本做向量化会导致每个样本的特征维度不一致,后续模型无法处理。
正确实现步骤
- 先收集所有预处理后的文本,统一构建Instances,再整体做向量化,确保所有样本的特征维度一致。
- 把文本和标签一起放进Instances后再应用过滤器,避免属性不匹配问题。
修改后的DataProcessor.java代码
package fnd; import com.fasterxml.jackson.databind.ObjectMapper; import weka.core.*; import weka.core.converters.ArffSaver; import weka.filters.Filter; import weka.filters.unsupervised.attribute.StringToWordVector; import java.io.File; import java.io.IOException; import java.util.ArrayList; public class DataProcessor { public static void main(String[] args) { try { String jsonFilePath = "src/main/resources/fnd_output.json"; ObjectMapper objectMapper = new ObjectMapper(); News[] newsArray = objectMapper.readValue(new File(jsonFilePath), News[].class); // 1. 定义原始属性:文本(字符串)+ 标签(标称) ArrayList<Attribute> attributes = new ArrayList<>(); attributes.add(new Attribute("text", (ArrayList<String>) null)); ArrayList<String> labelValues = new ArrayList<>(); labelValues.add("0"); // 匹配数据集的假新闻标签 labelValues.add("1"); // 匹配数据集的真实新闻标签 Attribute labelAttribute = new Attribute("label", labelValues); attributes.add(labelAttribute); Instances rawInstances = new Instances("RawNewsData", attributes, newsArray.length); rawInstances.setClassIndex(1); // 设置标签为分类属性 // 2. 批量添加预处理后的文本和标签 for (News news : newsArray) { String processedText = TextPreprocessor.preprocessText(news.getText()); Instance instance = new DenseInstance(2); instance.setValue(attributes.get(0), processedText); instance.setValue(labelAttribute, String.valueOf(news.getLabel())); rawInstances.add(instance); } // 3. 整体应用StringToWordVector过滤器 StringToWordVector filter = new StringToWordVector(); // 可按需添加参数:停用词、TF-IDF转换等 // filter.setStopwords(new File("stopwords.txt")); // filter.setTFTransform(true); // filter.setIDFTransform(true); filter.setInputFormat(rawInstances); Instances vectorizedInstances = Filter.useFilter(rawInstances, filter); // 4. 保存向量化后的数据集 ArffSaver arffSaver = new ArffSaver(); arffSaver.setInstances(vectorizedInstances); arffSaver.setFile(new File("vectorized_news_data.arff")); arffSaver.writeBatch(); System.out.println("文本向量化完成,已保存为vectorized_news_data.arff"); // 5. 后续可直接用vectorizedInstances训练模型 // 示例:NaiveBayes nb = new NaiveBayes(); // nb.buildClassifier(vectorizedInstances); } catch (IOException e) { e.printStackTrace(); } catch (Exception ex) { ex.printStackTrace(); } } }
关键说明
- 统一向量化:所有样本一起处理,确保特征维度一致,这是模型训练的必要前提。
- 标签匹配:标称属性的取值要和数据集的标签值完全对应,避免赋值失败。
- 参数优化:可根据需求给
StringToWordVector添加停用词、TF-IDF转换等参数,提升向量化效果。
内容的提问来源于stack exchange,提问作者DIJIN

