You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Java假新闻检测项目:StringToWordVector向量化报错及入模问题

Java假新闻检测应用文本向量化问题

我正在开发一个Java假新闻检测应用,数据集包含Text(新闻文章)和Label(0:假新闻/1:真实新闻)两列,已转换为JSON文件。用Regex预处理文本(转小写、去URL/HTML标签、特殊字符)后,尝试用Weka的StringToWordVector做文本向量化时遇到错误,注释掉instance.setValue(textAttr, vectorizedInstance.instance(0).stringValue(0));后程序能运行,但不知道怎么正确完成向量化并把数据喂给模型。

错误信息

java.lang.IllegalArgumentException: Attribute isn't nominal, string or date!
at weka.core.AbstractInstance.stringValue(AbstractInstance.java:674)
at weka.core.AbstractInstance.stringValue(AbstractInstance.java:644)
at fnd.DataProcessor.main(DataProcessor.java:60)

相关代码

DataProcessor.java

package fnd;

import com.fasterxml.jackson.databind.ObjectMapper;
import weka.core.Attribute;
import weka.core.DenseInstance;
import weka.core.Instance;
import weka.core.Instances;
import weka.core.converters.ArffSaver;

import java.io.File;
import java.io.IOException;
import java.util.ArrayList;

public class DataProcessor {

    public static void main(String[] args) {
        try {
            // Specify the path to your JSON file containing news data
            String jsonFilePath = "src/main/resources/fnd_output.json";

            // Create ObjectMapper instance to read JSON
            ObjectMapper objectMapper = new ObjectMapper();

            // Deserialize JSON array into an array of News objects
            News[] newsArray = objectMapper.readValue(new File(jsonFilePath), News[].class);

            // Prepare attributes for the Instances
            ArrayList<Attribute> attributes = new ArrayList<>();
            attributes.add(new Attribute("text", (ArrayList<String>) null)); // Text attribute as string

            // Define nominal values for the label attribute
            ArrayList<String> labelValues = new ArrayList<>();
            labelValues.add("positive");
            labelValues.add("negative");
            Attribute labelAttribute = new Attribute("label", labelValues); // Label attribute as nominal
            attributes.add(labelAttribute);

            // Create an empty Instances object
            Instances instances = new Instances("TextInstances", attributes, 0);

            // Set the index of the class attribute (label attribute)
            instances.setClassIndex(attributes.size() - 1);

            // Process each News object and add to Instances
            for (News news : newsArray) {
                String processedText = TextPreprocessor.preprocessText(news.getText());

                // Vectorize the processed text
                Instances vectorizedInstance = TextVectorization.vectorizeText(processedText);

                // Create a new Instance
                Instance instance = new DenseInstance(attributes.size());

                // Set the dataset for the instance
                instance.setDataset(instances);

                // Handle text attribute (assuming it's a string attribute)
                Attribute textAttr = attributes.get(0);
                if (textAttr.isString()) {
                    instance.setValue(textAttr, vectorizedInstance.instance(0).stringValue(0));
                } else {
                    System.err.println("Text attribute is not a string attribute.");
                }

                // Handle label attribute (assuming it's a nominal attribute)
                Attribute labelAttr = labelAttribute;
                if (labelAttr.isNominal()) {
                    instance.setValue(labelAttr, news.getLabel());
                } else {
                    System.err.println("Label attribute is not a nominal attribute.");
                }

                // Add the instance to Instances
                instances.add(instance);
            }

            // Output instances to ARFF file
            ArffSaver arffSaver = new ArffSaver();
            arffSaver.setInstances(instances);
            arffSaver.setFile(new File("vectorized_text_with_labels.arff"));
            arffSaver.writeBatch();

            System.out.println("Text vectorization complete with labels. Saved as vectorized_text_with_labels.arff");

        } catch (IOException e) {
            e.printStackTrace();
        } catch (Exception ex) {
            ex.printStackTrace();
        }
    }
}

TextVectorization.java

package fnd;

import java.io.File;import java.io.IOException;import java.util.ArrayList;

import weka.core.Attribute;import weka.core.Instances;import weka.core.DenseInstance;import weka.core.converters.ArffSaver;import weka.filters.Filter;import weka.filters.unsupervised.attribute.StringToWordVector;import weka.core.Instance;

public class TextVectorization {

 // Method to perform text vectorization (convert string to word vector)
public static Instances vectorizeText(String text) throws Exception {
    // Create ArrayList to hold attributes
    ArrayList<Attribute> attributes = new ArrayList<>();
    
    // Create a single attribute named "text"
    Attribute textAttribute = new Attribute("text", (ArrayList<String>) null);
    attributes.add(textAttribute);
    
    // Create Instances object with the specified attribute
    Instances instances = new Instances("TextInstances", attributes, 0);
    instances.setClass(textAttribute); // Set the class attribute to "text"
    
    // Create a new Instance with the provided text
 // Create a new Instance
    Instance instance = new DenseInstance(instances.numAttributes());
    instance.setValue(textAttribute, text);
    instances.add(instance);
    
    // Apply StringToWordVector filter to vectorize the text
    StringToWordVector filter = new StringToWordVector();
    filter.setInputFormat(instances);
    Instances vectorizedData = Filter.useFilter(instances, filter);
    
    return vectorizedData;
}

public static void saveInstancesToArff(Instances instances, String filename) throws IOException {
    ArffSaver arffSaver = new ArffSaver();
    arffSaver.setInstances(instances);
    arffSaver.setFile(new File(filename));
    arffSaver.writeBatch();
}
}

TextPreprocessor.java

package fnd;

import java.util.regex.Matcher;import java.util.regex.Pattern;

public class TextPreprocessor {

private static final Pattern URL_PATTERN = Pattern.compile("http[s]?://\\S+|www\\.\\S+");
private static final Pattern HTML_TAG_PATTERN = Pattern.compile("<[^>]+>");

public static String preprocessText(String text) {
    if (text == null || text.isEmpty()) {
        return "";
    }

    // Convert text to lowercase
    text = text.toLowerCase();

    // Remove URLs and HTML tags
    text = removeUrlsAndHtmlTags(text);

    // Remove non-word characters (except spaces), digits, and newline characters
    text = removeSpecialCharacters(text);

    return text;
}

private static String removeUrlsAndHtmlTags(String text) {
    Matcher urlMatcher = URL_PATTERN.matcher(text);
    text = urlMatcher.replaceAll("");

    Matcher htmlTagMatcher = HTML_TAG_PATTERN.matcher(text);
    text = htmlTagMatcher.replaceAll("");

    return text;
}

private static String removeSpecialCharacters(String text) {
    StringBuilder processedText = new StringBuilder(text.length());

    for (char ch : text.toCharArray()) {
        if (Character.isLetter(ch) || Character.isWhitespace(ch)) {
            processedText.append(ch);
        }
    }

    return processedText.toString();
}
}

问题原因与解决方法

错误原因

调用vectorizedInstance.instance(0).stringValue(0)时出错,因为经过StringToWordVector处理后,原字符串属性已经被转换成数值型的词向量属性,不再是字符串类型,所以无法用stringValue()方法获取值。同时,单独对每个文本做向量化会导致每个样本的特征维度不一致,后续模型无法处理。

正确实现步骤

  1. 先收集所有预处理后的文本,统一构建Instances,再整体做向量化,确保所有样本的特征维度一致。
  2. 把文本和标签一起放进Instances后再应用过滤器,避免属性不匹配问题。

修改后的DataProcessor.java代码

package fnd;

import com.fasterxml.jackson.databind.ObjectMapper;
import weka.core.*;
import weka.core.converters.ArffSaver;
import weka.filters.Filter;
import weka.filters.unsupervised.attribute.StringToWordVector;

import java.io.File;
import java.io.IOException;
import java.util.ArrayList;

public class DataProcessor {

    public static void main(String[] args) {
        try {
            String jsonFilePath = "src/main/resources/fnd_output.json";
            ObjectMapper objectMapper = new ObjectMapper();
            News[] newsArray = objectMapper.readValue(new File(jsonFilePath), News[].class);

            // 1. 定义原始属性:文本(字符串)+ 标签(标称)
            ArrayList<Attribute> attributes = new ArrayList<>();
            attributes.add(new Attribute("text", (ArrayList<String>) null));
            ArrayList<String> labelValues = new ArrayList<>();
            labelValues.add("0"); // 匹配数据集的假新闻标签
            labelValues.add("1"); // 匹配数据集的真实新闻标签
            Attribute labelAttribute = new Attribute("label", labelValues);
            attributes.add(labelAttribute);

            Instances rawInstances = new Instances("RawNewsData", attributes, newsArray.length);
            rawInstances.setClassIndex(1); // 设置标签为分类属性

            // 2. 批量添加预处理后的文本和标签
            for (News news : newsArray) {
                String processedText = TextPreprocessor.preprocessText(news.getText());
                Instance instance = new DenseInstance(2);
                instance.setValue(attributes.get(0), processedText);
                instance.setValue(labelAttribute, String.valueOf(news.getLabel()));
                rawInstances.add(instance);
            }

            // 3. 整体应用StringToWordVector过滤器
            StringToWordVector filter = new StringToWordVector();
            // 可按需添加参数:停用词、TF-IDF转换等
            // filter.setStopwords(new File("stopwords.txt"));
            // filter.setTFTransform(true);
            // filter.setIDFTransform(true);
            filter.setInputFormat(rawInstances);
            Instances vectorizedInstances = Filter.useFilter(rawInstances, filter);

            // 4. 保存向量化后的数据集
            ArffSaver arffSaver = new ArffSaver();
            arffSaver.setInstances(vectorizedInstances);
            arffSaver.setFile(new File("vectorized_news_data.arff"));
            arffSaver.writeBatch();

            System.out.println("文本向量化完成,已保存为vectorized_news_data.arff");

            // 5. 后续可直接用vectorizedInstances训练模型
            // 示例:NaiveBayes nb = new NaiveBayes();
            // nb.buildClassifier(vectorizedInstances);

        } catch (IOException e) {
            e.printStackTrace();
        } catch (Exception ex) {
            ex.printStackTrace();
        }
    }
}

关键说明

  • 统一向量化:所有样本一起处理,确保特征维度一致,这是模型训练的必要前提。
  • 标签匹配:标称属性的取值要和数据集的标签值完全对应,避免赋值失败。
  • 参数优化:可根据需求给StringToWordVector添加停用词、TF-IDF转换等参数,提升向量化效果。

内容的提问来源于stack exchange,提问作者DIJIN

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.25 15:35:56