【问题标题】:How to add an "id" to each column in .arff file for k-means clustering in weka如何在 .arff 文件中的每一列中添加一个“id”,以便在 weka 中进行 k-means 聚类
【发布时间】:2013-07-24 02:15:59
【问题描述】:

这是我的 arff 文件(links.arff):

@relation links

@attribute isLink1Present numeric
@attribute isLink2Present numeric
@attribute isLink3Present numeric
@attribute isLink4Present numeric
@attribute isLink6Present numeric
@attribute isLink7Present numeric
@attribute isLink8Present numeric
@attribute isLink9Present numeric

@data
0,0,0,0,0,0,0,0,0
1,0,0,0,0,0,0,0,0
1,0,0,0,0,0,0,0,0
1,1,0,0,0,0,0,0,0
1,0,0,0,0,0,0,0,0
1,0,1,0,0,0,0,0,0
1,1,0,0,0,0,0,0,0
1,1,1,0,0,0,0,0,0
1,0,0,0,0,0,0,0,0
1,0,0,1,0,0,0,0,0
1,0,1,0,0,0,0,0,0
1,0,1,1,0,0,0,0,0
1,1,0,0,0,0,0,0,0
1,1,0,1,0,0,0,0,0
1,1,1,0,0,0,0,0,0
1,1,1,1,0,0,0,0,0
1,0,0,0,0,0,0,0,0
1,0,0,0,1,0,0,0,0
1,0,0,1,0,0,0,0,0
1,0,0,1,1,0,0,0,0
1,0,1,0,0,0,0,0,0
1,0,1,0,1,0,0,0,0
1,0,1,1,0,0,0,0,0
1,0,1,1,1,0,0,0,0
1,1,0,0,0,0,0,0,0
1,1,0,0,1,0,0,0,0
1,1,0,1,0,0,0,0,0
1,1,0,1,1,0,0,0,0
1,1,1,0,0,0,0,0,0
1,1,1,0,1,0,0,0,0
1,1,1,1,0,0,0,0,0
1,1,1,1,1,0,0,0,0
1,0,0,0,0,0,0,0,0
1,0,0,0,0,1,0,0,0
1,0,0,0,1,0,0,0,0
1,0,0,0,1,1,0,0,0
1,0,0,1,0,0,0,0,0
1,0,0,1,0,1,0,0,0
1,0,0,1,1,0,0,0,0
1,0,0,1,1,1,0,0,0
1,0,1,0,0,0,0,0,0
1,0,1,0,0,1,0,0,0
1,0,1,0,1,0,0,0,0
1,0,1,0,1,1,0,0,0
1,0,1,1,0,0,0,0,0
1,0,1,1,0,1,0,0,0
1,0,1,1,1,0,0,0,0
1,0,1,1,1,1,0,0,0
1,1,0,0,0,0,0,0,0
1,1,0,0,0,1,0,0,0
1,1,0,0,1,0,0,0,0
1,1,0,0,1,1,0,0,0
1,1,0,1,0,0,0,0,0
1,1,0,1,0,1,0,0,0
1,1,0,1,1,0,0,0,0
1,1,0,1,1,1,0,0,0
1,1,1,0,0,0,0,0,0
1,1,1,0,0,1,0,0,0
1,1,1,0,1,0,0,0,0
1,1,1,0,1,1,0,0,0
1,1,1,1,0,0,0,0,0
1,1,1,1,0,1,0,0,0
1,1,1,1,1,0,0,0,0
1,1,1,1,1,1,0,0,0
1,0,0,0,0,0,0,0,0
1,0,0,0,0,0,1,0,0
1,0,0,0,0,1,0,0,0
1,0,0,0,0,1,1,0,0
1,0,0,0,1,0,0,0,0
1,0,0,0,1,0,1,0,0
1,0,0,0,1,1,0,0,0
1,0,0,0,1,1,1,0,0
1,0,0,1,0,0,0,0,0
1,0,0,1,0,0,1,0,0
1,0,0,1,0,1,0,0,0
1,0,0,1,0,1,1,0,0
1,0,0,1,1,0,0,0,0
1,0,0,1,1,0,1,0,0
1,0,0,1,1,1,0,0,0
1,0,0,1,1,1,1,0,0
1,0,1,0,0,0,0,0,0
1,0,1,0,0,0,1,0,0
1,0,1,0,0,1,0,0,0
1,0,1,0,0,1,1,0,0
1,0,1,0,1,0,0,0,0
1,0,1,0,1,0,1,0,0
1,0,1,0,1,1,0,0,0
1,0,1,0,1,1,1,0,0
1,0,1,1,0,0,0,0,0
1,0,1,1,0,0,1,0,0
1,0,1,1,0,1,0,0,0
1,0,1,1,0,1,1,0,0
1,0,1,1,1,0,0,0,0
1,0,1,1,1,0,1,0,0
1,0,1,1,1,1,0,0,0
1,0,1,1,1,1,1,0,0
1,1,0,0,0,0,0,0,0
1,1,0,0,0,0,1,0,0
1,1,0,0,0,1,0,0,0
1,1,0,0,0,1,1,0,0

这是我实现 k-means 的方式:

public void runKMeans(int numClusters){
    try {
        SimpleKMeans kmeans = new SimpleKMeans();

        //DistanceFunction df = new weka.core.ManhattanDistance();
        DistanceFunction df = new weka.core.EuclideanDistance();

        kmeans.setDistanceFunction(df);
        kmeans.setSeed(10);

        kmeans.setPreserveInstancesOrder(true);
        kmeans.setNumClusters(numClusters);

        String arffFile = new PropertyUtils().getProperty("datafiles-home")+"\\links.arff";
        DataSource source = new DataSource(arffFile);
        Instances instances = source.getDataSet();

        //inst.setDataset(instances);
        kmeans.buildClusterer(instances);
        System.out.println(kmeans.displayStdDevsTipText());

        // This array returns the cluster number (starting with 0) for each instance
        // The array has as many elements as the number of instances
        int[] assignments = kmeans.getAssignments();

        int i=0;

        List<Cluster> lc = new ArrayList<Cluster>();
        for(int clusterNum : assignments) {
            lc.add(new Cluster((i+1) , clusterNum));
          //  System.out.println("Instance "+(i+1)+" -> Cluster "+clusterNum);
            i++;

        }
        Collections.sort(lc);

        for(Cluster c : lc){
            PrintUtils.println("Instance : "+c.getInstance()+" Cluster "+c.getCluster());
        }

        }
        catch(Exception e){
            e.printStackTrace();
        }
}

我想将每一列数据与一个“名称”属性相关联,这样我就可以识别每一列。我怎样才能做到这一点?我不认为我可以将 String 属性添加到@data,因为这会影响 k-means 算法的实现?还有其他方法吗?

【问题讨论】:

    标签: java weka k-means


    【解决方案1】:

    是的,您可以添加一个附加属性来命名实例。

    然后,对于EuclideanDistance,您可以使用-R 选项或setAttributeIndices 来决定要用于计算距离的属性范围。删除name 属性将起作用!

    【讨论】:

      猜你喜欢
      • 2011-08-13
      • 2012-11-08
      • 2012-03-24
      • 2016-01-01
      • 2012-11-05
      • 2020-10-30
      • 2011-10-04
      • 2019-09-19
      • 2019-03-16
      相关资源
      最近更新 更多