spark取一个班级的排名topN

该博客介绍了如何利用Java和Scala两种语言在Spark平台上实现计算一个班级成绩的TopN排名。通过示例代码展示了具体的操作步骤和测试过程。
摘要由CSDN通过智能技术生成

java:

package cn.spark.sparktest;



import java.util.Arrays;
import java.util.Iterator;

import org.apache.spark.SparkConf;
import org.apache.spark.api.java.JavaPairRDD;
import org.apache.spark.api.java.JavaRDD;
import org.apache.spark.api.java.JavaSparkContext;
import org.apache.spark.api.java.function.PairFunction;
import org.apache.spark.api.java.function.VoidFunction;

import scala.Tuple2;

/**
 * 分组取top3
 * @author Administrator
 *
 */
public class Topic3 {

    public static void main(String[] args) {
        SparkConf conf = new SparkConf()
                .setAppName("Top3")
                .setMaster("local");
        JavaSparkContext sc = new JavaSparkContext(conf);

        JavaRDD<String> lines = sc.textFile("C://Users//Administrator//Desktop//score.txt");

        JavaPairRDD<String, Integer> pairs = lines.mapToPair(

                new PairFunction<String, String, Integer>() {

                    private static final long serialVersionUID = 1L;

                    @Override
                    public Tuple2<String, Integer> call(String line) throws Exception {
                        String[] lineSplited = line.split(" ");
                        return new Tuple2<String, Integer>(lineSplited[0],
                                Integer.valueOf(lineSplited[1]));
                    }

                });

        JavaPairRDD<String, Iterable<Integer>> groupedPairs = pairs.groupByKey();

        JavaPairRDD<String, Iterable<Integer>> top3Score = groupedPairs.mapToPair(

                new PairFunction<Tuple2<String,Iterable<Integer>>, String, Iterable<Integer>>() {

                    private static final long serialVersionUID = 1L;

                    @Override
                    public Tuple2<String, Iterable<Integer>> call(
                            Tuple2<String, Iterable<Integer>> classScores)
                            throws Exception {
                        Integer[] top3 = new Integer[3];

                        String className = classScores._1;
                        Iterator<Integer> scores = classScores._2.iterator();

                        while(scores.hasNext()) {
                            Integer score = scores.next();

                            for(int i = 0; i < 3; i++) {
                                if(top3[i] == null) {
                                    top3[i] = score;
                                    break;
                                } else if(score > top3[i]) {
                                    for(int j = 2; j > i; j--) {
                                        top3[j] = top3[j - 1];
                                    }

                                    top3[i] = score;

                                    break;
                                }
                            }
                        }

                        return new Tuple2<String,
                                Iterable<Integer>>(className, Arrays.asList(top3));
                    }

                });

        top3Score.foreach(new VoidFunction<Tuple2<String,Iterable<Integer>>>() {

            private static final long serialVersionUID = 1L;

            @Override
            public void call(Tuple2<String, Iterable<Integer>> t) throws Exception {
                System.out.println("class: " + t._1);
                Iterator<Integer> scoreIterator = t._2.iterator();
                while(scoreIterator.hasNext()) {
                    Integer score = scoreIterator.next();
                    System.out.println(score);
                }
                System.out.println("=======================================");
            }

        });

        sc.close();
    }

}

测试:

scala:

package cn.spark.study.core

import org.apache.spark.{SparkConf, SparkContext}

object GroupTop3 {
  def main(args: Array[String]): Unit = {
    val conf = new SparkConf()
      .setAppName("GroupTop3")
      .setMaster("local")
    val sc = new SparkContext(conf)

    val lines = sc.textFile("C://Users//gaochen//Desktop//score.txt")

    val line = lines.map(line => (line.split(" ")(0),
      line.split(" ")(1).toInt))
    val groups = line.groupByKey()
    val groupSort = groups.map(tu =>{
      val key = tu._1
      val value = tu._2
      val sortValues = value.toList.sortWith(_ > _).take(3)
      (key,sortValues)
    })
    groupSort.sortBy(tu => tu._1,false,1).foreach(x =>{
      println(x._1)
      x._2.foreach((v => println("\t"+ v)))
      println("================")
    })
    sc.stop()
  }
}

测试:

评论
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值