spark 解析非结构化数据存储至hive的scala代码
//提交代码包
// /usr/local/spark/bin$ spark-submit --class "getkv" /data/chun/sparktes.jar import org.apache.spark.sql.{DataFrame, Row, SQLContext, SaveMode}
import org.apache.spark.{SparkConf, SparkContext}
import org.apache.spark.sql.hive.HiveContext
object split {
def main(args:Array[String])
{ val cf = new SparkConf().setAppName("ass").setMaster("local")
val sc = new SparkContext(cf)
val sqlContext = new SQLContext(sc)
val hc = new HiveContext(sc)
val format=new java.text.SimpleDateFormat("yyyy-MM-dd")
val date=format.format(new java.util.Date().getTime-****) val lg= sc.textFile("hdfs://master:9000/data/"+date+"/*/*.gz") val filed1=lg.map(l=>(l.split("android_id\":\"").last.split("\"").head.toString,
l.split("anylst_ver\":").last.split(",").head.toString,
l.split("area\":\"").last.split("\"").head,
l.split("build_CPU_ABI\":\"").last.split("\"").head,
l.split("build_board\":\"").last.split("\"").head,
l.split("build_model\":\"").last.split("\"").head,
l.split("\"city\":\"").last.split("\"").head,
l.split("country\":\"").last.split("\"").head,
l.split("cpuCount\":").last.split(",").head,
l.split("cpuName\":\"").last.split("\"").head,
l.split("custom_uuid\":\"").last.split("\"").head,
l.split("cid\":\"").last.split("\"").head,
l.split("definition\":\"").last.split("\"").head,
l.split("firstTitle\":\"").last.split("\"").head,
l.split("modeType\":\"").last.split("\"").head,
l.split("pageName\":\"").last.split("\"").head,
l.split("playIndex\":\"").last.split("\"").head,
l.split("rectime\":").last.split(",").head,
l.split("time\":\"").last.split("\"").head))
//val F1=filed1.toDF("custom_uuid","region","screenHeight","screenWidth","serial_number","touchMode","umengChannel","vercode","vername","wlan0_mac","rectime","time")
val scoreDataFrame1 = hc.createDataFrame(filed1).toDF("android_id","anylst_ver","area","build_CPU_ABI","build_board","build_model","city","country","cpuCount","cpuName","custom_uuid","cid","definition","firstTitle","modeType","pageName","playIndex","rectime","time")
scoreDataFrame1.write.mode(SaveMode.Append).saveAsTable("test.f1") val filed2=lg.map(l=>(l.split("custom_uuid\":\"").last.split("\"").head,
l.split("playType\":\"").last.split("\"").head,
l.split("prevName\":\"").last.split("\"").head,
l.split("prevue\":").last.split(",").head,
l.split("siteName\":\"").last.split("\"").head,
l.split("title\":\"").last.split("\"").head,
l.split("uuid\":\"").last.split("\"").head,
l.split("vod_seek\":\"").last.split("\"").head,
l.split("device_id\":\"").last.split("\"").head,
l.split("device_name\":\"").last.split("\"").head,
l.split("dpi\":").last.split(",").head,
l.split("eth0_mac\":\"").last.split("\"").head,
l.split("ip\":\"").last.split("\"").head,
l.split("ipaddr\":\"").last.split("\"").head,
l.split("isp\":\"").last.split("\"").head,
l.split("largeMem\":").last.split(",").head,
l.split("limitMem\":").last.split(",").head,
l.split("packageName\":\"").last.split("\"").head,
l.split("rectime\":").last.split(",").head,
l.split("time\":\"").last.split("\"").head))
import sqlContext.implicits._
val scoreDataFrame2 = hc.createDataFrame(filed2).toDF("custom_uuid","playType","prevName","prevue","siteName","title","uuid","vod_seek","device_id","device_name","dpi","eth0_mac","ip","ipaddr","isp","largeMem","limitMem","packageName","rectime","time")
scoreDataFrame2.write.mode(SaveMode.Append).saveAsTable("test.f2")
// val filed3=lg.map(l=>(l.split("custom_uuid\":\"").last.split("\"").head,
l.split("region\":\"").last.split("\"").head,
l.split("screenHeight\":").last.split(",").head,
l.split("screenWidth\":").last.split(",").head,
l.split("serial_number\":\"").last.split("\"").head,
l.split("touchMode\":").last.split(",").head,
l.split("umengChannel\":\"").last.split("\"").head,
l.split("vercode\":").last.split(",").head,
l.split("vername\":\"").last.split("\"").head,
l.split("wlan0_mac\":\"").last.split("\"").head,
l.split("rectime\":").last.split(",").head,
l.split("time\":\"").last.split("\"").head
)) import sqlContext.implicits._
val scoreDataFrame3= hc.createDataFrame(filed3).toDF("custom_uuid","region","screenHeight","screenWidth","serial_number","touchMode","umengChannel","vercode","vername","wlan0_mac","rectime","time")
scoreDataFrame3.write.mode(SaveMode.Append).saveAsTable("test.f3") }
}
spark 解析非结构化数据存储至hive的scala代码的更多相关文章
- MySQL 5.7:非结构化数据存储的新选择
本文转载自:http://www.innomysql.net/article/23959.html (只作转载, 不代表本站和博主同意文中观点或证实文中信息) 工作10余年,没有一个版本能像MySQL ...
- Spark如何与深度学习框架协作,处理非结构化数据
随着大数据和AI业务的不断融合,大数据分析和处理过程中,通过深度学习技术对非结构化数据(如图片.音频.文本)进行大数据处理的业务场景越来越多.本文会介绍Spark如何与深度学习框架进行协同工作,在大数 ...
- Python爬虫(九)_非结构化数据与结构化数据
爬虫的一个重要步骤就是页面解析与数据提取.更多内容请参考:Python学习指南 页面解析与数据提取 实际上爬虫一共就四个主要步骤: 定(要知道你准备在哪个范围或者网站去搜索) 爬(将所有的网站的内容全 ...
- 结构化数据(structured),半结构化数据(semi-structured),非结构化数据(unstructured)
概念 结构化数据:即行数据,存储在数据库里,可以用二维表结构来逻辑表达实现的数据. 半结构化数据:介于完全结构化数据(如关系型数据库.面向对象数据库中的数据)和完全无结构的数据(如声音.图像文件等)之 ...
- 结构化数据、半结构化数据、非结构化数据——Hadoop处理非结构化数据
刚开始接触Hadoop ,指南中说Hadoop处理非结构化数据,学习数据库的时候,老师总提结构化数据,就是一张二维表,那非结构化数据是什么呢?难道是文本那样的文件?经过上网搜索,感觉这个帖子不错 网址 ...
- Scrapy系列教程(2)------Item(结构化数据存储结构)
Items 爬取的主要目标就是从非结构性的数据源提取结构性数据,比如网页. Scrapy提供 Item 类来满足这种需求. Item 对象是种简单的容器.保存了爬取到得数据. 其提供了 类似于词典(d ...
- hbase非结构化数据库与结构化数据库比较
目的:了解hbase与支持海量数据查询的特性以及实现方式 传统关系型数据库特点及局限 传统数据库事务性特别强,要求数据完整性及安全性,造成系统可用性以及伸缩性大打折扣.对于高并发的访问量,数据库性能不 ...
- 利用Gson和SharePreference存储结构化数据
问题的导入 Android互联网产品通常会有很多的结构化数据需要保存,比如对于登录这个流程,通常会保存诸如username.profile_pic.access_token等等之类的数据,这些数据可以 ...
- Spark读取结构化数据
读取结构化数据 Spark可以从本地CSV,HDFS以及Hive读取结构化数据,直接解析为DataFrame,进行后续分析. 读取本地CSV 需要指定一些选项,比如留header,比如指定delimi ...
随机推荐
- null与“ ”
http://blog.csdn.net/eroswang/article/details/8529817 MySQL数据库是一个基于结构化数据的开源数据库.SQL语句是mysql数据库中核心语言.不 ...
- j2me必备之网络开发数据处理
第9章 无线网络开发MIDP提供了一组通用的网络开发接口,用来针对不同的无线网络应用可以采取不同的开发接口.基于CLDC的网络支持是由统一网络连接框架(Generic Connection Frame ...
- Profiling Java Application with Systemtap
https://laurent-leturgez.com/2017/12/22/profiling-java-application-with-systemtap/ https://myaut.git ...
- 查询返回JSON数据结果集
查询返回JSON数据结果集 设计目标: 1)一次性可以返回N个数据表的JSON数据 2)跨数据库引擎 { "tables": [ { "cols": [ { & ...
- nssm和AlwaysUp来包装exe文件为windows服务
最近遇到要把windows exe文件部署为service,因为原先开发为exe程序,现在有不想修改code改为service,但是部署必须是service服务, 所以我们需要一个包装器来包装exe为 ...
- SSE图像算法优化系列一:一段BGR2Y的SIMD代码解析。
一个同事在github上淘到一个基于SIMD的RGB转Y(彩色转灰度或者转明度)的代码,我抽了点时间看了下,顺便学习了一些SIMD指令,这里把学习过程中的一些理解和认识共享给大家. github上相关 ...
- 郑晔谈 Java 开发:新工具、新框架、新思维【转载】【整理】
原文地址 导语:"我很惊讶地发现,现在许多程序员讨论的内容几乎和我十多年前刚开始做 Java 时几乎完全一样.要知道,我们生存的这个行业号称是变化飞快的.其实,这十几年时间,在开发领域已经有 ...
- linux 切分文件
linux经常需要处理文件,如果文件比较大,那么需要切分成为若干的小文件再处理. 命令:split 比如有一个文件: ll -h 1431531915758 -rw-r--r-- 1 ticketde ...
- VirtualBox中出现UUID have already exists 解决方法
虚拟机更换VDI文件,启动时会出现 "UUID already exists"的错误,这是因为删除虚拟机时候没有选择"删除所有",只是选择移除造成的. 方法一: ...
- 什么是同源策略,什么是跨域,如何跨域,Jsonp/CORS跨域
同源策略 同源策略(Same origin policy)是一种约定,它是浏览器最核心也最基本的安全功能,如果缺少了同源策略,则浏览器的正常功能可能都会受到影响. 可以说Web是构建在同源策略基础之上 ...