通过HA方式操作HDFS
之前操作hdfs的时候,都是固定namenode的地址,然后去操作。这个时候就必须判断namenode的状态为active还是standby,比较繁琐,如果集群使用了HA的形式,就很方便了
直接上代码,看注释:
package com.ideal.template.openbigdata.util; import java.io.IOException;
import java.net.URI;
import java.sql.ResultSet;
import java.sql.ResultSetMetaData;
import java.sql.SQLException;
import java.sql.Timestamp;
import java.text.SimpleDateFormat; import java.util.LinkedList;
import java.util.List; //import org.anarres.lzo.LzoAlgorithm;
//import org.anarres.lzo.LzoDecompressor;
//import org.anarres.lzo.LzoInputStream;
//import org.anarres.lzo.LzoLibrary;
import org.apache.commons.logging.Log;
import org.apache.commons.logging.LogFactory;
import org.apache.hadoop.conf.Configuration;
import org.apache.hadoop.fs.FSDataOutputStream;
import org.apache.hadoop.fs.FileStatus;
import org.apache.hadoop.fs.FileSystem;
import org.apache.hadoop.fs.Path;
import org.apache.hadoop.security.UserGroupInformation;
import org.apache.log4j.Logger; public class HadoopUse
{
private static Log log = LogFactory.getLog(HadoopUse.class); /**
* 设置hdfs配置信息
* @return
*/
private static Configuration getConf()
{
Configuration conf = new Configuration(); //设置配置相关的信息,分别对应hdfs-site.xml core-site.xml
conf.set("fs.defaultFS", "hdfs://dragoncluster");
conf.set("dfs.nameservices", "dragoncluster");
conf.set("dfs.ha.namenodes.dragoncluster", "nn1,nn2");
conf.set("dfs.namenode.rpc-address.dragoncluster.nn1", "n01.dragon.com:8020");
conf.set("dfs.namenode.rpc-address.dragoncluster.nn2", "n02.dragon.com:8020");
conf.set("dfs.client.failover.proxy.provider.dragoncluster", "org.apache.hadoop.hdfs.server.namenode.ha.ConfiguredFailoverProxyProvider"); //设置实现类,因为会出现类覆盖的问题
conf.set("fs.hdfs.impl", org.apache.hadoop.hdfs.DistributedFileSystem.class.getName());
conf.set("fs.file.impl", org.apache.hadoop.fs.LocalFileSystem.class.getName());
return conf;
} /**
* 设置kerberos认证
* @param conf
* @throws Exception
*/
private static void kerberosLogin(Configuration conf) throws Exception
{
conf.set("hadoop.security.authentication", "kerberos");
UserGroupInformation.setConfiguration(conf);
UserGroupInformation.loginUserFromKeytab("openbigdata@DRAGON.COM", "/etc/security/keytabs/openbigdata.keytab");
} public static long getSize(String uri, String user)
{
Path path = new Path(URI.create(uri)); Configuration conf = new Configuration();
try
{
FileSystem fs = FileSystem.get(URI.create(uri), conf, user);
return fs.getContentSummary(path).getLength() / 1024 / 1024; // 单位为MB
}
catch (Exception ex)
{
log.error("HadoopUse.getSize" + ex.getMessage(), ex);
return 0;
}
} /**
* 在hdfs上创建文件,并写入内容
*
* @param uri
* @param content
* @param user
* @return
*/
public static boolean createHdfsFile(String uri, String user, String fullName, String content)
{
if (fullName == null || fullName.length() == 0)
{// 本地路径不正确
return false;
}
if (content == null || content.length() == 0)
{// hdfs路径不正确
return false;
} try
{
Configuration conf = new Configuration(); FileSystem fs = FileSystem.get(URI.create(uri), conf, user);
FSDataOutputStream os = null; if (fs.exists(new Path(fullName)) == true)
{// 如果该路径存在
// os = fs.append(new Path(fullName));
fs.delete(new Path(fullName), true);
}
os = fs.create(new Path(fullName));
os.write(content.getBytes());
os.close();
fs.close();
return true;
}
catch (Exception ex)
{
log.error("HadoopUse.createHdfsFile" + ex.getMessage(), ex);
return false;
}
} /**
* 删除hdfs上的文件
* @param uri
* @param user
* @param fullName
* @return
*/
public static boolean deleteHdfsFile(String uri, String user, String fullName)
{
if (fullName == null || fullName.length() == 0)
{// 本地路径不正确
log.error("HadoopUse.deleteHdfsFile文件名不合法");
return false;
} try
{
Configuration conf = new Configuration(); FileSystem fs = FileSystem.get(URI.create(uri), conf, user);
//FSDataOutputStream os = null; if (fs.exists(new Path(fullName)) == true)
{// 如果该路径存在
// os = fs.append(new Path(fullName));
fs.delete(new Path(fullName), true);
}
return true;
}
catch (Exception ex)
{
log.error("HadoopUse.createHdfsFile" + ex.getMessage(), ex);
}
return false;
} /**
* 根据resultset将值写入到hdfs上
* @param uri
* @param user
* @param fullName
* @param resultSet
* @param terminated
* @return
* @throws InterruptedException
* @throws IOException
* @throws SQLException
*/
public void createHdfsFile(String fullName, ResultSet resultSet, String terminated, FlagUtil flag)
throws IOException, InterruptedException, SQLException, Exception
{
if (resultSet == null)
{ // 如果查询出来的游标为空,直接退出
return;
}
SimpleDateFormat sdf = new SimpleDateFormat("yyyy-MM-dd HH:mm:ss"); FileSystem fs = null;
FSDataOutputStream out = null;
Configuration conf = getConf();
kerberosLogin(conf); fs = FileSystem.get(conf);
if (fs.exists(new Path(fullName)) == true)
{// 如果该路径存在
fs.delete(new Path(fullName), true);
} // 获取文件句柄
out = fs.create(new Path(fullName)); // 写入文件内容
ResultSetMetaData rsmd = resultSet.getMetaData();
int rowCnt = rsmd.getColumnCount();
int count = 0;
while (resultSet.next())
{
count++;
if(count >= 1000)
{//每1000条记录检查一次需要终止任务
if(flag.getTeminalStatus() == true)
{
break;
}
count = 0;
} for (int i = 1; i <= rowCnt; i++)
{
if (resultSet.getObject(i) == null)
{// 如果是空的数据
out.write("".getBytes("utf-8"));
}
else
{
String item = null;
if("DATE".equals(rsmd.getColumnTypeName(i).toUpperCase()))
{//如果是日期类型
Timestamp date = resultSet.getTimestamp(i);
item = sdf.format(date);
}
else
{
item = String.valueOf(resultSet.getObject(i));
}
if (item != null)
{
out.write(item.getBytes("utf-8"));
}
else
{
out.write("".getBytes("utf-8"));
}
}
if (i < rowCnt)
{// 如果写完一列,则插入分隔符
out.write(terminated.getBytes("utf-8"));
}
}
// 切换到下一行
out.write("\r\n".getBytes("utf-8"));
}
log.info("fullName:" + fullName + "写入成功"); if (out != null)
{
out.flush();
out.close();
}
if (fs != null)
{
fs.close();
}
} /**
* 查询路径
* @param path
* @return
* @throws Exception
*/
public static List<String> listDir(String path) throws Exception
{
Configuration conf = getConf();
kerberosLogin(conf);
FileSystem fs = FileSystem.get(conf); Path hdfs = new Path(path);
List<String> pathList = null;
FileStatus files[] = fs.listStatus(hdfs);
if(files!=null && files.length >0)
{
pathList = new LinkedList<String>();
for (FileStatus file : files)
{
pathList.add(file.getPath().toString());
}
}
return pathList;
} public static void main(String[] args) throws Exception
{
List<String> pathList = listDir(args[0]);
for(String path: pathList)
{
System.out.println(path);
}
}
}
注意,这用到了HA,以及kerberos认证,
通过HA方式操作HDFS的更多相关文章
- 使用命令行的方式操作hdfs
必须要用打全路径,没有相对路径的概念,或者cd的概念 打印报告: 所有的命令显示出来: 以下的操作分别是创建创建文件夹,删除文件夹,显示文件夹,可见删除文件夹只能够使用-rmr . 从本地拷贝文件到h ...
- Java API操作HA方式下的Hadoop
通过java api连接Hadoop集群时,如果集群支持HA方式,那么可以通过如下方式设置来自动切换到活动的master节点上.其中,ClusterName 是可以任意指定的,跟集群配置无关,dfs. ...
- 用流的方式来操作hdfs上的文件
import java.io.FileInputStream; import java.io.FileOutputStream; import java.io.IOException; import ...
- 使用javaAPI操作hdfs
欢迎到https://github.com/huabingood/everyDayLanguagePractise查看源码. 一.构建环境 在hadoop的安装包中的share目录中有hadoop所有 ...
- 使用Java方式连接HDFS
IDEA中新建Maven工程,添加POM依赖, 在IDE的提示中, 点击 Import Changes 等待自动下载完成相关的依赖包. <?xml version="1.0" ...
- Hadoop Java API操作HDFS文件系统(Mac)
1.下载Hadoop的压缩包 tar.gz https://mirrors.tuna.tsinghua.edu.cn/apache/hadoop/common/stable/ 2.关联jar包 在 ...
- 使用Java API方式连接HDFS Client测试
IDEA中新建Maven工程,添加POM依赖, 在IDE的提示中, 点击 Import Changes 等待自动下载完成相关的依赖包. <?xml version="1.0" ...
- Linux -- 之HDFS实现自动切换HA(全新HDFS)
Linux -- 之HDFS实现自动切换HA(全新HDFS) JDK规划 1.7及以上 https://blog.csdn.net/meiLin_Ya/article/details/8065094 ...
- (第3篇)HDFS是什么?HDFS适合做什么?我们应该怎样操作HDFS系统?
摘要: 这篇文章会详细介绍HDFS是什么,HDFS的作用,适合和不适合的场景,我们该如何操作HDFS? HDFS文件系统 Hadoop 附带了一个名为 HDFS(Hadoop分布式文件系统)的分布 ...
随机推荐
- [推荐]Silverlight 2 开发者海报
从Brad Abrams的Blog上看到了一张Silverlight 2开发者海报,非常酷,拿出来与大家分享. [JPG版本 5.8MB] [PNG版本 6.5MB] [TIF版本 19.9 MB] ...
- CALayer和UIView
前言 本次分享将从以下方面进行展开: 曾被面试官问倒过的问题:层与视图的关系 CALayer类介绍及层与视图的关系 CAShapeLayer类介绍 UIBezierPath贝塞尔曲线讲解 CoreAn ...
- SPOJ:Strange Waca(不错的搜索&贪心&剪枝)
Waca loves maths,.. a lot. He always think that 1 is an unique number. After playing in hours, Waca ...
- BZOJ_4311_向量_线段树按时间分治
BZOJ_4311_向量_CDQ分治+线段树按时间分治 Description 你要维护一个向量集合,支持以下操作: 1.插入一个向量(x,y) 2.删除插入的第i个向量 3.查询当前集合与(x,y) ...
- 斯坦福CS231n—深度学习与计算机视觉----学习笔记 课时6
课时6 线性分类器损失函数与最优化(上) 多类SVM损失:这是一个两分类支持向量机的泛化 SVM损失计算了所有不正确的例子,将所有不正确的类别的评分,与正确类别的评分之差加1,将得到的数值与0作比较, ...
- c++常见面试题30道
1.new.delete.malloc.free关系 delete会调用对象的析构函数,和new对应free只会释放内存,new调用构造函数.malloc与free是C++/C语言的标准库函数,new ...
- YCOJ-DFS
DFS搜索是搜索中的一种,即深度优先搜索(Depth First Search),其过程简要来说是对每一个可能的分支路径深入到不能再深入为止,而且每个节点只能访问一次. 图示: 如图,这是邻接矩阵,我 ...
- Java中JRE、JDK和JVM的区别
一.三者的基本概念: JRE(Java Development Kit):Java的运行环境: JDK(Java Runtime Enviroment):Java开发工具包: JVM(Java Vir ...
- NSA互联网公开情报收集指南:迷宫中的秘密·上
猫宁!!! 参考链接: https://www.nsa.gov/news-features/declassified-documents/assets/files/Untangling-the-Web ...
- Luogu P1607 庙会班车【线段树】By cellur925
题目传送门 据说可以用贪心做?算了算了...我都不会贪.... 开始想的是用线段树,先建出一颗空树,然后输进区间操作后就维护最大值,显然开始我忽视了班车的容量以及可以有多组奶牛坐在一起的信息. 我们肯 ...