网页抓取
来源:互联网 发布:2014年网络歌曲排行 编辑:程序博客网 时间:2024/05/18 09:17
import java.io.File;
import java.io.FileWriter;
import java.io.IOException;
import java.io.PrintWriter;
public class Test {
private static String getStaticPage(String surl) {
String htmlContent = "";
try {
java.io.InputStream inputStream;
java.net.URL url = new java.net.URL(surl);
java.net.HttpURLConnection connection = (java.net.HttpURLConnection) url.openConnection();
connection.connect();
inputStream = connection.getInputStream();
byte[] bytes = new byte[1024 * 2000];
int index = 0;
int count = inputStream.read(bytes, index, 1024 * 2000);
while (count != -1) {
index += count;
count = inputStream.read(bytes, index, 1);
}
htmlContent = new String(bytes, "UTF-8");
connection.disconnect();
} catch (Exception ex) {
ex.printStackTrace();
}
return htmlContent.trim();
}
public static void main(String[] args) {
try {
String src = getStaticPage("http://www.baidu.com");
File file = new File("D:\\data\\aa.html");
FileWriter resultFile = new FileWriter(file);
PrintWriter myFile = new PrintWriter(resultFile);// 写文件
myFile.println(src);
resultFile.close();
myFile.close();
} catch (IOException e) {
// TODO Auto-generated catch block
e.printStackTrace();
}
}
}
import java.io.FileWriter;
import java.io.IOException;
import java.io.PrintWriter;
public class Test {
private static String getStaticPage(String surl) {
String htmlContent = "";
try {
java.io.InputStream inputStream;
java.net.URL url = new java.net.URL(surl);
java.net.HttpURLConnection connection = (java.net.HttpURLConnection) url.openConnection();
connection.connect();
inputStream = connection.getInputStream();
byte[] bytes = new byte[1024 * 2000];
int index = 0;
int count = inputStream.read(bytes, index, 1024 * 2000);
while (count != -1) {
index += count;
count = inputStream.read(bytes, index, 1);
}
htmlContent = new String(bytes, "UTF-8");
connection.disconnect();
} catch (Exception ex) {
ex.printStackTrace();
}
return htmlContent.trim();
}
public static void main(String[] args) {
try {
String src = getStaticPage("http://www.baidu.com");
File file = new File("D:\\data\\aa.html");
FileWriter resultFile = new FileWriter(file);
PrintWriter myFile = new PrintWriter(resultFile);// 写文件
myFile.println(src);
resultFile.close();
myFile.close();
} catch (IOException e) {
// TODO Auto-generated catch block
e.printStackTrace();
}
}
}
0 0
- 网页抓取
- 网页抓取
- 抓取网页
- 网页抓取
- 抓取网页
- 网页抓取
- 网页抓取
- 抓取网页
- 网页抓取
- 抓取网页
- 网页抓取
- 网页抓取
- 网页抓取
- perl 网页抓取 网页解析
- 用PHP抓取网页
- 抓取网页中的链接
- 抓取网页中的链接
- PHP进行网页抓取
- iOS国际化
- 纯js 消灭星星游戏,js 消灭星星游戏实现原理,有道具的消灭星星
- android事件分发 入口(dispatchTouchEvent)
- 年关将至,老板,你拿什么留住人?
- Netty4学习笔记(8)-- Channel接口
- 网页抓取
- SQL高效去重语句
- LeetCode 219:Contains Duplicate II
- 重构的那些事儿
- Unity Shader 学习笔记(十) 滚动效果Shader实例
- apk反编译修改后重新打包
- 自定义控件(六)- 百分比圆形
- 又见嵌入式
- Android内存泄露自动检测神器LeakCanary