PHP实现抓取百度搜索结果页⾯【相关搜索词】并存储到txt⽂件⽰例
本⽂实例讲述了PHP实现抓取百度搜索结果页⾯【相关搜索词】并存储到txt⽂件。分享给⼤家供⼤家参考,具体如下:
⼀、百度搜索关键词【】
【】搜索链接
搜索结果部分源代码:
<div id="rs"><div class="tt">相关搜索</div><table cellpadding="0"><tbody><tr><th><a href="/s?
wd=%E6%B8%B8%E6%88%8F%E8%84%9A%E6%9C%AC%E4%B8%80%E8%88%AC%E9%83%BD%E5%9C%A8%E5%93%AA%E6%89%BE&rsf=4562&rsp=0&f=1&oq=%E8%84% 8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >游戏脚本⼀般都在哪</a></th><td></td><th><a href="/s?
wd=%E8%84%9A%E6%9C%AC%E6%80%8E%E4%B9%88%E5%86%99&rsf=4562&rsp=1&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE%B6&ie=utf-8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >脚本怎么写</a></th><td></td><th><a href="/s?
wd=%E8%84%9A%E6%9C%AC%E6%98%AF%E4%BB%80%E4%B9%88%E6%84%8F%E6%80%9D&rsf=4562&rsp=2&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE% 8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >脚本是什么意思</a></th></tr><tr><th><a href="/s?
wd=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE%B6app&rsf=4562&rsp=3&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE%B6&ie=utf-
8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >app</a></th><td></td><th><a href="/s?
wd=%E6%89%8B%E6%9C%BA%E8%84%9A%E6%9C%AC%E5%88%B6%E4%BD%9C&rsf=4562&rsp=4&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE%B6&ie=u 8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >⼿机脚本制作</a></th><td></td><th><a href="/s?
wd=%E6%89%8B%E6%9C%BA%E8%84%9A%E6%9C%AC%E5%A4%A7%E5%85%A8&rsf=4562&rsp=5&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE%B6&ie=u 8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >⼿机脚本⼤全</a></th></tr><tr><th><a href="/s?
wd=%E8%84%9A%E6%9C%AC%E6%B8%B8%E6%88%8F%E5%88%B6%E4%BD%9C%E5%A4%A7%E5%B8%88&rsf=4562&rsp=6&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9% 8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >脚本游戏制作⼤师</a></th><td></td><th><a href="/s?
wd=%E6%B8%B8%E6%88%8F%E8%84%9A%E6%9C%AC%E5%88%B6%E4%BD%9C%E6%95%99%E7%A8%8B&rsf=4562&rsp=7&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9% 8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >游戏脚本制作教程</a></th><td></td><th><a href="/s?
wd=%E8%84%9A%E6%9C%AC%E7%B2%BE%E7%81%B5&rsf=4562&rsp=8&f=1&oq=%E8%84%9A%E6%9C%AC%E4%B9%8B%E5%AE%B6&ie=utf-
8&rsv_idx=1&rsv_pq=c1ff4bdb000208b4&rsv_t=a1f2OCsgS6vkkBcxsdqfBfehkXoR65%2FtFlpSI30%2F%2FMmk6jQJEukZbv30XaM&rqlang=cn&rs_src=0&rsv_pq=c1ff4bdb0002 rel="external nofollow" >脚本精灵</a></th></tr></tbody></table></div>
⼆、抓取并保存本地
源代码
index.php:
<form action="index.php" method="post">
<input name="q" type="text" />
<input type="submit" value="Get Keywords" />
php文件下载源码
</form>
<?php
header('Content-Type:text/html;charset=gbk');
class ComBaike{
private $o_String=NULL;
public function __construct(){
include('cls.StringEx.php');
$this->o_String=new StringEx();
}
public function getItem($word){
$url = "www.baidu/s?wd=".$word;
/
/ 构造包头,模拟浏览器请求
$header = array (
"Host:www.baidu",
"Content-Type:application/x-www-form-urlencoded",//post请求
"Connection: keep-alive",
'Referer:www.baidu',
'User-Agent: Mozilla/5.0 (compatible; MSIE 9.0; Windows NT 6.1; WOW64; Trident/5.0; BIDUBrowser 2.6)'
);
$ch = curl_init ();
curl_setopt ( $ch, CURLOPT_URL, $url );
curl_setopt ( $ch, CURLOPT_HTTPHEADER, $header );
curl_setopt ( $ch, CURLOPT_RETURNTRANSFER, 1 );
$content = curl_exec ( $ch );
if ($content == FALSE) {
echo "error:" . curl_error ( $ch );
}
curl_close ( $ch );
//输出结果echo $content;
$this->o_String->string=$content;
$s_begin='<div id="rs">';
$s_end='</div>';
$summary=$this->o_String->getPart($s_begin,$s_end);
$s_begin='<div class="tt">相关搜索</div><table cellpadding="0"><tr><th>';
$s_end='</th></tr></table></div>';
$content=$this->o_String->getPart($s_begin,$s_end);
return $content;
}
public function __destruct(){
unset($this->o_String);
}
}
if($_POST){
$com = new ComBaike();
$q = $_POST['q'];
$str = $com->getItem($q); //获取搜索内容
$pat = '/<a(.*?)href="(.*?)" rel="external nofollow" (.*?)>(.*?)<\/a>/i';
preg_match_all($pat, $str, $m);
//print_r($m[4]); 链接⽂字
$con = implode(",", $m[4]);
//⽣成⽂件夹
$dates = date("Ymd");
$path="./Search/".$dates."/";
if(!is_dir($path)){
mkdir($path,0777,true);
}
//⽣成⽂件
$file = fopen($path.iconv("UTF-8","GBK",$q).".txt",'w');
if(fwrite($file,$con)){
echo $con;
echo '<script>alert("success")</script>';
}else{
echo '<script>alert("error")</script>';
}
fclose($file);
}
>
cls.StringEx.php:
<?php
header('Content-Type: text/html; charset=UTF-8');
class StringEx{
public $string='';
public function __construct($string=''){
$this->string=$string;
}
public function pregGetPart($s_begin,$s_end){
$s_begin==preg_quote($s_begin);
$s_begin=str_replace('/','\/',$s_begin);
$s_end=preg_quote($s_end);
$s_end=str_replace('/','\/',$s_end);
$pattern='/'.$s_begin.'(.*?)'.$s_end.'/';
$result=preg_match($pattern,$this->string,$a_match);
if(!$result){
return $result;
}else{
return isset($a_match[1])?$a_match[1]:'';
}
}
public function strstrGetPart($s_begin,$s_end){
$string=strstr($this->string,$s_begin);
$string=strstr($string,$s_end,true);
$string=str_replace($s_begin,'',$string);
$string=str_replace($s_end,'',$string);
return $string;
}
public function getPart($s_begin,$s_end){
$result=$this->pregGetPart($s_begin,$s_end);
if(!$result){
$result=$this->strstrGetPart($s_begin,$s_end);
}
return $result;
}
}
>
更多关于PHP相关内容感兴趣的读者可查看本站专题:《》、《》、《》、《》、《》及《》希望本⽂所述对⼤家PHP程序设计有所帮助。

版权声明:本站内容均来自互联网,仅供演示用,请勿用于商业和其他非法用途。如果侵犯了您的权益请与我们联系QQ:729038198,我们将在24小时内删除。