<?php
function get_url_content($Url,$Method = 'c') {
//引入需要的語言編碼.如果沒有, 就會預設為utf-8,不必擔心.
global $Charset;
$Urlarr = parse_url($Url);
//如果偵測不出網域名稱,就回傳.
if (!isset($Urlarr['host'])) {
return false;
}
//我們用智慧方式定義header頭倍資訊.
foreach (@getallheaders() as $key => $val){
$key==='Host' && $val = $Urlarr['host'];
$key==='Referer' && $val ='http://'.$Urlarr['host'];
$str .= "'$key:$val', n";
}
//虛擬來路.
!eregi('Referer',$str) && $str .="'Referer:http://{$Urlarr['host']}', n";
//經過修正, 基本上, 來路也是那個站, 主機也是Url站點.
$Header = array(trim($str));
//下面只是選擇用哪個程式來採集.
if($Method === 'f'&&function_exists('file_get_contents')) {
$opts = array(
'http'=>array(
'method'=>"GET",
'header'=>$Header,
)
);
$cxContext = stream_context_create($opts);
$file_contents = @file_get_contents($Url, false, $cxContext);
} elseif ($Method === 'c'&&function_exists('curl_init')) {
$Ch = curl_init();
$Timeout = 5;
curl_setopt($Ch,CURLOPT_HTTPHEADER,$Header);
curl_setopt ($Ch, CURLOPT_URL, $Url);
curl_setopt ($Ch, CURLOPT_RETURNTRANSFER,1);
curl_setopt ($Ch, CURLOPT_CONNECTTIMEOUT, $Timeout);
$file_contents = curl_exec($Ch);
curl_close($Ch);
}
//為了讓樣式顯示得漂亮,我們給它加一句目標引向.
$file_contents = str_replace('</title>',"</title>n<base href=" http://{$Urlarr['host']}/ " />",$file_contents);
//處理最常見的幾種編碼, 如果目標網站沒有編碼, 就預設為GBK
!preg_match('/charset=([^<>"]*)"/isU',$file_contents,$lang) && $lang[1]='GBK';
function_exists('mb_convert_encoding') && $file_contents = mb_convert_encoding($file_contents,empty($Charset)?'UTF-8':$Charset,$lang[1]);
//註銷部分程式碼;
unset($Url,$lang,$Timeout,$Urlarr,$Charset);
return $file_contents;
}
//測試開始測試用file_get_contents方式
HEADER("CONTENT-TYPE:TEXT/HTML; CHARSET=UTF-8");
//http://www.xtzj.com/read-htm-tid-347550.html 這是採集不到.
$file = get_url_content(" http://www.hao123.com",'f' );
$file = strip_tags($file,'<a>');
preg_match_all('/(http:[^"<>]*)>/isU',$file,$link);unset($link[0]);
$link = $link[1];
//我們來模擬取得數據. 自己換數位.0-151 以下是用curl方式
$x = 10;
$file = get_url_content($link[$x]);
echo $file;
?>
全部寫上說明, 註..
有不明白的回應..我來給採集普及一下知識.
原文網址: http ://bbs.phpchina.com/viewthread.php?tid=99263