الفريق العربي للبرمجةأرشيف المنتديات · 2000 – 2023
نسخة أرشيفية للقراءة فقط — التسجيل والمشاركة مغلقان، والمحتوى محفوظ كما كان.

ويب كرولر والروابطs

بدأه Khaled_Sulaiman في 15 يوليو 2010 · 8 رد · 1,058 مشاهدة · في لغة Python
مشاركة: واتساب X فيسبوك تيليجرام
#1 صاحب الموضوع

السلام عليكم..

ابي الفزعة يا جماعة الخير...

انا جالس اجرب ويب كرولر يستخرج فقط الروابط من صفحات الويب ويحفظها ب list

أبغى الويب كرولر يمشي لمستوى 3 مثلا.

المشكله:

يبدو انه الويب كرولر يمشي مزبوط للمستوى الاول والثاني بس يعك بالمستوى الثالث لسبب اجهله. هل فيه احد جرب ويب كرولرز بالبايثون؟ هل فيه احد ممكن يفهم ايش مشكلته هالبرنامج؟

الكود اللي جالس اجربه موجود اون لاين ويستخدم البيوتفل سوب انا بس عدلت فيه الاشياء اللي محتاجها...

import sys
import urllib2
import urlparse
from BeautifulSoup import BeautifulSoup

__version__ = "0.1"
__copyright__ = "CopyRight (C) 2008 by James Mills"
__license__ = "GPL"
__author__ = "James Mills"
__author_email__ = "James Mills, James dot Mills st dotred dot com dot au"

USAGE = "%prog [options] <url>"
VERSION = "%prog v" + __version__

AGENT = "%s/%s" % (__name__, __version__)

def encodeHTML(s=""):
    """encodeHTML(s) -> str

    Encode HTML special characters from their ASCII form to
    HTML entities.
    """

    return s.replace("&", "&") \
            .replace("<", "<") \
            .replace(">", ">") \
            .replace("\"", """) \
            .replace("'", "'") \
            .replace("--", "—")

class Fetcher(object):

    def __init__(self, url):
        self.url = url
        self.urls = []

    def __contains__(self, x):
        return x in self.urls

    def __getitem__(self, x):
        return self.urls[x]

    def _addHeaders(self, request):
        request.add_header("User-Agent", AGENT)

    def open(self):
        url = self.url
        #print "\nFollowing %s" % url
        try:
            request = urllib2.Request(url)
            handle = urllib2.build_opener()
        except IOError:
            return None
        return (request, handle)

    def fetch(self):
        request, handle = self.open()
        self._addHeaders(request)
        if handle:
            soup = BeautifulSoup()
            try:
                content = unicode(handle.open(request).read(), errors="ignore")
                soup.feed(content)
                #soup = BeautifulSoup(content)
                tags = soup('a')
            except urllib2.HTTPError, error:
                if error.code == 404:
                    print >> sys.stderr, "ERROR: %s -> %s" % (error, error.url)
                else:
                    print >> sys.stderr, "ERROR: %s" % error
                tags = []
            except urllib2.URLError, error:
                print >> sys.stderr, "ERROR: %s" % error
                tags = []
            for tag in tags:
                try:
                    href = tag["href"]
                    if href is not None:
                        url = urlparse.urljoin(self.url, encodeHTML(href))
                        if url not in self:
                            #print " Found: %s" % url
                            self.urls.append(url)
                except KeyError:
                    pass


# I created 3 lists (root, level2 and level3).                                 
# Each list saves the URLs of that level i.e. depth. I chose to create 3       

# Level1: 
root = Fetcher('http://www.wantasimplewebsite.co.uk/index.html')
root.fetch()
for url in root:
    if url not in root: # Avoid duplicate links 
       root.append(url)

print "\nRoot URLs are:"
for i, url in enumerate(root):
   print "%d. %s" % (i+1, url)


# Level2: 
level2 = []
for url in root: # Traverse every element(i.e URL) in root and fetch the URLs from it
    temp = Fetcher(url)
    temp.fetch()
    for url in temp:
        if url not in level2: # Avoid duplicate links 
           level2.append(url)


print "\nLevel2 URLs are:"
for i, url in enumerate(level2):
   print "%d. %s" % (i+1, url)


# Level3: 
level3 = []
for url in level2: # Traverse every element(i.e URL) in level2 and fetch the URLs from it
    temp = Fetcher(url)
    temp.fetch()
    for url in temp:
        if url not in level3: # Avoid duplicate links
           level3.append(url)


print "\nLevel3 URLs are:"
for i, url in enumerate(level3):
   print "%d. %s" % (i+1, url)
#2

تقدر تجرب السكربت دا

#       crawlme.py
#       
#       Copyright 2010 ahmed youssef <xmonader@gmail.com>
#       
#       This program is free software; you can redistribute it and/or modify
#       it under the terms of the GNU General Public License as published by
#       the Free Software Foundation; either version 2 of the License, or
#       (at your option) any later version.
#       
#       This program is distributed in the hope that it will be useful,
#       but WITHOUT ANY WARRANTY; without even the implied warranty of
#       MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
#       GNU General Public License for more details.
#       
#       You should have received a copy of the GNU General Public License
#       along with this program; if not, write to the Free Software
#       Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston,
#       MA 02110-1301, USA.

from BeautifulSoup import BeautifulSoup
import urllib2 as ulib

def get_source(url):
	src=""
	try:
		src=ulib.urlopen(url).read()
	except:
		pass		
	return src

def get_links(src):
	soup=BeautifulSoup(src)
	links=[]
	els=soup.findAll("a")
	if els:
		for el in els:
			try:
				if "http" in el['href']:
					links.append(el['href'])
			except:
				pass
	return links

def get_links_of(url):
	#print "visiting url: ", url
	res=get_links(get_source(url))
	#print "res: ", res
	return res

def crawl(link):
	total={}
	links=get_links_of(link)
	total[link]=links
	for link in links:
		links2=get_links_of(link)
		total[link]=links2
		for lnk in links2:
			links3=get_links_of(lnk)
			total[lnk]=links3
	return total

print crawl("http://ahmedyoussef.wordpress.com/")

دا كود مبدأى اكيد محتاج تحسين ومراجعه

عموما فى مواقع كتير مش بتكون بادئه بعنوان كامل بتستخدم relative path ففى الحالة دى هتحتاج تحلل الurl وتقدر تستخدم urlparse للمهمة دى

تم تعديل هذه المشاركة بواسطة ahmed_youssef في 15 يوليو 2010 في 20:38

(map share people)

فضلا لاتقم بمراسلتي من أجل أسئلة لها أقسامها في المنتدى حتى تعم الفائدة على الجميع وللحصول على إجابات أفضل من أعضاء أكثر خبرة.
Weblog
@bitbucket
@xmonader

#3

شكرا استاذ أحمد.. جزاك الله خير

ممكن اعرف ايش العرف بالنسبه للروابط .. قصدي هناك روابط كثيره فيها مشاكل مثل error403 and error404

ايش ممكن يسوي الواحد معاها؟ او بس الواحد يتجاهلها ويكمل لباقي الروابط؟ لاني لاحظت مجموعه كبيره من الروابط عباره عن اخطاء وتتسبب ببطء فظيع او حتى انهاء مفاجيء للبرنامج..

هل يوجد قائمة معينه بجميع الاخطاء والاهم كيفية التعامل معاها؟

مشكورين

#4

عفوا تحت امرك

دى القايمة

http://en.wikipedia.org/wiki/List_of_HTTP_status_codes

لكن بصورة عامة انت اهتمامك هيكون مقتصر على لينكات ال http وزى ماقلتلك اقرب اضافة للسكربت هى معالجة العناوين اللى بتستخدم ال relative paths

هنا فى المثال السابق بيتجاهلها فى حال ظهور اى ايرور فى عملية فتح اللينك وقراءة المحتوى

(map share people)

فضلا لاتقم بمراسلتي من أجل أسئلة لها أقسامها في المنتدى حتى تعم الفائدة على الجميع وللحصول على إجابات أفضل من أعضاء أكثر خبرة.
Weblog
@bitbucket
@xmonader

#5
اقتباس
لكن بصورة عامة انت اهتمامك هيكون مقتصر على لينكات ال http وزى ماقلتلك اقرب اضافة للسكربت هى معالجة العناوين اللى بتستخدم ال relative paths

هنا فى المثال السابق بيتجاهلها فى حال ظهور اى ايرور فى عملية فتح اللينك وقراءة المحتوى

عذرا أستاذ أحمد أنا انسان تنح جدا :blush: ما فهمت بس قرأت عن ال urlparse ويبدو انها تقسم الروابط وتعيد ربطها بس ما فهمت كيف هذا رح يساعدني. آسف على التناحه

#6

على سبيل المثال مثلا رابط زى دا http://zetcode.com/tutorials/pyqt4/

>>> p=up.urlparse("http://zetcode.com/tutorials/pyqt4/")
>>> p
ParseResult(scheme='http', netloc='zetcode.com', path='/tutorials/pyqt4/', params='', query='', fragment='')

هتلاقى الروابط للدروس اللى فيه بدون العنوان الكامل فهتعمل ايه ؟ هتشوف اذا العنوان صالح للفتح مباشرة او لأ اذا لأ تقدر تضيف العنوان السابق قبل المقطع بحيث ان العنوان firstprograms يتحول ل http://zetcode.com/tutorials/pyqt4/firstprograms

(map share people)

فضلا لاتقم بمراسلتي من أجل أسئلة لها أقسامها في المنتدى حتى تعم الفائدة على الجميع وللحصول على إجابات أفضل من أعضاء أكثر خبرة.
Weblog
@bitbucket
@xmonader

#7

الله يجزاك الجنه يا أستاذ أحمد ويسعدك ويفرج عنك همومك

عندي سؤال ثاني يتعلق بالبيوتفل سوب لو تكرمت. انا الحين كتبت كود عشان استخرج جميع الكلمات اللي موجوده في صفحه انترنت ثم اضع الكلمات في ملف.

فاللي عملته اني استخدمت regular expressions عشان احذف ال tags وكل شي بين الtags يعني اي شي بين <> ينحذف مع الاقواس وبذلك يبقى الكلمات اللي خارج الtags بس فيه مشكلة: الtags اللي فيها span or style or <!--.....> لاحظت انها ما تنحذف مثل باقي ال tags هل فيه طريقه ممكن اتخلص منها؟

مشكور

#8

السكربت دا GPL

http://www.aaronsw.com/2002/html2text/

اذا عايز تتحصل على كل النصوص مش اكتر ففى طريقة سهلة بعيد عن احتمالات ال RE

انك تستخدم http://docs.python.org/library/htmlparser.html

فقط قم بتعريف handle_data وضيف المحتوى ل string او list واتعامل معاه زى ماتحب

(map share people)

فضلا لاتقم بمراسلتي من أجل أسئلة لها أقسامها في المنتدى حتى تعم الفائدة على الجميع وللحصول على إجابات أفضل من أعضاء أكثر خبرة.
Weblog
@bitbucket
@xmonader

#9

جزاك الله كل خير أستاذ أحمد.. الله يسعدك ان شاء الله...

مشكور

مواضيع مشابهة