On Sat, 16 Jun 2001 22:07:18 +0100, I wrote:
>There is a problem with the way depth-limited recursive retrievals are
>performed. This is because the recursion depth tests assume the pages
>of the web-site are related to each other in a hierarchical tree,
>rather than a directed graph.
>
>To illustrate the problem, here is a diagram:
>
>test.html ---> a.html ---> b.html ---> c.html
> --------------->
>
>test.html has a link to a.html and a link to b.html in that order.
>a.html has a link to b.html and b.html has a link to c.html.
>
>Now do a level 2 recursive retrieval on test.html :
>
>test.html is retrieved and marked as visited.
>+ URLs to a.html and b.html are extracted from test.html.
>+ a.html is retrieved, marked as visited and recursed into.
>+ + URL to b.html is extracted from a.html.
>+ + b.html is retrieved, marked as visited and recursed into.
>+ + + recursion depth exceeded!
>+ + + (possibly load page requisites here)
>+ + + RETURN
>+ + Reached end of extracted URL list for a.html
>+ + RETURN
>+ b.html is skipped because it has already been seen.
>+ Reached end of extracted URL list for test.html.
>+ RETURN
>END
>
>Note that c.html has not been retrieved, even though it can be
>considered to be within the range of a level 2 recursive retrieval.
>
>If the order of the URLs in test.html is reversed then c.html would be
>retrieved. If the URLs were retrieved in a breadth first fashion
>rather than depth first, the problem can be avoided altogether. I'll
>have a play around in recur.c to do this - it shouldn't be too hard.
I have attached a patch to fix the above bug. The patched version
works by storing a list of all the retrieved files (and their URLs)
which are to be recursed into. This list is only examined when the
current recursion level is finished.
The patch is a bit ugly, but is all in one file (recur.c). It's mostly
a complete reorganization of the recursive_retrieve function (which is
no longer called recursively).
Regards,
Ian Abbott.
Index: src/recur.c
===================================================================
RCS file: /pack/anoncvs/wget/src/recur.c,v
retrieving revision 1.22
diff -u -r1.22 recur.c
--- src/recur.c 2001/06/14 21:48:00 1.22
+++ src/recur.c 2001/06/23 16:19:48
@@ -64,9 +64,6 @@
/* List of forbidden locations. */
static char **forbidden = NULL;
-/* Current recursion depth. */
-static int depth;
-
/* Base directory we're recursing from (used by no_parent). */
static char *base_dir;
@@ -134,6 +131,10 @@
char *constr, *filename, *newloc;
char *canon_this_url = NULL;
int dt, inl, dash_p_leaf_HTML = FALSE;
+ int depth = 0; /* Current recursion depth */
+ int depth_exceeded = FALSE;
+ int at_maximum_depth = FALSE;
+ int recurse_it;
int meta_disallow_follow;
int this_url_ftp; /* See below the explanation */
uerr_t err;
@@ -141,6 +142,8 @@
urlpos *url_list, *cur_url;
char *rfile; /* For robots */
struct urlinfo *u;
+ slist *cur_file_url_stack;
+ slist *next_file_url_stack;
assert (this_url != NULL);
assert (file != NULL);
@@ -175,355 +178,526 @@
base_dir = NULL;
}
freeurl (u, 1);
- depth = 1;
robots_host = NULL;
forbidden = NULL;
first_time = 0;
}
- else
- ++depth;
- if (opt.reclevel != INFINITE_RECURSION && depth > opt.reclevel)
- /* We've exceeded the maximum recursion depth specified by the user. */
+ /* Check if recursion level allows any recursive retrieval at all. */
+ if (opt.reclevel == 0 && !opt.page_requisites)
{
- if (opt.page_requisites && depth <= opt.reclevel + 1)
- /* When -p is specified, we can do one more partial recursion from the
- "leaf nodes" on the HTML document tree. The recursion is partial in
- that we won't traverse any <A> or <AREA> tags, nor any <LINK> tags
- except for <LINK REL="stylesheet">. */
- dash_p_leaf_HTML = TRUE;
- else
- /* Either -p wasn't specified or it was and we've already gone the one
- extra (pseudo-)level that it affords us, so we need to bail out. */
- {
- DEBUGP (("Recursion depth %d exceeded max. depth %d.\n",
- depth, opt.reclevel));
- --depth;
- return RECLEVELEXC;
- }
+ DEBUGP (("Recursive retrieval prevented by zero recursion level.\n"));
+ FREE_MAYBE(canon_this_url);
+ return RECLEVELEXC;
}
- /* Determine whether this_url is an FTP URL. If it is, it means
- that the retrieval is done through proxy. In that case, FTP
- links will be followed by default and recursion will not be
- turned off when following them. */
- this_url_ftp = (urlproto (this_url) == URLFTP);
-
- /* Get the URL-s from an HTML file: */
- url_list = get_urls_html (file, canon_this_url ? canon_this_url : this_url,
- dash_p_leaf_HTML, &meta_disallow_follow);
+ /* Initialize for first level. */
+ next_file_url_stack = NULL;
+ next_file_url_stack = slist_prepend (next_file_url_stack, this_url);
+ next_file_url_stack = slist_prepend (next_file_url_stack, file);
- if (opt.use_robots && meta_disallow_follow)
+ /* Do successive recursion levels. */
+ do
{
- /* The META tag says we are not to follow this file. Respect
- that. */
- free_urlpos (url_list);
- url_list = NULL;
- }
-
- /* Decide what to do with each of the URLs. A URL will be loaded if
- it meets several requirements, discussed later. */
- for (cur_url = url_list; cur_url; cur_url = cur_url->next)
- {
- /* If quota was exceeded earlier, bail out. */
- if (downloaded_exceeds_quota ())
- break;
- /* Parse the URL for convenient use in other functions, as well
- as to get the optimized form. It also checks URL integrity. */
- u = newurl ();
- if (parseurl (cur_url->url, u, 0) != URLOK)
- {
- DEBUGP (("Yuck! A bad URL.\n"));
- freeurl (u, 1);
- continue;
- }
- if (u->proto == URLFILE)
- {
- DEBUGP (("Nothing to do with file:// around here.\n"));
- freeurl (u, 1);
- continue;
- }
- assert (u->url != NULL);
- constr = xstrdup (u->url);
+ ++depth;
- /* Several checkings whether a file is acceptable to load:
- 1. check if URL is ftp, and we don't load it
- 2. check for relative links (if relative_only is set)
- 3. check for domain
- 4. check for no-parent
- 5. check for excludes && includes
- 6. check for suffix
- 7. check for same host (if spanhost is unset), with possible
- gethostbyname baggage
- 8. check for robots.txt
-
- Addendum: If the URL is FTP, and it is to be loaded, only the
- domain and suffix settings are "stronger".
-
- Note that .html and (yuck) .htm will get loaded regardless of
- suffix rules (but that is remedied later with unlink) unless
- the depth equals the maximum depth.
-
- More time- and memory- consuming tests should be put later on
- the list. */
-
- /* inl is set if the URL we are working on (constr) is stored in
- undesirable_urls. Using it is crucial to avoid unnecessary
- repeated continuous hits to the hash table. */
- inl = string_set_contains (undesirable_urls, constr);
-
- /* If it is FTP, and FTP is not followed, chuck it out. */
- if (!inl)
- if (u->proto == URLFTP && !opt.follow_ftp && !this_url_ftp)
- {
- DEBUGP (("Uh, it is FTP but i'm not in the mood to follow FTP.\n"));
- string_set_add (undesirable_urls, constr);
- inl = 1;
- }
- /* If it is absolute link and they are not followed, chuck it
- out. */
- if (!inl && u->proto != URLFTP)
- if (opt.relative_only && !cur_url->link_relative_p)
- {
- DEBUGP (("It doesn't really look like a relative link.\n"));
- string_set_add (undesirable_urls, constr);
- inl = 1;
- }
- /* If its domain is not to be accepted/looked-up, chuck it out. */
- if (!inl)
- if (!accept_domain (u))
- {
- DEBUGP (("I don't like the smell of that domain.\n"));
- string_set_add (undesirable_urls, constr);
- inl = 1;
- }
- /* Check for parent directory. */
- if (!inl && opt.no_parent
- /* If the new URL is FTP and the old was not, ignore
- opt.no_parent. */
- && !(!this_url_ftp && u->proto == URLFTP))
+ if (opt.reclevel != INFINITE_RECURSION)
{
- /* Check for base_dir first. */
- if (!(base_dir && frontcmp (base_dir, u->dir)))
+ if (depth > opt.reclevel)
{
- /* Failing that, check for parent dir. */
- struct urlinfo *ut = newurl ();
- if (parseurl (this_url, ut, 0) != URLOK)
- DEBUGP (("Double yuck! The *base* URL is broken.\n"));
- else if (!frontcmp (ut->dir, u->dir))
+ /* When -p is specified, we can do one more partial recursion
+ from the "leaf nodes" on the HTML document tree. The
+ recursion is partial in that we won't traverse any <A> or
+ <AREA> tags, nor any <LINK> tags except for
+ <LINK REL="stylesheet">. */
+ dash_p_leaf_HTML = TRUE;
+ /* We won't do any more recursion levels. */
+ at_maximum_depth = TRUE;
+ }
+ else if (depth == opt.reclevel && !opt.page_requisites)
+ /* Doing the last recursion level and -p was not specified,
+ so we won't do any more recursion levels. */
+ at_maximum_depth = TRUE;
+ }
+ DEBUGP (("Doing depth %d%s ....\n", depth,
+ dash_p_leaf_HTML ? " (page requisites only)" : ""));
+
+ /* Do the URLs in the reverse of the order we found them. (It would
+ be more aesthetic to do them in the order they were found.) */
+ cur_file_url_stack = next_file_url_stack;
+ next_file_url_stack = NULL;
+ do
+ {
+ slist *tmp_slist;
+
+ /* Get the file and URL from the stack. */
+ file = cur_file_url_stack->string;
+ this_url = cur_file_url_stack->next->string;
+
+ /* Determine whether this_url is an FTP URL. If it is, it means
+ that the retrieval is done through proxy. In that case, FTP
+ links will be followed by default and recursion will not be
+ turned off when following them. */
+ this_url_ftp = (urlproto (this_url) == URLFTP);
+
+ /* Get the URL-s from an HTML file: */
+ url_list = get_urls_html (file,
+ canon_this_url ? canon_this_url : this_url,
+ dash_p_leaf_HTML, &meta_disallow_follow);
+
+ if (opt.use_robots && meta_disallow_follow)
+ {
+ /* The META tag says we are not to follow this file. Respect
+ that. */
+ free_urlpos (url_list);
+ url_list = NULL;
+ }
+
+ /* Decide what to do with each of the URLs. A URL will be loaded if
+ it meets several requirements, discussed later. */
+ for (cur_url = url_list; cur_url; cur_url = cur_url->next)
+ {
+ /* If quota was exceeded earlier, bail out. */
+ if (downloaded_exceeds_quota ())
+ break;
+ /* Parse the URL for convenient use in other functions, as well
+ as to get the optimized form. It also checks URL integrity. */
+ u = newurl ();
+ if (parseurl (cur_url->url, u, 0) != URLOK)
{
- /* Failing that too, kill the URL. */
- DEBUGP (("Trying to escape parental guidance with no_parent on.\n"));
- string_set_add (undesirable_urls, constr);
- inl = 1;
+ DEBUGP (("Yuck! A bad URL.\n"));
+ freeurl (u, 1);
+ continue;
}
- freeurl (ut, 1);
- }
- }
- /* If the file does not match the acceptance list, or is on the
- rejection list, chuck it out. The same goes for the
- directory exclude- and include- lists. */
- if (!inl && (opt.includes || opt.excludes))
- {
- if (!accdir (u->dir, ALLABS))
- {
- DEBUGP (("%s (%s) is excluded/not-included.\n", constr, u->dir));
- string_set_add (undesirable_urls, constr);
- inl = 1;
- }
- }
- if (!inl)
- {
- char *suf = NULL;
- /* We check for acceptance/rejection rules only for non-HTML
- documents. Since we don't know whether they really are
- HTML, it will be deduced from (an OR-ed list):
-
- 1) u->file is "" (meaning it is a directory)
- 2) suffix exists, AND:
- a) it is "html", OR
- b) it is "htm"
-
- If the file *is* supposed to be HTML, it will *not* be
- subject to acc/rej rules, unless a finite maximum depth has
- been specified and the current depth is the maximum depth. */
- if (!
- (!*u->file
- || (((suf = suffix (constr)) != NULL)
- && ((!strcmp (suf, "html") || !strcmp (suf, "htm"))
- && ((opt.reclevel != INFINITE_RECURSION) &&
- (depth != opt.reclevel))))))
- {
- if (!acceptable (u->file))
+ if (u->proto == URLFILE)
+ {
+ DEBUGP (("Nothing to do with file:// around here.\n"));
+ freeurl (u, 1);
+ continue;
+ }
+ assert (u->url != NULL);
+ constr = xstrdup (u->url);
+
+ /* Several checkings whether a file is acceptable to load:
+ 1. check if URL is ftp, and we don't load it
+ 2. check for relative links (if relative_only is set)
+ 3. check for domain
+ 4. check for no-parent
+ 5. check for excludes && includes
+ 6. check for suffix
+ 7. check for same host (if spanhost is unset), with possible
+ gethostbyname baggage
+ 8. check for robots.txt
+
+ Addendum: If the URL is FTP, and it is to be loaded, only the
+ domain and suffix settings are "stronger".
+
+ Note that .html and (yuck) .htm will get loaded regardless of
+ suffix rules (but that is remedied later with unlink) unless
+ the depth equals the maximum depth.
+
+ More time- and memory- consuming tests should be put later on
+ the list. */
+
+ /* inl is set if the URL we are working on (constr) is stored in
+ undesirable_urls. Using it is crucial to avoid unnecessary
+ repeated continuous hits to the hash table. */
+ inl = string_set_contains (undesirable_urls, constr);
+
+ /* If it is FTP, and FTP is not followed, chuck it out. */
+ if (!inl)
+ if (u->proto == URLFTP && !opt.follow_ftp && !this_url_ftp)
+ {
+ DEBUGP (("Uh, it is FTP but i'm not in the mood to follow
+FTP.\n"));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ /* If it is absolute link and they are not followed, chuck it
+ out. */
+ if (!inl && u->proto != URLFTP)
+ if (opt.relative_only && !cur_url->link_relative_p)
+ {
+ DEBUGP (("It doesn't really look like a relative link.\n"));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ /* If its domain is not to be accepted/looked-up, chuck it out. */
+ if (!inl)
+ if (!accept_domain (u))
+ {
+ DEBUGP (("I don't like the smell of that domain.\n"));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ /* Check for parent directory. */
+ if (!inl && opt.no_parent
+ /* If the new URL is FTP and the old was not, ignore
+ opt.no_parent. */
+ && !(!this_url_ftp && u->proto == URLFTP))
{
- DEBUGP (("%s (%s) does not match acc/rej rules.\n",
- constr, u->file));
+ /* Check for base_dir first. */
+ if (!(base_dir && frontcmp (base_dir, u->dir)))
+ {
+ /* Failing that, check for parent dir. */
+ struct urlinfo *ut = newurl ();
+ if (parseurl (this_url, ut, 0) != URLOK)
+ DEBUGP (("Double yuck! The *base* URL is broken.\n"));
+ else if (!frontcmp (ut->dir, u->dir))
+ {
+ /* Failing that too, kill the URL. */
+ DEBUGP (("Trying to escape parental guidance with no_parent
+on.\n"));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ freeurl (ut, 1);
+ }
+ }
+ /* If the file does not match the acceptance list, or is on the
+ rejection list, chuck it out. The same goes for the
+ directory exclude- and include- lists. */
+ if (!inl && (opt.includes || opt.excludes))
+ {
+ if (!accdir (u->dir, ALLABS))
+ {
+ DEBUGP (("%s (%s) is excluded/not-included.\n",
+ constr, u->dir));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ }
+ if (!inl)
+ {
+ char *suf = NULL;
+ /* We check for acceptance/rejection rules only for non-HTML
+ documents. Since we don't know whether they really are
+ HTML, it will be deduced from (an OR-ed list):
+
+ 1) u->file is "" (meaning it is a directory)
+ 2) suffix exists, AND:
+ a) it is "html", OR
+ b) it is "htm"
+
+ If the file *is* supposed to be HTML, it will *not* be
+ subject to acc/rej rules, unless a finite maximum depth has
+ been specified and the current depth is the maximum depth.
+ */
+ if (!
+ (!*u->file
+ || (((suf = suffix (constr)) != NULL)
+ && ((!strcmp (suf, "html") || !strcmp (suf, "htm"))
+ && ((opt.reclevel != INFINITE_RECURSION) &&
+ (depth != opt.reclevel))))))
+ {
+ if (!acceptable (u->file))
+ {
+ DEBUGP (("%s (%s) does not match acc/rej rules.\n",
+ constr, u->file));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ }
+ FREE_MAYBE (suf);
+ }
+ /* Optimize the URL (which includes possible DNS lookup) only
+ after all other possibilities have been exhausted. */
+ if (!inl)
+ {
+ if (!opt.simple_check)
+ opt_url (u);
+ else
+ {
+ char *p;
+ /* Just lowercase the hostname. */
+ for (p = u->host; *p; p++)
+ *p = TOLOWER (*p);
+ xfree (u->url);
+ u->url = str_url (u, 0);
+ }
+ xfree (constr);
+ constr = xstrdup (u->url);
+ /* After we have canonicalized the URL, check if we have it
+ on the black list. */
+ if (string_set_contains (undesirable_urls, constr))
+ inl = 1;
+ /* This line is bogus. */
+ /*string_set_add (undesirable_urls, constr);*/
+
+ if (!inl && !((u->proto == URLFTP) && !this_url_ftp))
+ if (!opt.spanhost && this_url
+ && !same_host (this_url, constr))
+ {
+ DEBUGP (("This is not the same hostname as the parent's.\n"));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ }
+ /* What about robots.txt? */
+ if (!inl && opt.use_robots && u->proto == URLHTTP)
+ {
+ /* Since Wget knows about only one set of robot rules at a
+ time, /robots.txt must be reloaded whenever a new host is
+ accessed.
+
+ robots_host holds the host the current `forbid' variable
+ is assigned to. */
+ if (!robots_host || !same_host (robots_host, u->host))
+ {
+ FREE_MAYBE (robots_host);
+ /* Now make robots_host the new host, no matter what the
+ result will be. So if there is no /robots.txt on the
+ site, Wget will not retry getting robots all the
+ time. */
+ robots_host = xstrdup (u->host);
+ free_vec (forbidden);
+ forbidden = NULL;
+ err = retrieve_robots (constr, ROBOTS_FILENAME);
+ if (err == ROBOTSOK)
+ {
+ rurl = robots_url (constr, ROBOTS_FILENAME);
+ rfile = url_filename (rurl);
+ forbidden = parse_robots (rfile);
+ freeurl (rurl, 1);
+ xfree (rfile);
+ }
+ }
+
+ /* Now that we have (or don't have) robots, we can check for
+ them. */
+ if (!robots_match (u, forbidden))
+ {
+ DEBUGP (("Stuffing %s because %s forbids it.\n", this_url,
+ ROBOTS_FILENAME));
+ string_set_add (undesirable_urls, constr);
+ inl = 1;
+ }
+ }
+
+ filename = NULL;
+ recurse_it = FALSE;
+ /* If it wasn't chucked out, do something with it. */
+ if (!inl)
+ {
+ DEBUGP (("I've decided to load it -> "));
+ /* Add it to the list of already-loaded URL-s. */
string_set_add (undesirable_urls, constr);
- inl = 1;
+ /* Automatically followed FTPs will *not* be downloaded
+ recursively. */
+ if (u->proto == URLFTP)
+ {
+ /* Don't you adore side-effects? */
+ opt.recursive = 0;
+ }
+ /* Reset its type. */
+ dt = 0;
+ /* Retrieve it. */
+ retrieve_url (constr, &filename, &newloc,
+ canon_this_url ? canon_this_url : this_url, &dt);
+ if (u->proto == URLFTP)
+ {
+ /* Restore... */
+ opt.recursive = 1;
+ }
+ if (newloc)
+ {
+ xfree (constr);
+ constr = newloc;
+ }
+ /* If there was no error, and the type is text/html, parse
+ it recursively. */
+ if (dt & TEXTHTML)
+ {
+ if (dt & RETROKF)
+ {
+ /* Do not exceed maximum depth except to get
+ page requisites if -p option used. */
+ if (at_maximum_depth)
+ {
+ DEBUGP (("%s at maximum depth so don't recurse.\n",
+ filename));
+ depth_exceeded = TRUE;
+ }
+ else
+ {
+ DEBUGP (("Stacking %s for recursion depth %d.\n",
+ filename, depth + 1));
+ recurse_it = TRUE;
+ next_file_url_stack
+ = slist_prepend (next_file_url_stack, constr);
+ next_file_url_stack
+ = slist_prepend (next_file_url_stack, filename);
+ }
+ }
+ else
+ DEBUGP (("%s not retrieved okay so don't recurse.\n",
+ filename));
+ }
+ else
+ DEBUGP (("%s is not text/html so we don't chase.\n",
+ filename ? filename: "(null)"));
+
+ if (opt.delete_after || (filename && !acceptable (filename)))
+
+ /* Either --delete-after was specified, or we loaded this
+ otherwise rejected (e.g. by -R) HTML file just so we
+ could harvest its hyperlinks -- in either case, delete
+ the local file. */
+ {
+ /* Defer deletion if marked for recursion. */
+ if (!recurse_it)
+ {
+ DEBUGP (("Removing file due to %s in
+recursive_retrieve():\n",
+ opt.delete_after ? "--delete-after" :
+ "recursive rejection criteria"));
+ logprintf (LOG_VERBOSE,
+ (opt.delete_after ? _("Removing %s.\n")
+ : _("Removing %s since it should be
+rejected.\n")),
+ filename);
+ if (unlink (filename))
+ logprintf (LOG_NOTQUIET, "unlink: %s\n",
+ strerror (errno));
+ }
+ else
+ DEBUGP (("Deferring removal of file due to %s in
+recursive_retrieve().\n",
+ opt.delete_after ? "--delete-after" :
+ "recursive rejection criteria"));
+
+ dt &= ~RETROKF;
+ }
+
+ /* If everything was OK, and links are to be converted, let's
+ store the local filename. */
+ if (opt.convert_links && (dt & RETROKF) && (filename != NULL))
+ {
+ cur_url->convert = CO_CONVERT_TO_RELATIVE;
+ cur_url->local_name = xstrdup (filename);
+ }
}
- }
- FREE_MAYBE (suf);
- }
- /* Optimize the URL (which includes possible DNS lookup) only
- after all other possibilities have been exhausted. */
- if (!inl)
- {
- if (!opt.simple_check)
- opt_url (u);
- else
- {
- char *p;
- /* Just lowercase the hostname. */
- for (p = u->host; *p; p++)
- *p = TOLOWER (*p);
- xfree (u->url);
- u->url = str_url (u, 0);
- }
- xfree (constr);
- constr = xstrdup (u->url);
- /* After we have canonicalized the URL, check if we have it
- on the black list. */
- if (string_set_contains (undesirable_urls, constr))
- inl = 1;
- /* This line is bogus. */
- /*string_set_add (undesirable_urls, constr);*/
-
- if (!inl && !((u->proto == URLFTP) && !this_url_ftp))
- if (!opt.spanhost && this_url && !same_host (this_url, constr))
- {
- DEBUGP (("This is not the same hostname as the parent's.\n"));
- string_set_add (undesirable_urls, constr);
- inl = 1;
- }
- }
- /* What about robots.txt? */
- if (!inl && opt.use_robots && u->proto == URLHTTP)
- {
- /* Since Wget knows about only one set of robot rules at a
- time, /robots.txt must be reloaded whenever a new host is
- accessed.
-
- robots_host holds the host the current `forbid' variable
- is assigned to. */
- if (!robots_host || !same_host (robots_host, u->host))
- {
- FREE_MAYBE (robots_host);
- /* Now make robots_host the new host, no matter what the
- result will be. So if there is no /robots.txt on the
- site, Wget will not retry getting robots all the
- time. */
- robots_host = xstrdup (u->host);
- free_vec (forbidden);
- forbidden = NULL;
- err = retrieve_robots (constr, ROBOTS_FILENAME);
- if (err == ROBOTSOK)
+ else
+ DEBUGP (("%s already in list, so we don't load.\n", constr));
+ /* Free filename and constr. */
+ FREE_MAYBE (filename);
+ FREE_MAYBE (constr);
+ freeurl (u, 1);
+ /* Increment the pbuf for the appropriate size. */
+ }
+ if (opt.convert_links && !opt.delete_after)
+ /* This is merely the first pass: the links that have been
+ successfully downloaded are converted. In the second pass,
+ convert_all_links() will also convert those links that have NOT
+ been downloaded to their canonical form. */
+ convert_links (file, url_list);
+ /* Free the linked list of URL-s. */
+ free_urlpos (url_list);
+ /* Free the canonical this_url. */
+ if (canon_this_url)
+ {
+ xfree(canon_this_url);
+ canon_this_url = NULL;
+ }
+
+ if (depth > 1)
+ {
+ /* Deal with deferred deletion. */
+ if (opt.delete_after || (file && !acceptable (file)))
+ /* Either --delete-after was specified, or we loaded this
+ otherwise rejected (e.g. by -R) HTML file just so we
+ could harvest its hyperlinks -- in either case, delete
+ the local file. */
{
- rurl = robots_url (constr, ROBOTS_FILENAME);
- rfile = url_filename (rurl);
- forbidden = parse_robots (rfile);
- freeurl (rurl, 1);
- xfree (rfile);
+ DEBUGP (("Previously deferred removal of file in
+recursive_retrieve():\n"));
+ logprintf (LOG_VERBOSE,
+ (opt.delete_after ? _("Removing %s.\n")
+ : _("Removing %s since it should be rejected.\n")),
+ file);
+ if (unlink (file))
+ logprintf (LOG_NOTQUIET, "unlink: %s\n",
+ strerror (errno));
}
}
- /* Now that we have (or don't have) robots, we can check for
- them. */
- if (!robots_match (u, forbidden))
- {
- DEBUGP (("Stuffing %s because %s forbids it.\n", this_url,
- ROBOTS_FILENAME));
- string_set_add (undesirable_urls, constr);
- inl = 1;
+ /* Split off and free the URL/file pair we've just finished. */
+ tmp_slist = cur_file_url_stack;
+ cur_file_url_stack = tmp_slist->next->next;
+ tmp_slist->next->next = NULL;
+ slist_free (tmp_slist);
+
+ /* If quota was exceeded earlier, bail out. */
+ if (downloaded_exceeds_quota ())
+ break;
+ }
+ while (cur_file_url_stack);
+
+ /* Tidy up uncompleted stuff for this recursion level. */
+ if (depth > 1)
+ {
+ slist *pair = cur_file_url_stack;
+
+ while (pair)
+ {
+ file = pair->string;
+ if (opt.delete_after || (file && !acceptable (file)))
+ /* Either --delete-after was specified, or we loaded this
+ otherwise rejected (e.g. by -R) HTML file just so we
+ could harvest its hyperlinks -- in either case, delete
+ the local file. */
+ {
+ DEBUGP (("Previously deferred removal of file in
+recursive_retrieve():\n"));
+ logprintf (LOG_VERBOSE,
+ (opt.delete_after ? _("Removing %s.\n")
+ : _("Removing %s since it should be rejected.\n")),
+ file);
+ if (unlink (file))
+ logprintf (LOG_NOTQUIET, "unlink: %s\n",
+ strerror (errno));
+ }
+ pair = pair->next->next;
}
}
+ slist_free (cur_file_url_stack);
+ /* Free the canonical this_url. */
+ if (canon_this_url)
+ {
+ xfree(canon_this_url);
+ canon_this_url = NULL;
+ }
- filename = NULL;
- /* If it wasn't chucked out, do something with it. */
- if (!inl)
+ /* If quota was exceeded earlier, bail out. */
+ if (downloaded_exceeds_quota ())
+ break;
+ }
+ while (next_file_url_stack);
+
+ /* Tidy up uncompleted stuff. */
+ if (depth > 1)
+ {
+ slist *pair = next_file_url_stack;
+
+ while (pair)
{
- DEBUGP (("I've decided to load it -> "));
- /* Add it to the list of already-loaded URL-s. */
- string_set_add (undesirable_urls, constr);
- /* Automatically followed FTPs will *not* be downloaded
- recursively. */
- if (u->proto == URLFTP)
+ file = pair->string;
+ if (opt.delete_after || (file && !acceptable (file)))
+ /* Either --delete-after was specified, or we loaded this
+ otherwise rejected (e.g. by -R) HTML file just so we
+ could harvest its hyperlinks -- in either case, delete
+ the local file. */
{
- /* Don't you adore side-effects? */
- opt.recursive = 0;
- }
- /* Reset its type. */
- dt = 0;
- /* Retrieve it. */
- retrieve_url (constr, &filename, &newloc,
- canon_this_url ? canon_this_url : this_url, &dt);
- if (u->proto == URLFTP)
- {
- /* Restore... */
- opt.recursive = 1;
- }
- if (newloc)
- {
- xfree (constr);
- constr = newloc;
- }
- /* If there was no error, and the type is text/html, parse
- it recursively. */
- if (dt & TEXTHTML)
- {
- if (dt & RETROKF)
- recursive_retrieve (filename, constr);
- }
- else
- DEBUGP (("%s is not text/html so we don't chase.\n",
- filename ? filename: "(null)"));
-
- if (opt.delete_after || (filename && !acceptable (filename)))
- /* Either --delete-after was specified, or we loaded this otherwise
- rejected (e.g. by -R) HTML file just so we could harvest its
- hyperlinks -- in either case, delete the local file. */
- {
- DEBUGP (("Removing file due to %s in recursive_retrieve():\n",
- opt.delete_after ? "--delete-after" :
- "recursive rejection criteria"));
+ DEBUGP (("Previously deferred removal of file in
+recursive_retrieve():\n"));
logprintf (LOG_VERBOSE,
(opt.delete_after ? _("Removing %s.\n")
: _("Removing %s since it should be rejected.\n")),
- filename);
- if (unlink (filename))
- logprintf (LOG_NOTQUIET, "unlink: %s\n", strerror (errno));
- dt &= ~RETROKF;
- }
-
- /* If everything was OK, and links are to be converted, let's
- store the local filename. */
- if (opt.convert_links && (dt & RETROKF) && (filename != NULL))
- {
- cur_url->convert = CO_CONVERT_TO_RELATIVE;
- cur_url->local_name = xstrdup (filename);
+ file);
+ if (unlink (file))
+ logprintf (LOG_NOTQUIET, "unlink: %s\n",
+ strerror (errno));
}
+ pair = pair->next->next;
}
- else
- DEBUGP (("%s already in list, so we don't load.\n", constr));
- /* Free filename and constr. */
- FREE_MAYBE (filename);
- FREE_MAYBE (constr);
- freeurl (u, 1);
- /* Increment the pbuf for the appropriate size. */
}
- if (opt.convert_links && !opt.delete_after)
- /* This is merely the first pass: the links that have been
- successfully downloaded are converted. In the second pass,
- convert_all_links() will also convert those links that have NOT
- been downloaded to their canonical form. */
- convert_links (file, url_list);
- /* Free the linked list of URL-s. */
- free_urlpos (url_list);
- /* Free the canonical this_url. */
- FREE_MAYBE (canon_this_url);
- /* Decrement the recursion depth. */
- --depth;
+ slist_free (next_file_url_stack);
if (downloaded_exceeds_quota ())
return QUOTEXC;
+ else if (depth_exceeded)
+ return RECLEVELEXC;
else
return RETROK;
}