2000-11-20 04:50:10 +08:00
|
|
|
|
/* Collect URLs from HTML source.
|
2002-04-12 09:23:23 +08:00
|
|
|
|
Copyright (C) 1998, 2000, 2001, 2002 Free Software Foundation, Inc.
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-05-28 03:35:15 +08:00
|
|
|
|
This file is part of GNU Wget.
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-05-28 03:35:15 +08:00
|
|
|
|
GNU Wget is free software; you can redistribute it and/or modify
|
2000-11-20 04:50:10 +08:00
|
|
|
|
it under the terms of the GNU General Public License as published by
|
|
|
|
|
the Free Software Foundation; either version 2 of the License, or
|
|
|
|
|
(at your option) any later version.
|
|
|
|
|
|
2001-05-28 03:35:15 +08:00
|
|
|
|
GNU Wget is distributed in the hope that it will be useful,
|
2000-11-20 04:50:10 +08:00
|
|
|
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
|
|
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
|
|
|
GNU General Public License for more details.
|
|
|
|
|
|
|
|
|
|
You should have received a copy of the GNU General Public License
|
2001-05-28 03:35:15 +08:00
|
|
|
|
along with Wget; if not, write to the Free Software
|
2000-11-20 04:50:10 +08:00
|
|
|
|
Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA. */
|
|
|
|
|
|
|
|
|
|
#include <config.h>
|
|
|
|
|
|
|
|
|
|
#include <stdio.h>
|
|
|
|
|
#ifdef HAVE_STRING_H
|
|
|
|
|
# include <string.h>
|
|
|
|
|
#else
|
|
|
|
|
# include <strings.h>
|
|
|
|
|
#endif
|
|
|
|
|
#include <stdlib.h>
|
|
|
|
|
#include <errno.h>
|
|
|
|
|
#include <assert.h>
|
|
|
|
|
|
|
|
|
|
#include "wget.h"
|
|
|
|
|
#include "html-parse.h"
|
|
|
|
|
#include "url.h"
|
|
|
|
|
#include "utils.h"
|
|
|
|
|
|
|
|
|
|
#ifndef errno
|
|
|
|
|
extern int errno;
|
|
|
|
|
#endif
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
struct map_context;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
typedef void (*tag_handler_t) PARAMS ((int, struct taginfo *,
|
|
|
|
|
struct map_context *));
|
|
|
|
|
|
|
|
|
|
#define DECLARE_TAG_HANDLER(fun) \
|
|
|
|
|
static void fun PARAMS ((int, struct taginfo *, struct map_context *))
|
|
|
|
|
|
|
|
|
|
DECLARE_TAG_HANDLER (tag_find_urls);
|
|
|
|
|
DECLARE_TAG_HANDLER (tag_handle_base);
|
2002-04-12 01:51:45 +08:00
|
|
|
|
DECLARE_TAG_HANDLER (tag_handle_form);
|
2001-12-12 23:43:01 +08:00
|
|
|
|
DECLARE_TAG_HANDLER (tag_handle_link);
|
|
|
|
|
DECLARE_TAG_HANDLER (tag_handle_meta);
|
|
|
|
|
|
|
|
|
|
/* The list of known tags and functions used for handling them. Most
|
|
|
|
|
tags are simply harvested for URLs. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
static struct {
|
|
|
|
|
const char *name;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
tag_handler_t handler;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
} known_tags[] = {
|
|
|
|
|
#define TAG_A 0
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "a", tag_find_urls },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_APPLET 1
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "applet", tag_find_urls },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_AREA 2
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "area", tag_find_urls },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_BASE 3
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "base", tag_handle_base },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_BGSOUND 4
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "bgsound", tag_find_urls },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_BODY 5
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "body", tag_find_urls },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_EMBED 6
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "embed", tag_find_urls },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
#define TAG_FIG 7
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "fig", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_FORM 8
|
|
|
|
|
{ "form", tag_handle_form },
|
|
|
|
|
#define TAG_FRAME 9
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "frame", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_IFRAME 10
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "iframe", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_IMG 11
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "img", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_INPUT 12
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "input", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_LAYER 13
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "layer", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_LINK 14
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "link", tag_handle_link },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_META 15
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "meta", tag_handle_meta },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_OVERLAY 16
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "overlay", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_SCRIPT 17
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "script", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_TABLE 18
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "table", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_TD 19
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "td", tag_find_urls },
|
2002-04-12 01:51:45 +08:00
|
|
|
|
#define TAG_TH 20
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ "th", tag_find_urls }
|
2000-11-20 04:50:10 +08:00
|
|
|
|
};
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* tag_url_attributes documents which attributes of which tags contain
|
|
|
|
|
URLs to harvest. It is used by tag_find_urls. */
|
2001-01-10 10:28:24 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* Defines for the FLAGS field; currently only one flag is defined. */
|
2001-01-10 10:28:24 +08:00
|
|
|
|
|
|
|
|
|
/* This tag points to an external document not necessary for rendering this
|
|
|
|
|
document (i.e. it's not an inlined image, stylesheet, etc.). */
|
2001-12-12 23:43:01 +08:00
|
|
|
|
#define TUA_EXTERNAL 1
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* For tags handled by tag_find_urls: attributes that contain URLs to
|
2000-11-20 04:50:10 +08:00
|
|
|
|
download. */
|
|
|
|
|
static struct {
|
|
|
|
|
int tagid;
|
|
|
|
|
const char *attr_name;
|
|
|
|
|
int flags;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
} tag_url_attributes[] = {
|
|
|
|
|
{ TAG_A, "href", TUA_EXTERNAL },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{ TAG_APPLET, "code", 0 },
|
2001-12-12 23:43:01 +08:00
|
|
|
|
{ TAG_AREA, "href", TUA_EXTERNAL },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{ TAG_BGSOUND, "src", 0 },
|
|
|
|
|
{ TAG_BODY, "background", 0 },
|
2001-12-13 15:18:59 +08:00
|
|
|
|
{ TAG_EMBED, "href", TUA_EXTERNAL },
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{ TAG_EMBED, "src", 0 },
|
|
|
|
|
{ TAG_FIG, "src", 0 },
|
|
|
|
|
{ TAG_FRAME, "src", 0 },
|
|
|
|
|
{ TAG_IFRAME, "src", 0 },
|
|
|
|
|
{ TAG_IMG, "href", 0 },
|
|
|
|
|
{ TAG_IMG, "lowsrc", 0 },
|
|
|
|
|
{ TAG_IMG, "src", 0 },
|
|
|
|
|
{ TAG_INPUT, "src", 0 },
|
|
|
|
|
{ TAG_LAYER, "src", 0 },
|
|
|
|
|
{ TAG_OVERLAY, "src", 0 },
|
|
|
|
|
{ TAG_SCRIPT, "src", 0 },
|
|
|
|
|
{ TAG_TABLE, "background", 0 },
|
|
|
|
|
{ TAG_TD, "background", 0 },
|
|
|
|
|
{ TAG_TH, "background", 0 }
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
/* The lists of interesting tags and attributes are built dynamically,
|
|
|
|
|
from the information above. However, some places in the code refer
|
|
|
|
|
to the attributes not mentioned here. We add them manually. */
|
|
|
|
|
static const char *additional_attributes[] = {
|
2002-04-12 01:51:45 +08:00
|
|
|
|
"rel", /* used by tag_handle_link */
|
|
|
|
|
"http-equiv", /* used by tag_handle_meta */
|
|
|
|
|
"name", /* used by tag_handle_meta */
|
|
|
|
|
"content", /* used by tag_handle_meta */
|
|
|
|
|
"action" /* used by tag_handle_form */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
static const char **interesting_tags;
|
|
|
|
|
static const char **interesting_attributes;
|
|
|
|
|
|
2001-12-09 09:24:41 +08:00
|
|
|
|
static void
|
2000-11-20 04:50:10 +08:00
|
|
|
|
init_interesting (void)
|
|
|
|
|
{
|
|
|
|
|
/* Init the variables interesting_tags and interesting_attributes
|
|
|
|
|
that are used by the HTML parser to know which tags and
|
|
|
|
|
attributes we're interested in. We initialize this only once,
|
|
|
|
|
for performance reasons.
|
|
|
|
|
|
|
|
|
|
Here we also make sure that what we put in interesting_tags
|
|
|
|
|
matches the user's preferences as specified through --ignore-tags
|
2001-12-12 23:43:01 +08:00
|
|
|
|
and --follow-tags.
|
|
|
|
|
|
|
|
|
|
This function is as large as this only because of the glorious
|
|
|
|
|
expressivity of the C programming language. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
|
|
|
|
{
|
|
|
|
|
int i, ind = 0;
|
|
|
|
|
int size = ARRAY_SIZE (known_tags);
|
|
|
|
|
interesting_tags = (const char **)xmalloc ((size + 1) * sizeof (char *));
|
|
|
|
|
|
|
|
|
|
for (i = 0; i < size; i++)
|
|
|
|
|
{
|
|
|
|
|
const char *name = known_tags[i].name;
|
|
|
|
|
|
|
|
|
|
/* Normally here we could say:
|
|
|
|
|
interesting_tags[i] = name;
|
|
|
|
|
But we need to respect the settings of --ignore-tags and
|
2001-01-10 10:54:52 +08:00
|
|
|
|
--follow-tags, so the code gets a bit hairier. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
|
|
|
|
if (opt.ignore_tags)
|
|
|
|
|
{
|
|
|
|
|
/* --ignore-tags was specified. Do not match these
|
|
|
|
|
specific tags. --ignore-tags takes precedence over
|
|
|
|
|
--follow-tags, so we process --ignore first and fall
|
|
|
|
|
through if there's no match. */
|
|
|
|
|
int j, lose = 0;
|
|
|
|
|
for (j = 0; opt.ignore_tags[j] != NULL; j++)
|
2001-01-10 10:54:52 +08:00
|
|
|
|
/* Loop through all the tags this user doesn't care about. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
if (strcasecmp(opt.ignore_tags[j], name) == EQ)
|
|
|
|
|
{
|
|
|
|
|
lose = 1;
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
if (lose)
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (opt.follow_tags)
|
|
|
|
|
{
|
2001-01-10 10:54:52 +08:00
|
|
|
|
/* --follow-tags was specified. Only match these specific tags, so
|
|
|
|
|
continue back to top of for if we don't match one of them. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
int j, win = 0;
|
|
|
|
|
for (j = 0; opt.follow_tags[j] != NULL; j++)
|
|
|
|
|
/* Loop through all the tags this user cares about. */
|
|
|
|
|
if (strcasecmp(opt.follow_tags[j], name) == EQ)
|
|
|
|
|
{
|
|
|
|
|
win = 1;
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
if (!win)
|
2001-01-10 10:54:52 +08:00
|
|
|
|
continue; /* wasn't one of the explicitly desired tags */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/* If we get to here, --follow-tags isn't being used or the
|
2001-01-10 10:54:52 +08:00
|
|
|
|
tag is among the ones that are followed, and --ignore-tags,
|
2000-11-20 04:50:10 +08:00
|
|
|
|
if specified, didn't include this tag, so it's an
|
|
|
|
|
"interesting" one. */
|
|
|
|
|
interesting_tags[ind++] = name;
|
|
|
|
|
}
|
|
|
|
|
interesting_tags[ind] = NULL;
|
|
|
|
|
}
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* The same for attributes, except we loop through tag_url_attributes.
|
2000-11-20 04:50:10 +08:00
|
|
|
|
Here we also need to make sure that the list of attributes is
|
|
|
|
|
unique, and to include the attributes from additional_attributes. */
|
|
|
|
|
{
|
|
|
|
|
int i, ind;
|
|
|
|
|
const char **att = xmalloc ((ARRAY_SIZE (additional_attributes) + 1)
|
|
|
|
|
* sizeof (char *));
|
|
|
|
|
/* First copy the "additional" attributes. */
|
|
|
|
|
for (i = 0; i < ARRAY_SIZE (additional_attributes); i++)
|
|
|
|
|
att[i] = additional_attributes[i];
|
|
|
|
|
ind = i;
|
|
|
|
|
att[ind] = NULL;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
for (i = 0; i < ARRAY_SIZE (tag_url_attributes); i++)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
|
|
|
|
int j, seen = 0;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
const char *look_for = tag_url_attributes[i].attr_name;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
for (j = 0; j < ind - 1; j++)
|
|
|
|
|
if (!strcmp (att[j], look_for))
|
|
|
|
|
{
|
|
|
|
|
seen = 1;
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
if (!seen)
|
|
|
|
|
{
|
|
|
|
|
att = xrealloc (att, (ind + 2) * sizeof (*att));
|
|
|
|
|
att[ind++] = look_for;
|
|
|
|
|
att[ind] = NULL;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
interesting_attributes = att;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
static int
|
|
|
|
|
find_tag (const char *tag_name)
|
|
|
|
|
{
|
|
|
|
|
int i;
|
|
|
|
|
|
|
|
|
|
/* This is linear search; if the number of tags grow, we can switch
|
|
|
|
|
to binary search. */
|
|
|
|
|
|
|
|
|
|
for (i = 0; i < ARRAY_SIZE (known_tags); i++)
|
|
|
|
|
{
|
|
|
|
|
int cmp = strcasecmp (known_tags[i].name, tag_name);
|
|
|
|
|
/* known_tags are sorted alphabetically, so we can
|
|
|
|
|
micro-optimize. */
|
|
|
|
|
if (cmp > 0)
|
|
|
|
|
break;
|
|
|
|
|
else if (cmp == 0)
|
|
|
|
|
return i;
|
|
|
|
|
}
|
|
|
|
|
return -1;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/* Find the value of attribute named NAME in the taginfo TAG. If the
|
2001-12-12 23:43:01 +08:00
|
|
|
|
attribute is not present, return NULL. If ATTRIND is non-NULL, the
|
|
|
|
|
index of the attribute in TAG will be stored there. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
static char *
|
2001-12-12 23:43:01 +08:00
|
|
|
|
find_attr (struct taginfo *tag, const char *name, int *attrind)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
|
|
|
|
int i;
|
|
|
|
|
for (i = 0; i < tag->nattrs; i++)
|
|
|
|
|
if (!strcasecmp (tag->attrs[i].name, name))
|
|
|
|
|
{
|
2001-12-12 23:43:01 +08:00
|
|
|
|
if (attrind)
|
|
|
|
|
*attrind = i;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
return tag->attrs[i].value;
|
|
|
|
|
}
|
|
|
|
|
return NULL;
|
|
|
|
|
}
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
struct map_context {
|
2000-11-20 04:50:10 +08:00
|
|
|
|
char *text; /* HTML text. */
|
|
|
|
|
char *base; /* Base URI of the document, possibly
|
|
|
|
|
changed through <base href=...>. */
|
|
|
|
|
const char *parent_base; /* Base of the current document. */
|
|
|
|
|
const char *document_file; /* File name of this document. */
|
|
|
|
|
int nofollow; /* whether NOFOLLOW was specified in a
|
|
|
|
|
<meta name=robots> tag. */
|
2001-12-12 23:43:01 +08:00
|
|
|
|
|
|
|
|
|
struct urlpos *head, *tail; /* List of URLs that is being
|
|
|
|
|
built. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
};
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* Append LINK_URI to the urlpos structure that is being built.
|
|
|
|
|
|
|
|
|
|
LINK_URI will be merged with the current document base. TAG and
|
|
|
|
|
ATTRIND are the necessary context to store the position and
|
|
|
|
|
size. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-11-26 02:40:55 +08:00
|
|
|
|
static struct urlpos *
|
2001-12-12 23:43:01 +08:00
|
|
|
|
append_one_url (const char *link_uri, int inlinep,
|
|
|
|
|
struct taginfo *tag, int attrind, struct map_context *ctx)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
2001-11-25 11:10:34 +08:00
|
|
|
|
int link_has_scheme = url_has_scheme (link_uri);
|
|
|
|
|
struct urlpos *newel;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
const char *base = ctx->base ? ctx->base : ctx->parent_base;
|
2001-11-25 11:10:34 +08:00
|
|
|
|
struct url *url;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
|
|
|
|
if (!base)
|
|
|
|
|
{
|
2001-11-25 11:10:34 +08:00
|
|
|
|
DEBUGP (("%s: no base, merge will use \"%s\".\n",
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx->document_file, link_uri));
|
2001-11-25 11:10:34 +08:00
|
|
|
|
|
|
|
|
|
if (!link_has_scheme)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
2001-12-13 01:01:26 +08:00
|
|
|
|
/* Base URL is unavailable, and the link does not have a
|
|
|
|
|
location attached to it -- we have to give up. Since
|
|
|
|
|
this can only happen when using `--force-html -i', print
|
|
|
|
|
a warning. */
|
|
|
|
|
logprintf (LOG_NOTQUIET,
|
2001-12-13 02:32:17 +08:00
|
|
|
|
_("%s: Cannot resolve incomplete link %s.\n"),
|
2001-12-13 01:01:26 +08:00
|
|
|
|
ctx->document_file, link_uri);
|
2001-11-26 02:40:55 +08:00
|
|
|
|
return NULL;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
2001-11-25 11:10:34 +08:00
|
|
|
|
|
|
|
|
|
url = url_parse (link_uri, NULL);
|
|
|
|
|
if (!url)
|
|
|
|
|
{
|
|
|
|
|
DEBUGP (("%s: link \"%s\" doesn't parse.\n",
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx->document_file, link_uri));
|
2001-11-26 02:40:55 +08:00
|
|
|
|
return NULL;
|
2001-11-25 11:10:34 +08:00
|
|
|
|
}
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
|
|
|
|
else
|
2001-11-25 11:10:34 +08:00
|
|
|
|
{
|
|
|
|
|
/* Merge BASE with LINK_URI, but also make sure the result is
|
|
|
|
|
canonicalized, i.e. that "../" have been resolved.
|
|
|
|
|
(parse_url will do that for us.) */
|
|
|
|
|
|
|
|
|
|
char *complete_uri = uri_merge (base, link_uri);
|
|
|
|
|
|
|
|
|
|
DEBUGP (("%s: merge(\"%s\", \"%s\") -> %s\n",
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx->document_file, base, link_uri, complete_uri));
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-11-25 11:10:34 +08:00
|
|
|
|
url = url_parse (complete_uri, NULL);
|
|
|
|
|
if (!url)
|
|
|
|
|
{
|
|
|
|
|
DEBUGP (("%s: merged link \"%s\" doesn't parse.\n",
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx->document_file, complete_uri));
|
2001-11-25 11:10:34 +08:00
|
|
|
|
xfree (complete_uri);
|
2001-11-26 02:40:55 +08:00
|
|
|
|
return NULL;
|
2001-11-25 11:10:34 +08:00
|
|
|
|
}
|
|
|
|
|
xfree (complete_uri);
|
|
|
|
|
}
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-13 01:01:26 +08:00
|
|
|
|
DEBUGP (("appending \"%s\" to urlpos.\n", url->url));
|
|
|
|
|
|
2001-11-25 11:10:34 +08:00
|
|
|
|
newel = (struct urlpos *)xmalloc (sizeof (struct urlpos));
|
2000-11-20 04:50:10 +08:00
|
|
|
|
memset (newel, 0, sizeof (*newel));
|
2001-12-12 23:43:01 +08:00
|
|
|
|
|
2000-11-20 04:50:10 +08:00
|
|
|
|
newel->next = NULL;
|
2001-11-25 11:10:34 +08:00
|
|
|
|
newel->url = url;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
newel->pos = tag->attrs[attrind].value_raw_beginning - ctx->text;
|
|
|
|
|
newel->size = tag->attrs[attrind].value_raw_size;
|
|
|
|
|
newel->link_inline_p = inlinep;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-11-19 08:12:05 +08:00
|
|
|
|
/* A URL is relative if the host is not named, and the name does not
|
|
|
|
|
start with `/'. */
|
2001-11-25 11:10:34 +08:00
|
|
|
|
if (!link_has_scheme && *link_uri != '/')
|
2000-11-21 10:06:36 +08:00
|
|
|
|
newel->link_relative_p = 1;
|
2001-11-25 11:10:34 +08:00
|
|
|
|
else if (link_has_scheme)
|
2000-11-21 10:06:36 +08:00
|
|
|
|
newel->link_complete_p = 1;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
if (ctx->tail)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx->tail->next = newel;
|
|
|
|
|
ctx->tail = newel;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
|
|
|
|
else
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx->tail = ctx->head = newel;
|
2001-11-26 02:40:55 +08:00
|
|
|
|
|
|
|
|
|
return newel;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
2001-12-12 23:43:01 +08:00
|
|
|
|
|
|
|
|
|
/* All the tag_* functions are called from collect_tags_mapper, as
|
|
|
|
|
specified by KNOWN_TAGS. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-13 01:01:26 +08:00
|
|
|
|
/* Default tag handler: collect URLs from attributes specified for
|
|
|
|
|
this tag by tag_url_attributes. */
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
|
|
|
|
static void
|
2001-12-12 23:43:01 +08:00
|
|
|
|
tag_find_urls (int tagid, struct taginfo *tag, struct map_context *ctx)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
2001-12-12 23:43:01 +08:00
|
|
|
|
int i, attrind, first = -1;
|
|
|
|
|
int size = ARRAY_SIZE (tag_url_attributes);
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
for (i = 0; i < size; i++)
|
|
|
|
|
if (tag_url_attributes[i].tagid == tagid)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* We've found the index of tag_url_attributes where the
|
2001-12-13 01:01:26 +08:00
|
|
|
|
attributes of our tag begin. */
|
2001-11-17 03:44:42 +08:00
|
|
|
|
first = i;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
break;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
2001-12-12 23:43:01 +08:00
|
|
|
|
assert (first != -1);
|
|
|
|
|
|
|
|
|
|
/* Loop over the "interesting" attributes of this tag. In this
|
|
|
|
|
example, it will loop over "src" and "lowsrc".
|
|
|
|
|
|
|
|
|
|
<img src="foo.png" lowsrc="bar.png">
|
|
|
|
|
|
|
|
|
|
This has to be done in the outer loop so that the attributes are
|
|
|
|
|
processed in the same order in which they appear in the page.
|
|
|
|
|
This is required when converting links. */
|
|
|
|
|
|
|
|
|
|
for (attrind = 0; attrind < tag->nattrs; attrind++)
|
|
|
|
|
{
|
|
|
|
|
/* Find whether TAG/ATTRIND is a combination that contains a
|
|
|
|
|
URL. */
|
2001-12-13 01:01:26 +08:00
|
|
|
|
char *link = tag->attrs[attrind].value;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
|
|
|
|
|
/* If you're cringing at the inefficiency of the nested loops,
|
2001-12-13 01:01:26 +08:00
|
|
|
|
remember that they both iterate over a laughably small
|
|
|
|
|
quantity of items. The worst-case inner loop is for the IMG
|
|
|
|
|
tag, which has three attributes. */
|
2001-12-12 23:43:01 +08:00
|
|
|
|
for (i = first; i < size && tag_url_attributes[i].tagid == tagid; i++)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
2001-12-12 23:43:01 +08:00
|
|
|
|
if (0 == strcasecmp (tag->attrs[attrind].name,
|
|
|
|
|
tag_url_attributes[i].attr_name))
|
|
|
|
|
{
|
|
|
|
|
int flags = tag_url_attributes[i].flags;
|
2001-12-13 01:01:26 +08:00
|
|
|
|
append_one_url (link, !(flags & TUA_EXTERNAL), tag, attrind, ctx);
|
2001-12-12 23:43:01 +08:00
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
2001-11-26 02:40:55 +08:00
|
|
|
|
|
2001-12-13 01:01:26 +08:00
|
|
|
|
/* Handle the BASE tag, for <base href=...>. */
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
static void
|
|
|
|
|
tag_handle_base (int tagid, struct taginfo *tag, struct map_context *ctx)
|
|
|
|
|
{
|
|
|
|
|
struct urlpos *base_urlpos;
|
|
|
|
|
int attrind;
|
|
|
|
|
char *newbase = find_attr (tag, "href", &attrind);
|
|
|
|
|
if (!newbase)
|
|
|
|
|
return;
|
|
|
|
|
|
|
|
|
|
base_urlpos = append_one_url (newbase, 0, tag, attrind, ctx);
|
|
|
|
|
if (!base_urlpos)
|
|
|
|
|
return;
|
|
|
|
|
base_urlpos->ignore_when_downloading = 1;
|
|
|
|
|
base_urlpos->link_base_p = 1;
|
|
|
|
|
|
|
|
|
|
if (ctx->base)
|
|
|
|
|
xfree (ctx->base);
|
|
|
|
|
if (ctx->parent_base)
|
|
|
|
|
ctx->base = uri_merge (ctx->parent_base, newbase);
|
|
|
|
|
else
|
|
|
|
|
ctx->base = xstrdup (newbase);
|
|
|
|
|
}
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2002-04-12 01:51:45 +08:00
|
|
|
|
/* Mark the URL found in <form action=...> for conversion. */
|
|
|
|
|
|
|
|
|
|
static void
|
|
|
|
|
tag_handle_form (int tagid, struct taginfo *tag, struct map_context *ctx)
|
|
|
|
|
{
|
|
|
|
|
int attrind;
|
|
|
|
|
char *action = find_attr (tag, "action", &attrind);
|
|
|
|
|
if (action)
|
|
|
|
|
{
|
|
|
|
|
struct urlpos *action_urlpos = append_one_url (action, 0, tag,
|
|
|
|
|
attrind, ctx);
|
|
|
|
|
if (action_urlpos)
|
|
|
|
|
action_urlpos->ignore_when_downloading = 1;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2001-12-13 01:01:26 +08:00
|
|
|
|
/* Handle the LINK tag. It requires special handling because how its
|
|
|
|
|
links will be followed in -p mode depends on the REL attribute. */
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
static void
|
|
|
|
|
tag_handle_link (int tagid, struct taginfo *tag, struct map_context *ctx)
|
|
|
|
|
{
|
|
|
|
|
int attrind;
|
|
|
|
|
char *href = find_attr (tag, "href", &attrind);
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-19 09:15:34 +08:00
|
|
|
|
/* All <link href="..."> link references are external, except those
|
|
|
|
|
known not to be, such as style sheet and shortcut icon:
|
|
|
|
|
|
|
|
|
|
<link rel="stylesheet" href="...">
|
|
|
|
|
<link rel="shortcut icon" href="...">
|
|
|
|
|
*/
|
2001-12-12 23:43:01 +08:00
|
|
|
|
if (href)
|
|
|
|
|
{
|
|
|
|
|
char *rel = find_attr (tag, "rel", NULL);
|
2001-12-19 09:15:34 +08:00
|
|
|
|
int inlinep = (rel
|
|
|
|
|
&& (0 == strcasecmp (rel, "stylesheet")
|
|
|
|
|
|| 0 == strcasecmp (rel, "shortcut icon")));
|
2001-12-12 23:43:01 +08:00
|
|
|
|
append_one_url (href, inlinep, tag, attrind, ctx);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2001-12-13 01:01:26 +08:00
|
|
|
|
/* Handle the META tag. This requires special handling because of the
|
|
|
|
|
refresh feature and because of robot exclusion. */
|
2001-12-12 23:43:01 +08:00
|
|
|
|
|
|
|
|
|
static void
|
|
|
|
|
tag_handle_meta (int tagid, struct taginfo *tag, struct map_context *ctx)
|
|
|
|
|
{
|
|
|
|
|
char *name = find_attr (tag, "name", NULL);
|
|
|
|
|
char *http_equiv = find_attr (tag, "http-equiv", NULL);
|
|
|
|
|
|
|
|
|
|
if (http_equiv && 0 == strcasecmp (http_equiv, "refresh"))
|
|
|
|
|
{
|
2001-12-13 01:01:26 +08:00
|
|
|
|
/* Some pages use a META tag to specify that the page be
|
|
|
|
|
refreshed by a new page after a given number of seconds. The
|
|
|
|
|
general format for this is:
|
|
|
|
|
|
|
|
|
|
<meta http-equiv=Refresh content="NUMBER; URL=index2.html">
|
|
|
|
|
|
|
|
|
|
So we just need to skip past the "NUMBER; URL=" garbage to
|
|
|
|
|
get to the URL. */
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
struct urlpos *entry;
|
|
|
|
|
int attrind;
|
|
|
|
|
int timeout = 0;
|
2002-02-01 11:34:31 +08:00
|
|
|
|
char *p;
|
|
|
|
|
|
|
|
|
|
char *refresh = find_attr (tag, "content", &attrind);
|
|
|
|
|
if (!refresh)
|
|
|
|
|
return;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
|
|
|
|
|
for (p = refresh; ISDIGIT (*p); p++)
|
|
|
|
|
timeout = 10 * timeout + *p - '0';
|
|
|
|
|
if (*p++ != ';')
|
|
|
|
|
return;
|
|
|
|
|
|
|
|
|
|
while (ISSPACE (*p))
|
|
|
|
|
++p;
|
|
|
|
|
if (!( TOUPPER (*p) == 'U'
|
|
|
|
|
&& TOUPPER (*(p + 1)) == 'R'
|
|
|
|
|
&& TOUPPER (*(p + 2)) == 'L'
|
|
|
|
|
&& *(p + 3) == '='))
|
|
|
|
|
return;
|
|
|
|
|
p += 4;
|
|
|
|
|
while (ISSPACE (*p))
|
|
|
|
|
++p;
|
|
|
|
|
|
|
|
|
|
entry = append_one_url (p, 0, tag, attrind, ctx);
|
|
|
|
|
if (entry)
|
|
|
|
|
{
|
|
|
|
|
entry->link_refresh_p = 1;
|
|
|
|
|
entry->refresh_timeout = timeout;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
else if (name && 0 == strcasecmp (name, "robots"))
|
|
|
|
|
{
|
|
|
|
|
/* Handle stuff like:
|
|
|
|
|
<meta name="robots" content="index,nofollow"> */
|
|
|
|
|
char *content = find_attr (tag, "content", NULL);
|
|
|
|
|
if (!content)
|
|
|
|
|
return;
|
|
|
|
|
if (!strcasecmp (content, "none"))
|
|
|
|
|
ctx->nofollow = 1;
|
|
|
|
|
else
|
|
|
|
|
{
|
|
|
|
|
while (*content)
|
|
|
|
|
{
|
|
|
|
|
/* Find the next occurrence of ',' or the end of
|
|
|
|
|
the string. */
|
|
|
|
|
char *end = strchr (content, ',');
|
|
|
|
|
if (end)
|
|
|
|
|
++end;
|
|
|
|
|
else
|
|
|
|
|
end = content + strlen (content);
|
|
|
|
|
if (!strncasecmp (content, "nofollow", end - content))
|
|
|
|
|
ctx->nofollow = 1;
|
|
|
|
|
content = end;
|
|
|
|
|
}
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
/* Examine name and attributes of TAG and take appropriate action
|
|
|
|
|
according to the tag. */
|
|
|
|
|
|
|
|
|
|
static void
|
|
|
|
|
collect_tags_mapper (struct taginfo *tag, void *arg)
|
|
|
|
|
{
|
|
|
|
|
struct map_context *ctx = (struct map_context *)arg;
|
|
|
|
|
int tagid;
|
|
|
|
|
tag_handler_t handler;
|
|
|
|
|
|
|
|
|
|
tagid = find_tag (tag->name);
|
|
|
|
|
assert (tagid != -1);
|
|
|
|
|
handler = known_tags[tagid].handler;
|
|
|
|
|
|
|
|
|
|
handler (tagid, tag, ctx);
|
|
|
|
|
}
|
|
|
|
|
|
2001-04-25 08:50:22 +08:00
|
|
|
|
/* Analyze HTML tags FILE and construct a list of URLs referenced from
|
2001-11-26 02:40:55 +08:00
|
|
|
|
it. It merges relative links in FILE with URL. It is aware of
|
2001-12-01 05:17:53 +08:00
|
|
|
|
<base href=...> and does the right thing. */
|
2001-11-25 11:10:34 +08:00
|
|
|
|
struct urlpos *
|
2001-12-01 05:17:53 +08:00
|
|
|
|
get_urls_html (const char *file, const char *url, int *meta_disallow_follow)
|
2000-11-20 04:50:10 +08:00
|
|
|
|
{
|
|
|
|
|
struct file_memory *fm;
|
2001-12-12 23:43:01 +08:00
|
|
|
|
struct map_context ctx;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
|
|
|
|
/* Load the file. */
|
|
|
|
|
fm = read_file (file);
|
|
|
|
|
if (!fm)
|
|
|
|
|
{
|
|
|
|
|
logprintf (LOG_NOTQUIET, "%s: %s\n", file, strerror (errno));
|
|
|
|
|
return NULL;
|
|
|
|
|
}
|
|
|
|
|
DEBUGP (("Loaded %s (size %ld).\n", file, fm->length));
|
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
ctx.text = fm->content;
|
|
|
|
|
ctx.head = ctx.tail = NULL;
|
|
|
|
|
ctx.base = NULL;
|
|
|
|
|
ctx.parent_base = url ? url : opt.base_href;
|
|
|
|
|
ctx.document_file = file;
|
|
|
|
|
ctx.nofollow = 0;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
|
|
|
|
if (!interesting_tags)
|
|
|
|
|
init_interesting ();
|
|
|
|
|
|
|
|
|
|
map_html_tags (fm->content, fm->length, interesting_tags,
|
2001-12-12 23:43:01 +08:00
|
|
|
|
interesting_attributes, collect_tags_mapper, &ctx);
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
DEBUGP (("no-follow in %s: %d\n", file, ctx.nofollow));
|
2000-11-20 04:50:10 +08:00
|
|
|
|
if (meta_disallow_follow)
|
2001-12-12 23:43:01 +08:00
|
|
|
|
*meta_disallow_follow = ctx.nofollow;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
|
2001-12-12 23:43:01 +08:00
|
|
|
|
FREE_MAYBE (ctx.base);
|
2000-11-20 04:50:10 +08:00
|
|
|
|
read_file_free (fm);
|
2001-12-12 23:43:01 +08:00
|
|
|
|
return ctx.head;
|
2000-11-20 04:50:10 +08:00
|
|
|
|
}
|
2000-11-23 06:15:45 +08:00
|
|
|
|
|
|
|
|
|
void
|
|
|
|
|
cleanup_html_url (void)
|
|
|
|
|
{
|
|
|
|
|
FREE_MAYBE (interesting_tags);
|
|
|
|
|
FREE_MAYBE (interesting_attributes);
|
|
|
|
|
}
|