| 
									
										
										
										
											1998-08-10 19:42:37 +00:00
										 |  |  | \section{\module{urlparse} --- | 
					
						
							| 
									
										
										
										
											2000-08-25 17:29:35 +00:00
										 |  |  |          Parse URLs into components} | 
					
						
							| 
									
										
										
										
											1998-07-23 17:59:49 +00:00
										 |  |  | \declaremodule{standard}{urlparse} | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-08-06 21:23:17 +00:00
										 |  |  | \modulesynopsis{Parse URLs into components.} | 
					
						
							| 
									
										
										
										
											1998-07-23 17:59:49 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | \index{WWW} | 
					
						
							| 
									
										
										
										
											1995-03-17 16:07:09 +00:00
										 |  |  | \index{World-Wide Web} | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | \index{URL} | 
					
						
							|  |  |  | \indexii{URL}{parsing} | 
					
						
							|  |  |  | \indexii{relative}{URL} | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											1995-02-28 17:14:32 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											2000-08-25 17:29:35 +00:00
										 |  |  | This module defines a standard interface to break Uniform Resource | 
					
						
							|  |  |  | Locator (URL) strings up in components (addressing scheme, network | 
					
						
							|  |  |  | location, path etc.), to combine the components back into a URL | 
					
						
							|  |  |  | string, and to convert a ``relative URL'' to an absolute URL given a | 
					
						
							|  |  |  | ``base URL.'' | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-01-21 04:55:02 +00:00
										 |  |  | The module has been designed to match the Internet RFC on Relative | 
					
						
							|  |  |  | Uniform Resource Locators (and discovered a bug in an earlier | 
					
						
							| 
									
										
										
										
											2000-08-24 04:58:25 +00:00
										 |  |  | draft!). | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | 
 | 
					
						
							|  |  |  | It defines the following functions: | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											1997-12-29 19:09:37 +00:00
										 |  |  | \begin{funcdesc}{urlparse}{urlstring\optional{, default_scheme\optional{, allow_fragments}}} | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | Parse a URL into 6 components, returning a 6-tuple: (addressing | 
					
						
							|  |  |  | scheme, network location, path, parameters, query, fragment | 
					
						
							|  |  |  | identifier).  This corresponds to the general structure of a URL: | 
					
						
							|  |  |  | \code{\var{scheme}://\var{netloc}/\var{path};\var{parameters}?\var{query}\#\var{fragment}}. | 
					
						
							|  |  |  | Each tuple item is a string, possibly empty. | 
					
						
							|  |  |  | The components are not broken up in smaller parts (e.g. the network | 
					
						
							|  |  |  | location is a single string), and \% escapes are not expanded. | 
					
						
							| 
									
										
										
										
											1995-03-17 16:07:09 +00:00
										 |  |  | The delimiters as shown above are not part of the tuple items, | 
					
						
							|  |  |  | except for a leading slash in the \var{path} component, which is | 
					
						
							|  |  |  | retained if present. | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | 
 | 
					
						
							|  |  |  | Example: | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \begin{verbatim} | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | urlparse('http://www.cwi.nl:80/%7Eguido/Python.html')
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \end{verbatim} | 
					
						
							| 
									
										
										
										
											2000-08-24 04:58:25 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | yields the tuple | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \begin{verbatim} | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | ('http', 'www.cwi.nl:80', '/%7Eguido/Python.html', '', '', '')
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \end{verbatim} | 
					
						
							| 
									
										
										
										
											2000-08-24 04:58:25 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | If the \var{default_scheme} argument is specified, it gives the | 
					
						
							|  |  |  | default addressing scheme, to be used only if the URL string does not | 
					
						
							|  |  |  | specify one.  The default value for this argument is the empty string. | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | If the \var{allow_fragments} argument is zero, fragment identifiers | 
					
						
							|  |  |  | are not allowed, even if the URL's addressing scheme normally does | 
					
						
							|  |  |  | support them.  The default value for this argument is \code{1}. | 
					
						
							|  |  |  | \end{funcdesc} | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | \begin{funcdesc}{urlunparse}{tuple} | 
					
						
							| 
									
										
										
										
											1998-01-21 04:55:02 +00:00
										 |  |  | Construct a URL string from a tuple as returned by \code{urlparse()}. | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | This may result in a slightly different, but equivalent URL, if the | 
					
						
							|  |  |  | URL that was parsed originally had redundant delimiters, e.g. a ? with | 
					
						
							|  |  |  | an empty query (the draft states that these are equivalent). | 
					
						
							|  |  |  | \end{funcdesc} | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-03-17 06:33:25 +00:00
										 |  |  | \begin{funcdesc}{urljoin}{base, url\optional{, allow_fragments}} | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | Construct a full (``absolute'') URL by combining a ``base URL'' | 
					
						
							|  |  |  | (\var{base}) with a ``relative URL'' (\var{url}).  Informally, this | 
					
						
							|  |  |  | uses components of the base URL, in particular the addressing scheme, | 
					
						
							|  |  |  | the network location and (part of) the path, to provide missing | 
					
						
							|  |  |  | components in the relative URL. | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | Example: | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \begin{verbatim} | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | urljoin('http://www.cwi.nl/%7Eguido/Python.html', 'FAQ.html')
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \end{verbatim} | 
					
						
							| 
									
										
										
										
											2000-08-24 04:58:25 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | yields the string | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \begin{verbatim} | 
					
						
							| 
									
										
										
										
											1995-04-10 11:34:00 +00:00
										 |  |  | 'http://www.cwi.nl/%7Eguido/FAQ.html'
 | 
					
						
							| 
									
										
										
										
											1998-02-13 06:58:54 +00:00
										 |  |  | \end{verbatim} | 
					
						
							| 
									
										
										
										
											2000-08-25 17:29:35 +00:00
										 |  |  | 
 | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | The \var{allow_fragments} argument has the same meaning as for | 
					
						
							| 
									
										
										
										
											1998-01-21 04:55:02 +00:00
										 |  |  | \code{urlparse()}. | 
					
						
							| 
									
										
										
										
											1995-02-27 17:53:25 +00:00
										 |  |  | \end{funcdesc} | 
					
						
							| 
									
										
										
										
											2000-08-24 04:58:25 +00:00
										 |  |  | 
 | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | \begin{seealso} | 
					
						
							|  |  |  |   \seerfc{1738}{Uniform Resource Locators (URL)}{ | 
					
						
							|  |  |  |         This specifies the formal syntax and semantics of absolute | 
					
						
							|  |  |  |         URLs.} | 
					
						
							|  |  |  |   \seerfc{1808}{Relative Uniform Resource Locators}{ | 
					
						
							|  |  |  |         This Request For Comments includes the rules for joining an | 
					
						
							|  |  |  |         absolute and a relative URL, including a fair normal of | 
					
						
							|  |  |  |         ``Abnormal Examples'' which govern the treatment of border | 
					
						
							|  |  |  |         cases.} | 
					
						
							| 
									
										
										
										
											2000-08-25 17:29:35 +00:00
										 |  |  |   \seerfc{2396}{Uniform Resource Identifiers (URI): Generic Syntax}{ | 
					
						
							|  |  |  |         Document describing the generic syntactic requirements for | 
					
						
							|  |  |  |         both Uniform Resource Names (URNs) and Uniform Resource | 
					
						
							|  |  |  |         Locators (URLs).} | 
					
						
							| 
									
										
										
										
											2000-08-24 04:58:25 +00:00
										 |  |  | \end{seealso} |