Whoops, forgot to edit WHATSNEW
[htmlpurifier.git] / library / HTMLPurifier / Injector / Linkify.php
blob74f83eaa7d16ad8a5e4455243756e46559b4082a
1 <?php
3 /**
4 * Injector that converts http, https and ftp text URLs to actual links.
5 */
6 class HTMLPurifier_Injector_Linkify extends HTMLPurifier_Injector
8 /**
9 * @type string
11 public $name = 'Linkify';
13 /**
14 * @type array
16 public $needed = array('a' => array('href'));
18 /**
19 * @param HTMLPurifier_Token $token
21 public function handleText(&$token)
23 if (!$this->allowsElement('a')) {
24 return;
27 if (strpos($token->data, '://') === false) {
28 // our really quick heuristic failed, abort
29 // this may not work so well if we want to match things like
30 // "google.com", but then again, most people don't
31 return;
34 // there is/are URL(s). Let's split the string.
35 // We use this regex:
36 // https://gist.github.com/gruber/249502
37 // but with @cscott's backtracking fix and also
38 // the Unicode characters un-Unicodified.
39 $bits = preg_split(
40 '/\\b((?:[a-z][\\w\\-]+:(?:\\/{1,3}|[a-z0-9%])|www\\d{0,3}[.]|[a-z0-9.\\-]+[.][a-z]{2,4}\\/)(?:[^\\s()<>]|\\((?:[^\\s()<>]|(?:\\([^\\s()<>]+\\)))*\\))+(?:\\((?:[^\\s()<>]|(?:\\([^\\s()<>]+\\)))*\\)|[^\\s`!()\\[\\]{};:\'".,<>?\x{00ab}\x{00bb}\x{201c}\x{201d}\x{2018}\x{2019}]))/iu',
41 $token->data, -1, PREG_SPLIT_DELIM_CAPTURE);
44 $token = array();
46 // $i = index
47 // $c = count
48 // $l = is link
49 for ($i = 0, $c = count($bits), $l = false; $i < $c; $i++, $l = !$l) {
50 if (!$l) {
51 if ($bits[$i] === '') {
52 continue;
54 $token[] = new HTMLPurifier_Token_Text($bits[$i]);
55 } else {
56 $token[] = new HTMLPurifier_Token_Start('a', array('href' => $bits[$i]));
57 $token[] = new HTMLPurifier_Token_Text($bits[$i]);
58 $token[] = new HTMLPurifier_Token_End('a');
64 // vim: et sw=4 sts=4