PageSourceSearch

https://tantek.com/2026/249/t1/cassis.js

js tantek.com collected 2026-10-01 07:27:30 UTC 58,558 bytes, 1,975 lines download raw bytes

1/* <!--
2   cassis.js Copyright 2008-2024 Tantek Çelik https://tantek.com 
3   http://cassisproject.com conceived:2008-254; created:2009-299;
4   license: https://creativecommons.org/licenses/by-sa/4.0/       -->
5if you see this or "/// var" in the browser, you need to 
6wrap your PHP include of cassis.js AND your use of functions therein 
7with calls to ob_start and ob_end_clean, e.g.:
8ob_start();
9include 'cassis.js';
10// your code that calls CASSIS functions like auto_link('@tantek.com') goes here
11ob_end_clean();
12/* <!-- <?php // CASSIS v0.1 start -->
13// ===================================================================
14// PHP-only block. Processed only by PHP. Use only // comments here.
15// -------------------------------------------------------------------
16function js() {
17  return false;
18}
19
20// global configuration
21
22if (php_min_version("5.1.0")) {
23  date_default_timezone_set("UTC");
24}
25
26function php_min_version($s) {
27  $s = explode(".",$s);
28  $phpv = explode(".",phpversion());
29  for ($i=0;$i<count($s);$i++) {
30    if ($s[$i]>$phpv[$i]) {
31      return false; 
32    }
33  }
34  return true;
35}
36
37
38
39// -------------------------------------------------------------------
40// string functions requiring separate js/php definitions
41
42function preg_matches($p, $s) {
43  $m = array();
44  if (preg_match_all($p, $s, $m, PREG_PATTERN_ORDER) !== FALSE) {
45    return $m[0];
46  }
47  else {
48    return array();
49  }
50}
51
52// -------------------------------------------------------------------
53// date time functions
54
55function date_get_full_year($d = "") {
56  if ($d == "") {
57    $d = new DateTime();
58  }
59  return $d->format('Y');
60} 
61
62function date_get_timestamp($d = "") { 
63  if ($d == "") {
64    $d = new DateTime();
65  }
66  return $d->format('U'); // $d->getTimestamp(); // in PHP 5.3+
67}
68
69function date_get_ordinal_days($d) {
70 return 1 + $d->format('z');
71}
72
73function date_get_rfc3339($d) {
74  return $d->format('c');
75}
76
77// -------------------------------------------------------------------
78// old wrappers. transition code away from these
79// ** do not use these in new code. **
80
81function getFullYear($d = "") {  
82  // 2010-020 obsoleted. Use date_get_full_year instead
83  return date_get_full_year($d);
84}
85
86// ===================================================================
87/*/ // This comment inverter switches from PHP only to JS only.
88// JS-only block. Processed only by JS. Use only // comments here.
89// -------------------------------------------------------------------
90function js() {
91  return true;
92}
93
94$debug = false; // change to true in developer console to debug in JS
95$GLOBALS = []; // pacify JS when PHP code access this array
96
97// array functions
98
99function array() { // makes an array from arbitrary parameter list.
100  return Array.prototype.slice.call(arguments);
101}
102
103function is_array(a) {
104  return (typeof(a) === "object") && (a instanceof Array);
105}
106
107function count(a) {
108  return a.length;
109}
110
111function array_slice(a, b, e) { // slice an array, begin, optional end
112  if (a === undefined) { return array(); }
113  if (b === undefined) { return a; }
114  if (e === undefined) { return a.slice(b); }
115  return a.slice(b, e);
116}
117
118// -------------------------------------------------------------------
119// math and numerical functions
120
121function floor(n) {
122  return Math.floor(n);
123}
124
125function intval(n) {
126  return parseInt(n, 10);
127}
128
129Array.min = function(a) { 
130// from http://ejohn.org/blog/fast-javascript-maxmin/
131  return Math.min.apply(Math, a);
132};
133
134function min() {
135  var m = arguments;
136  if (m.length < 1) {
137    return false;
138  } 
139  if (m.length === 1) {
140    m = m[0];
141    if (!is_array(m)) {
142      return m;
143    }
144  }
145  return Array.min(m);
146}
147
148function ctype_digit(s) {
149 return (/^[0-9]+$/).test(s);
150}
151
152function ctype_lower(s) {
153 return (/^[a-z]+$/).test(s);
154}
155
156function ctype_space(s) {
157 return (/\s/).test(s);
158}
159
160// -------------------------------------------------------------------
161// date time functions
162
163function date_create(s) {
164  if (s) return new Date(s);
165  else return new Date();
166}
167
168function date_get_full_year(d) {
169  if (arguments.length < 1) {
170    d = new Date();
171  }
172  return d.getFullYear();
173}
174
175function date_get_timestamp(d) {
176  return floor(d.getTime() / 1000);
177}
178
179function date_get_rfc3339($d) {
180  return strcat($d.getFullYear(),'-',
181                str_pad_left(1 + $d.getUTCMonth(), 2, "0"), '-',
182                str_pad_left($d.getDate(), 2, "0"), 'T',
183                str_pad_left($d.getUTCHours(), 2, "0"), ':',
184                str_pad_left($d.getUTCMinutes(), 2, "0"), ':',
185                str_pad_left($d.getUTCSeconds(), 2, "0"), 'Z');
186}
187
188// newcal
189
190function date_get_ordinal_days($d) {
191  return ymdp_to_d($d.getFullYear(), 1 + $d.getMonth(), $d.getDate());
192}
193
194
195// -------------------------------------------------------------------
196// character and string functions 
197
198function ord(s) {
199  return s.charCodeAt(0);
200}
201
202function strlen(s) {
203  return s.length;
204} 
205
206function substr(s, o, n) {
207  var m = strlen(s);
208  if ((o < 0 ? -1-o : o) >= m) { return ""; }
209  if (o < 0) { o = m + o; }
210  if (n === undefined) { n = m - o; }
211  if (n < 0) { n = m - o + n; }
212  return s.substring(o, o + n);
213}
214
215function substr_count(s, n) {
216 return s.split(n).length - 1;
217}
218
219function strpos(h, n, o) {
220  // clients must triple-equal test return for === false for no match!
221  // or use offset(n, h) instead (0 = not found, else 1-based index)
222  if (arguments.length === 2) {
223    o = 0;
224  }
225  o = h.indexOf(n, o);
226  if (o === -1) { return false; }
227  else { return o; }
228}
229
230function stripos(h, n, o) {
231  // clients must triple-equal test return for === false for no match!
232  if (arguments.length === 2) {
233    o = 0;
234  }
235  o = h.toLowerCase().indexOf(n.toLowerCase(), o);
236  if (o === -1) { return false; }
237  else { return o; }
238}
239
240function strncmp(s1, s2, n) {
241  s1 = substr(String(s1), 0, n);
242  s2 = substr(String(s2), 0, n);
243  return (s1 === s2) ? 0 :
244         ((s1 < s2) ? -1 : 1);
245}
246
247function explode(d, s, n) {
248  if (arguments.length === 2) {
249    return s.split(d);
250  }
251  return s.split(d, n);
252}
253
254function implode(d, a) {
255  return a.join(d);
256}
257
258function rawurlencode(s) {
259  return encodeURIComponent(s);
260}
261
262function htmlspecialchars(s) {
263 var c= [["&","&amp;"],["<","&lt;"],[">","&gt;"],["'","&#039;"],['"',"&quot;"]];
264 for (i=0;i<c.length;i++) {
265  s = s.replace(new RegExp(c[i][0],"g"),c[i][1]); // s.replace(c[i][0],c[i][1]);
266 }
267 return s;
268}
269
270function str_ireplace(a, b, s) {
271 var i;
272 if (!is_array(a)) {
273   return s.replace(new RegExp(a, "gi"), is_array(b) ? b[0] : b);
274 }
275 else {
276   for (i=0; i<a.length; i++) {
277     s = s.replace(new RegExp(a[i], "gi"), is_array(b) ? b[i] : b);
278   }
279   return s;
280 }
281}
282
283function preg_match(p, s) {
284  return (s.match(trim_slashes(p)) ? 1 : 0);
285}
286
287function preg_split(p, s) {
288  return s.split(new RegExp(trim_slashes(p),"gi")); // possibly off by one
289}
290
291function trim() {
292 var m = arguments;
293 var s = m[0];
294 var c = count(m)>1 ? m[1] : " \t\n\r\f\x00\x0b\xa0";
295 var i = 0;
296 var j = strlen(s);
297 while (contains(c,s[i]) && i<j) {
298   i++;
299 }
300 --j;
301 while (j>i && contains(c,s[j])) {
302   --j;
303 }
304 j++;
305 if (j>i) {
306   return substr(s,i,j-i);
307 }
308 else {
309   return '';
310 }
311}
312
313function rtrim() {
314 var m = arguments;
315 var s = m[0];
316 var c = count(m)>1 ? m[1] : " \t\n\r\f\x00\x0b\xa0";
317 var j = strlen(s)-1;
318 while (j>=0 && contains(c,s[j])) {
319   --j;
320 }
321 if (j>=0) {
322   return substr(s,0,j+1);
323 }
324 else {
325   return '';
326 }
327}
328
329function strtolower(s) {
330  return s.toLowerCase();
331}
332
333function ucfirst(s) {
334  return s.charAt(0).toUpperCase() + substr(s, 1);
335}
336
337// -------------------------------------------------------------------
338// more javascript-only php-equivalent functions here 
339
340
341// javascript-only framework functions
342function targetelement(e) {
343  var t;
344  e = e ? e : window.event;
345  t = e.target ? e.target : e.srcElement;
346  t = (t.nodeType == 3) ? t.parentNode : t; // Safari workaround
347  return t;
348}
349
350function doevent(el, evt) {
351  if (evt=="click" && el.tagName=='A') {
352  // note: dispatch/fireEvent not work FF3.5+/IE8+ on [a href] w "click" event
353    window.location = el.href; // workaround
354    return true;
355  }
356  if (document.createEvent) {
357    var eo = document.createEvent("HTMLEvents");
358    eo.initEvent(evt, true, true);
359    return !el.dispatchEvent(eo);
360  } 
361  else if (document.createEventObject) {
362    return el.fireEvent("on"+evt);
363  }
364}
365
366
367// -------------------------------------------------------------------
368// string functions requiring separate js/php definitions
369
370function preg_matches($p, $s) {
371  return $s.match(new RegExp(trim_slashes($p),"gi")); // match is a keyword in PHP 8.0
372}
373
374
375// old wrappers. transition code away from them, do not use them in new code.
376//function getFullYear(d) {       // use date_get_full_year instead
377//  return date_get_full_year(d);
378//}
379
380
381// end cassis0php.js
382// --------------------------------------------------------------------
383
384/**/ // unconditional comment closer enters PHP+javascript processing
385/* ------------------------------------------------------------------ */
386/* cassis0.js - processed by both PHP and javascript */
387
388
389// -------------------------------------------------------------------
390// character and string functions
391
392function strcat() { // takes as many strings as you want to give it.
393 $strcatr = "";
394 $isjs = js();
395 $args = $isjs ? arguments : func_get_args();
396 for ($strcati=count($args)-1; $strcati>=0; $strcati--) {
397    $strcatr = $isjs ? $args[$strcati] + $strcatr : $args[$strcati] . $strcatr;
398 }
399 return $strcatr;
400}
401
402function number($s) {
403 return $s - 0;
404}
405
406function string($n) {
407 if (js()) { 
408   if (typeof($n)=="number")
409     return Number($n).toString(); 
410   else if (typeof($n)=="undefined")
411     return "";
412   else return $n.toString();
413 }
414 else { return "" . $n; }
415}
416
417function str_pad_left($s1,$n,$s2) {
418 $s1 = string($s1);
419 $s2 = string($s2);
420 if (js()) {
421   $n -= strlen($s1);
422   while ($n >= strlen($s2)) { 
423     $s1 = strcat($s2,$s1); 
424     $n -= strlen($s2);
425   }
426   if ($n > 0) {
427     $s1 = strcat(substr($s2,0,$n),$s1);
428   }
429   return $s1;
430 }
431 else { return str_pad($s1,$n,$s2,STR_PAD_LEFT); }
432}
433
434function trim_slashes($s) {
435  if ($s[0]=="/") { // strip unnecessary / delimiters that PHP regexp funcs want
436    return substr($s,1,strlen($s)-2);
437  }
438  return $s;
439}
440
441/* end cassis0.js */
442
443function ctype_post_slug($s) {
444 // Falcon: post slugs should only have lowercase, numbers, or '-', or '_'
445 return (preg_match("/^[a-z0-9]+([_-][a-z0-9]+)*$/", $s));
446}
447
448function ctype_email_local($s) {
449 // close enough. no '.' because this is used for last char of.
450 return (preg_match("/^[a-zA-Z0-9_%+-]+$/", $s));
451}
452
453function ctype_uri_scheme($s) {
454 return (preg_match("/^[a-zA-Z][a-zA-Z0-9+.-]*$/", $s));
455}
456
457function ctype_time($s) { // whether start of a string is a time
458  switch (offset(':', $s)) {
459  case 2:
460    return ctype_digit(substr($s, 0, 1)) && ctype_digit(substr($s, 2, 2));
461    break;
462  case 3:
463    return ctype_digit(substr($s, 0, 2)) && ctype_digit(substr($s, 4, 2));  
464    break;
465  default:
466    return false;
467  }
468}
469
470// -------------------------------------------------------------------
471// newbase60
472
473function num_to_sxg($n) {
474 $s = "";
475 $p = "";
476 $m = "0123456789ABCDEFGHJKLMNPQRSTUVWXYZ_abcdefghijkmnopqrstuvwxyz";
477 if ($n==="" || $n===0) { return "0"; }
478 if ($n<0) {
479   $n = 0-$n;
480   $p = "-";
481 }
482 while ($n>0) {
483   $d = $n % 60;
484   $s = strcat($m[$d],$s);
485   $n = ($n-$d)/60;
486 }
487 return strcat($p,$s);
488}
489
490function num_to_sxgf($n, $f) {
491  if (!$f) { $f=1; }
492  return str_pad_left(num_to_sxg($n), $f, "0");
493}
494
495function sxg_to_num($s) {
496 $n = 0;
497 $m = 1;
498 $j = strlen($s);
499 if ($s[0]=="-") {
500   $m= -1;
501   $j--;
502   $s = substr($s,1,$j);
503 }
504 for ($i=0;$i<$j;$i++) { // iterate from first to last char of $s
505   $c = ord($s[$i]); //  put current ASCII of char into $c  
506   if ($c>=48 && $c<=57) { $c=$c-48; }
507   else if ($c>=65 && $c<=72) { $c-=55; }
508   else if ($c==73 || $c==108) { $c=1; } // typo capital I, lowercase l to 1
509   else if ($c>=74 && $c<=78) { $c-=56; }
510   else if ($c==79) { $c=0; } // error correct typo capital O to 0
511   else if ($c>=80 && $c<=90) { $c-=57; }
512   else if ($c==95 || $c==45) { $c=34; } // _ underscore and correct dash - to _
513   else if ($c>=97 && $c<=107) { $c-=62; }
514   else if ($c>=109 && $c<=122) { $c-=63; }
515   else break; // treat all other noise as end of number
516   $n = 60*$n + $c;
517 }
518 return $n*$m;
519}
520
521function sxg_to_numf($s, $f) {
522  if ($f===undefined) { $f=1; }
523  return str_pad_left(sxg_to_num($s), $f, "0");
524}
525
526// -------------------------------------------------------------------
527// == newbase60 compat functions only == (before 2011-149)
528function numtosxg($n) {
529  return num_to_sxg($n);
530}
531
532function numtosxgf($n, $f) {
533  return num_to_sxgf($n, $f);
534}
535
536function sxgtonum($s) {
537  return sxg_to_num($s);
538}
539
540function sxgtonumf($s, $f) {
541  return sxg_to_numf($s, $f);
542}
543/* == end compat functions == */
544
545// -------------------------------------------------------------------
546// date and time
547
548function date_create_ymd($s) {
549 if (!$s) {
550   return (js() ? new Date() : new DateTime());
551 }
552 if (js()) { 
553   if (substr($s,4,1)=='-') {
554      $s=strcat(strcat(substr($s,0,4),substr($s,5,2)),substr($s,8,2));
555   }
556   $d = new Date(substr($s,0,4),substr($s,4,2)-1,substr($s,6,2));
557   $d.setHours(0); // was setUTCHours, avoiding bc JS has no default timezone
558   return $d;
559 }
560 else { return date_create(strcat($s," 00:00:00")); }
561}
562
563function date_create_timestamp($s) {
564  if (js()) {
565    return new Date(1000*$s);
566  }
567  else {
568    return new DateTime(strcat("@", string($s)));
569  }
570}
571
572// function date_get_timestamp($d) // in PHP/JS specific code above.
573
574// function date_get_rfc3339($d) // in PHP/JS specific code above.
575
576function dt_to_time($dt) {
577  $dt = explode("T", $dt);
578  if (count($dt)==1) {
579    $dt = explode(" ", $dt);
580  }
581  return (count($dt)>1) ? $dt[1] : "0:00";
582}
583
584function dt_to_date($dt) {
585  $dt = explode("T", $dt);
586  if (count($dt)==1) {
587    $dt = explode(" ", $dt);
588  }
589  return $dt[0];
590}
591
592function dt_to_ordinal_date($dt) {
593  return ymd_to_yd(dt_to_date($dt));
594}
595
596// -------------------------------------------------------------------
597// newcal
598
599function isleap($y) {
600  return ($y % 4 === 0 && ($y % 100 !== 0 || $y % 400 === 0));
601}
602
603function ymdp_to_d($y,$m,$d) {
604  $md = array(
605         array(0,31,59,90,120,151,181,212,243,273,304,334),
606         array(0,31,60,91,121,152,182,213,244,274,305,335));
607  return $md[number(isleap($y))][$m-1] + number($d);
608}
609
610function ymd_to_d($d) {
611  if (substr($d, 4, 1)==='-') {
612    return ymdp_to_d(substr($d,0,4),substr($d,5,2),substr($d,8,2));
613  }
614  else {
615    return ymdp_to_d(substr($d,0,4),substr($d,4,2),substr($d,6,2));
616  }
617}
618
619function ymdp_to_yd($y, $m, $d) {
620  return strcat(str_pad_left($y, 4, "0"), '-',
621                str_pad_left(ymdp_to_d($y, $m, $d), 3, "0"));
622}
623
624function ymd_to_yd($d) {
625  if (substr($d, 4, 1)==='-') {
626    return ymdp_to_yd(substr($d,0,4),substr($d,5,2),substr($d,8,2));
627  }
628  else {
629    return ymdp_to_yd(substr($d,0,4),substr($d,4,2),substr($d,6,2));
630  }
631}
632
633// function date_get_ordinal_days($d) // in PHP/JS specific code above
634
635function bim_from_od($d) {
636  return 1+floor(($d-1)/61);
637}
638
639function date_get_bim() {
640  $args = js() ? arguments : func_get_args();
641  return bim_from_od(
642          date_get_ordinal_days(
643           date_create_ymd((count($args) > 0) ? $args[0] : 0)));
644}
645
646function get_nm_str($m) {
647  $a = array("New January", "New February", "New March", "New April", "New May", "New June", "New July", "New August", "New September", "New October", "
647New November", "New December");
648  return $a[($m-1)];
649}
650
651function nm_from_od($d) {
652  return ((($d-1) % 61) > 29) ? 2+2*(bim_from_od($d)-1) : 1+2*(bim_from_od($d)-1);
653}
654
655function date_get_ordinal_date(/* $d = "" */) {
656  $args = js() ? arguments : func_get_args();
657  $d = date_create_ymd((count($args) > 0) ? $args[0] : 0);
658  return strcat(date_get_full_year($d), '-',
659                str_pad_left(date_get_ordinal_days($d), 3, "0"));
660}
661
662// -------------------------------------------------------------------
663// begin epochdays
664
665function y_to_days($y) {
666  // convert y-01-01 to epoch days
667  return floor(
668   (date_get_timestamp(date_create_ymd(strcat($y, "-01-01"))) -
669    date_get_timestamp(date_create_ymd("1970-01-01")))/86400);
670}
671
672// convert ymd to epoch days and sexagesimal epoch days (sd)
673
674function ymd_to_days($d) {
675  return yd_to_days(ymd_to_yd($d));
676}
677
678/* old:
679function ymd_to_days($d) {
680  // fails in JS, "2013-03-10" and "2013-03-11" both return 15774 
681  return floor((date_get_timestamp(date_create_ymd($d))-date_get_timestamp(date_create_ymd("1970-01-01")))/86400);
682}
683*/
684
685function ymd_to_sd($d) {
686  return num_to_sxg(ymd_to_days($d));
687}
688
689function ymd_to_sdf($d, $f) {
690  return num_to_sxgf(ymd_to_days($d), $f);
691}
692
693// ordinal date (YYYY-DDD) to ymd, epoch days, sexagesimal epoch days
694
695function ydp_to_ymd($y,$d) {
696  $md = array(
697         array(0,31,59,90,120,151,181,212,243,273,304,334,365),
698         array(0,31,60,91,121,152,182,213,244,274,305,335,366));
699  $d -= 1;
700  $m = trunc($d / 29);
701  if ($md[isleap($y) - 0][$m] > $d) $m -= 1;
702  $d = $d - $md[isleap($y)-0][$m] + 1;
703  $m += 1;
704  return strcat($y, '-', str_pad_left($m, 2, '0'), 
705                    '-', str_pad_left($d, 2, '0'));
706}
707
708function yd_to_ymd($d) {
709  return ydp_to_ymd(substr($d, 0, 4), substr($d, 5, 3));
710}
711
712function yd_to_days($d) {
713  return y_to_days(substr($d, 0, 4)) - 1 + number(substr($d, 5, 3));
714}
715
716function yd_to_sd($d) {
717  return num_to_sxg(yd_to_days($d));
718}
719
720function yd_to_sdf($d, $f) {
721  return num_to_sxgf(yd_to_days($d), $f);
722}
723
724// convert epoch days or sexagesimal epoch days (sd) to ordinal date
725
726function days_to_yd($d) {
727  $d = date_create_timestamp(
728         date_get_timestamp(
729           date_create_ymd("1970-01-01")) + $d*86400);
730  $y = date_get_full_year($d);
731  $a = date_create_ymd(strcat($y,"-01-01"));
732  return strcat($y, strcat("-", str_pad_left(1+floor((date_get_timestamp($d)-date_get_timestamp($a))/86400), 3, "0")));
733}
734
735function sd_to_yd($d) {
736  return days_to_yd(sxg_to_num($d));
737}
738
739// -------------------------------------------------------------------
740// compat as of 2011-143
741function bimfromod($d) { return bim_from_od($d); }
742function getnmstr($m) { return get_nm_str($m); }
743function nmfromod($d) { return nm_from_od($d); }
744function ymdptod($y,$m,$d) { return ymdp_to_d($y,$m,$d); }
745function ymdptoyd($y,$m,$d) { return ymdp_to_yd($y,$m,$d); }
746function ymdtoyd($d) { return ymd_to_yd($d); }
747function ymdtodays($d) { return ymd_to_days($d); }
748function ymdtosd($d) { return ymd_to_sd($d); }
749function ymdtosdf($d,$f) { return ymd_to_sdf($d, $f); }
750function ydtodays($d) { return yd_to_days($d); }
751function ydtosd($d) { return yd_to_sd($d); }
752function ydtosdf($d,$f) { return yd_to_sdf($d, $f); }
753function daystoyd($d) { return days_to_yd($d); }
754function sdtoyd($d) { return sd_to_yd($d); }
755
756/* end epochdays */
757
758
759/* ------------------------------------------------------------------ */
760
761
762// -------------------------------------------------------------------
763// webaddress
764
765function web_address_to_uri($wa, $addhttp) {
766  if ($wa=='' 
767      || (substr($wa, 0, 7) == "http://") 
768      || (substr($wa, 0, 8) == "https://") 
769      || (substr($wa, 0, 6) == "irc://")) {
770    return $wa;
771  }
772  if ((substr($wa, 0, 7) == "Http://") 
773      || (substr($wa, 0, 8) == "Https://")) { // handle iPad overcapitalization of input entries
774    return strcat('h', substr($wa, 1, strlen($wa)));
775  }
776  
777  if (substr($wa, 0, 1) == "@") {
778    return strcat("https://twitter.com/", substr($wa, 1, strlen($wa)));
779  }
780
781  if ($addhttp) { // NOTE: does not handle protocol relative URLs
782    $wa = strcat('http://', $wa);
783  }
784  return $wa;
785}
786
787function uri_clean($uri) {
788  $uri = web_address_to_uri($uri, false);
789  // prune the optional http:// for a neater param
790  if (substr($uri, 0, 7) === 'http://') {
791    $uri = explode('://', $uri);
792    $uri = array_slice($uri, 1);
793    $uri = implode('://', $uri);
794  }
795  // URL encode
796  return str_ireplace("%3A", ":", 
797                      str_ireplace("%2F", "/", rawurlencode($uri)));
798}
799
800// returns e.g. http:
801function protocol_of_uri($uri) {
802  if (offset(':', $uri) === 0) { return ""; }
803  $uri = explode(':', $uri, 2);
804  if (!ctype_uri_scheme($uri[0])) { return ""; }
805  return strcat($uri[0], ':');
806}
807
808// returns e.g. //ttk.me/b/4DY1?seriously=yes#ud
809function relative_uri_hash($uri) {
810  if (offset(':', $uri) === 0) { return ""; }
811  $uri = explode(':', $uri, 2);
812  if (!ctype_uri_scheme($uri[0])) { return ""; }
813  return $uri[1];
814}
815
816// returns e.g. ttk.me
817function hostname_of_uri($uri) {
818  $uri = explode('/', $uri, 4);
819  if (count($uri) > 2) {
820    $uri = $uri[2];
821    if (offset(':', $uri) !== 0) {
822      $uri = explode(':', $uri, 2);
823      $uri = $uri[0];
824    }
825    return $uri;
826  }   
827  return '';
828}
829
830function sld_of_uri($uri) {
831  $uri = hostname_of_uri($uri);
832  $uri = explode('.', $uri);
833  if (count($uri) > 1) {
834    return $uri[count($uri) - 2];
835  }
836  return "";
837}
838
839function path_of_uri($uri) {
840  $uri = explode('/', $uri);
841  if (count($uri) > 3) {
842    $uri = array_slice($uri, 3);
843    $uri = strcat('/', implode('/', $uri));
844    if (offset('?', $uri) !== 0) {
845      $uri = explode('?', $uri, 2);
846      $uri = $uri[0];
847    }
848    if (offset('#', $uri) !== 0) {
849      $uri = explode('#', $uri, 2);
850      $uri = $uri[0];
851    }
852    return $uri;    
853  }
854  return '/';
855}
856
857function prepath_of_uri($uri) {
858  $uri = explode('/', $uri);
859  $uri = array_slice($uri, 0, 3);
860  return implode('/', $uri);
861}
862
863function segment_of_uri($n, $u) {
864   /* nth starting at 1 */
865   $u = path_of_uri($u);
866   $u = explode('/', $u);
867   if ($n>=0 && $n<count($u))
868     return $u[$n];
869   else return "";
870}
871
872function fragment_of_uri($u) {
873  if (offset('#', $u) !== 0) {
874    $u = explode('#', $u, 2);
875    return $u[1];
876  }
877  return "";
878}
879
880function is_http_uri($uri) {
881  $uri = explode(':', $uri, 2);
882  return !!strncmp($uri[0], 'http', 4);
883}
884
885function get_absolute_uri($uri, $base) {
886  if (protocol_of_uri($uri) != "") { return $uri; }
887  if (substr($uri, 0, 2) === '//') { 
888    return strcat(protocol_of_uri($base), $uri);
889  }
890  if (substr($uri, 0, 1) === '/') {
891    return strcat(prepath_of_uri($base), $uri);
892  }
893  // TBI # relative
894  return strcat(prepath_of_uri($base), path_of_uri($base), $uri);
895}
896
897// -------------------------------------------------------------------
898// compat as of 2011-149
899function webaddresstouri($wa, $addhttp) { 
900  return web_address_to_uri($wa, $addhttp);
901}
902function uriclean($uri) { return uri_clean($uri); }
903
904// -------------------------------------------------------------------
905// HTTP related
906
907function is_html_type($ct) {
908  $ct = explode(';', $ct, 2);
909  $ct = $ct[0];
910  return ($ct === 'text/html' || $ct === 'application/xhtml+xml');
911}
912
913// -------------------------------------------------------------------
914// hexatridecimal
915
916function numtohxt($n) {
917 $s = "";
918 $m = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ";
919 if ($n===undefined || $n===0) { return "0"; }
920 while ($n>0) {
921   $d = $n % 36;
922   $s = strcat($m[$d],$s);
923   $n = ($n-$d)/36;
924 }
925 return $s;
926}
927
928function numtohxtf($n,$f) {
929 if ($f===undefined) { $f=1; }
930 return str_pad_left(numtohxt($n), $f, "0");
931}
932
933function hxttonum($h) {
934 $n = 0;
935 $j = strlen($h);
936 for ($i=0;$i<$j;$i++) { // iterate from first to last char of $h
937   $c = ord($h[$i]); //  put current ASCII of char into $c  
938   if ($c>=48 && $c<=57) { $c=$c-48; } // 0-9
939   else if ($c>=65 && $c<=90) { $c-=55; } // A-Z
940   else if ($c>=97 && $c<=122) { $c-=87; } // a-z case-insensitive treat as A-Z
941   else { $c = 0; } // treat all other noise as 0
942   $n = 36*$n + $c;
943 }
944 return $n;
945}
946
947/* end hexatridecimal */
948
949
950/* ------------------------------------------------------------------ */
951
952
953/* ISBN-10 */
954
955function numtoisbn10($n) {
956 $n=string($n);
957 $d=0;
958 $f=2;
959 for ($i=strlen($n)-1;$i>=0;$i--) {
960  $d += $n[$i]*$f;
961  $f++;  
962 }
963 $d = 11-($d % 11);
964 if ($d==10) {$d="X";}
965 else if ($d==11) {$d="0";}
966 else {$d=string($d);}
967 return strcat(str_pad_left($n,9,"0"),$d);
968}
969/* end ISBN-10 */
970
971
972/* ------------------------------------------------------------------ */
973
974
975/* ASIN */
976
977function asintorsxg($a) { // ASIN to reversible sexagesimal; prefix ISBN-10 w ~
978 $a = amazontoasin($a); // extract ASIN from Amazon URL if necessary
979 if ($a[0]=='B') {
980   $a=num_to_sxg(hxttonum(substr($a,1,9)));
981 }
982 else {
983   $a = implode("",explode("-",$a)); // eliminate presentational hyphens
984   if (strlen($a)>10 && substr($a,0,3)=="978") {
985     $a = substr($a,3,9);
986   }
987   else {
988     $a = substr($a,0,9);
989   }
990   $a = strcat("~",num_to_sxg($a));
991 }
992 return $a;
993}
994
995function amazontoasin($a) {
996 // idempotent
997 if (preg_match("/[\.\/]+/",$a)) {
998   $a = explode("/",$a);
999   for ($i=count($a)-1; $i>=0; $i--) {
1000     if (preg_match("/^[0-9A-Za-z]{10}$/",$a[$i])) {
1001       $a = $a[$i];
1002       break;
1003     }
1004   }
1005   if ($i==-1) { // no ASIN was found in URL
1006     $a=""; // reset $a to a string (instead of an array)
1007   }
1008 }
1009 return $a;
1010}
1011
1012/* end ASIN */
1013
1014
1015/* ------------------------------------------------------------------ */
1016
1017
1018/* Unicode */
1019
1020function nstr_to_usup($s) {
1021  if ($s===undefined || $s===0) { return '⁰'; }
1022  $r = '';
1023  $usups = array('⁰', '¹', '²', '³', '⁴', '⁵', '⁶', '⁷', '⁸', '⁹');
1024  for ($i=0; $i<strlen($s); $i++) {
1025    $r = strcat($r, $usups[number($s[$i])]);
1026  }
1027  return $r;
1028}
1029
1030
1031/* ------------------------------------------------------------------ */
1032
1033
1034// -------------------------------------------------------------------
1035// HyperTalk
1036
1037function trunc($n) { // just an alias from BASIC days
1038  return floor($n);
1039}
1040
1041function offset($n, $h) {
1042 $n = strpos($h, $n);
1043 if ($n===false) { return 0; }
1044 else            { return $n+1; }
1045}
1046
1047function contains($h, $n) {
1048 // actual HT syntax:haystack contains needle: if ("abc" contains "b")
1049 // return ($n !== '') && !(strpos($h, $n)===false);
1050 return !(strpos($h, $n)===false);
1051}
1052
1053function last_character_of($s) {
1054  return (strlen($s) > 0) ? $s[strlen($s)-1] : '';
1055}
1056/* end HyperTalk */
1057
1058
1059/* ------------------------------------------------------------------ */
1060
1061
1062// -------------------------------------------------------------------
1063// microformats
1064
1065// xpath expressions to extract microformats
1066function xp_has_class($s) {
1067  return strcat("//*[contains(concat(' ',@class,' '),' ",$s," ')]");
1068}
1069
1070function xpr_has_class($s) {
1071  return strcat(".//*[contains(concat(' ',@class,' '),' ",$s," ')]");
1072}
1073
1074function xp_has_id($s) {
1075  return strcat("//*[@id='", $s, "']");
1076}
1077
1078function xp_attr_starts_with($a, $s) {
1079  return strcat("//*[starts-with(@", $a, ",'", $s, "')]");
1080}
1081
1082function xp_has_rel($s) {
1083  return strcat("//*[@href and contains(concat(' ',@rel,' '),' ", $s, " ')]");
1084}
1085
1086function xpr_has_rel($s) {
1087  return strcat(".//*[@href and contains(concat(' ',@rel,' '),' ", $s, " ')]");
1088}
1089
1090function xpr_attr_starts_with_has_rel($a, $s, $r) {
1091  return strcat(".//*[@href and contains(concat(' ',@rel,' '),' ", $r, 
1092                " ') and starts-with(@", $a, ",'", $s, "')]");
1093}
1094
1095function xpr_attr_starts_with_has_class($a, $s, $c) {
1096  return strcat(".//*[contains(concat(' ',@class,' '),' ", $c, " ') and starts-with(@", $a, ",'", $s, "')]");
1097}
1098
1099/* end XPath */
1100
1101
1102/* ------------------------------------------------------------------ */
1103
1104
1105/* microformats */
1106
1107/* value class pattern readable date time from ISO8601 datetime */
1108function vcp_dt_readable($d) {
1109  $d = explode("T", $d);
1110  $r = "";
1111  if (count($d)>1) { 
1112     $r = explode("-", $d[1]);
1113     if (count($d)==1) {
1114			 $r = explode("+", $d[1]);
1115     }
1116     if (count($d)>1) {
1117       $r = strcat('<time class="value" datetime="',$d[1],'">', 
1118                   $r[0],'</time> on ');
1119     }
1120     else {
1121       $r = strcat('<time class="value">', $d[1], '</time> on ');
1122     }
1123  }
1124  return strcat($r, '<time class="value">', $d[0], '</time>');
1125}
1126
1127
1128// -------------------------------------------------------------------
1129// compat as of 2011-149
1130function xphasclass($s) { return xp_has_class($s); }
1131function xprhasclass($s) { return xpr_has_class($s); }
1132function xphasid($s) { return xp_has_id($s); }
1133function xpattrstartswith($a, $s) { 
1134  return xp_attr_starts_with($a, $s); 
1135}
1136function xphasrel($s) { return xp_has_rel($s); }
1137function xprhasrel($s) { return xpr_has_rel($s); }
1138function xprattrstartswithhasrel($a, $s, $r) {
1139  return xpr_attr_starts_with_has_rel($a, $s, $r);
1140}
1141function xprattrstartswithhasclass($a, $s, $c) {
1142  return xpr_attr_starts_with_has_class($a, $s, $c);
1143}
1144function vcpdtreadable($d) { return vcp_dt_readable($d); }
1145
1146
1147// -------------------------------------------------------------------
1148// whistle
1149// algorithmic URL shortener core
1150// YYYY/DDD/tnnn to tdddss 
1151// ordinal date, type, decimal #, to sexagesimal epoch days, sexagesimal #
1152function whistle_short_path($p) {
1153  return strcat(substr($p, 9, 1),
1154                ((substr($p, 9, 1)!=='t') ? "/" : ""),
1155                yd_to_sdf(substr($p, 0, 8), 3),
1156                num_to_sxg(substr($p, 10, 3)));
1157}
1158/* end Whistle */
1159
1160
1161// -------------------------------------------------------------------
1162// Falcon
1163
1164function html_unesc_amp_only($s) {
1165  return str_ireplace('&amp;', '&', $s);
1166}
1167
1168function html_esc_amper_once($s) {
1169  return str_ireplace('&', '&amp;', html_unesc_amp_only($s));
1170}
1171
1172function html_esc_amp_ang($s) {
1173  return str_ireplace('<', '&lt;',
1174         str_ireplace('>', '&gt;', html_esc_amper_once($s)));
1175}
1176
1177function ellipsize_to_word($s, $max, $e, $min) {
1178  if (strlen($s)<=$max) {
1179    return $s; // no need to ellipsize
1180  }
1181
1182  $elen = strlen($e);
1183  $slen = $max-$elen;
1184
1185  // if last characters before $max+1 are ': ', truncate w/o ellipsis.
1186  // no need to take length of ellipsis into account
1187  if ($e=='...') {
1188    for ($ii=1;$ii<=$elen+1;$ii++) {
1189      if (substr($s,$max-$ii,2)==': ') {
1190        return substr($s,0,$max-$ii+1);
1191      }
1192    }
1193  }
1194
1195  if ($min) {
1196    // if a non-zero minimum is provided, then
1197    // find previous space or word punctuation to break at.
1198    // do not break at %`'"&.!?^ - reasons why to be documented.
1199    while ($slen>$min && !contains('@$ -~*()_+[]\{}|;,<>',$s[$slen-1])) {
1200      --$slen;
1201    }
1202  }
1203  // at this point we've got a min length string, 
1204  // only do minimum trimming necessary to avoid a punctuation error.
1205  
1206  // trim slash after colon or slash
1207  if ($s[$slen-1]=='/' && $slen > 2) {
1208    if ($s[$slen-2]==':') {
1209      --$slen;    
1210    }
1211    if ($s[$slen-2]=='/') {
1212      $slen -= 2;
1213    }
1214  }
1215
1216  //if trimmed at a ":" in a URL, trim the whole thing
1217    //or trimmed at "http", trim the whole URL
1218  if ($s[$slen-1]==':' && $slen > 5 && substr($s,$slen-5,5)=='http:') {
1219    $slen -= 5;
1220  }
1221  else if ($s[$slen-1]=='p' && $slen > 4 && substr($s,$slen-4,4)=='http') {
1222    $slen -= 4;
1223  }
1224  else if ($s[$slen-1]=='t' && $slen > 4 && (substr($s,$slen-3,4)=='http' || substr($s,$slen-3,4)==' htt')) {
1225    $slen -= 3;
1226  }
1227  else if ($s[$slen-1]=='h' && $slen > 4 && substr($s,$slen-1,4)=='http') {
1228    $slen -= 1;
1229  }
1230  
1231  // if char immediately before ellipsis would be @$ then trim it
1232  if ($slen > 0 && contains('@$', $s[$slen-1])) {
1233    --$slen;
1234  }
1235 
1236  //if char before ellipsis would be sentence terminator, trim 2 more
1237  while ($slen > 1 && contains('.!?', $s[$slen-1])) {
1238    $slen-=2;
1239  }
1240
1241  // trim extra whitespace before ellipsis down to one space
1242  if ($slen >
1242 2 && contains("\n\r ", $s[$slen-1])) {
1243    while (contains("\n\r ", $s[$slen-2]) && $slen > 2) {
1244      --$slen;
1245    }
1246  }
1247
1248  if ($slen < 1) { // somehow shortened too much
1249    return $e; // or ellipsis by itself exceeded max, return ellipsis.
1250  }
1251
1252  // if last two chars are ': ', omit ellipsis. 
1253  if ($e==='...' && substr($s, $slen-2, 2)===': ') {
1254    return substr($s, 0, $slen);
1255  }
1256
1257  return strcat(substr($s, 0, $slen), $e);
1258}
1259
1260function get_leading_images_alts($s) {
1261  // note: alt text is unescaped, e.g. may contain ' & '
1262  return parse_leading_urls($s, true, false);  
1263}
1264
1265function trim_leading_urls($s) {
1266  // deliberately trim URLs with explicit http: / https: from start
1267  // keep schemeless URLs, @-names as expected user-visible text
1268  // if empty or just space after trimming, just return original
1269  return parse_leading_urls($s, false, true);
1270}
1271
1272function parse_leading_urls($s, $images_only, $remainder) {
1273  // parse for leading URLs with explicit http: / https: from start
1274  // including alt text after image URLs
1275  // if $images_only then also stop at first non-image URL
1276  // if $remainder return remaining string if non-empty or original
1277  // else return array of url,alt strings
1278  $r = trim($s);
1279  $u = array();
1280  while ($r!='' && (substr($r, 0, 5) == 'http:' || substr($r, 0, 6) == 'https:'))
1281  {
1282    $ws = offset(' ', $r);
1283    $rs = offset("\r", $r);
1284    if ($rs == 0) { $rs = offset("\n", $r); }
1285    if ($rs != 0 && $rs < $ws) { $ws = $rs; }
1286    if ($ws == 0) { 
1287      if ($remainder) return $s; 
1288    } else {
1289      $r[$ws-1] = ' ';
1290    }
1291    if (!$remainder) {
1292      $us = ($ws > 0) ? substr($r, 0, $ws-1) : $r;
1293      $as = '';
1294    }
1295    if ($ws > 0) {
1296      $rlen = $ws;
1297    }
1298    else {
1299      $r = strcat($r, ' ');
1300      $rlen = strlen($r);    
1301    }
1302    if (substr($r, ($fe = $rlen-5), 1) === '.' ||
1303        substr($r, ($fe = $rlen-6), 1) === '.')
1304    {
1305      $fe = strtolower(substr($r, $fe, 5)); 
1306      if ($fe == '.gif ' || $fe == '.jpeg' || $fe == '.jpg ' ||
1307          $fe == '.png ' || $fe == '.svg ' || $fe == '.webp')
1308      {
1309        // parse alt text after an image link also
1310
1311        if ($ws > 0 && substr($r, $ws, 1) == '('/*)*/ ) {
1312          // balance for close paren, allow balanced parens in alt
1313          $paren_depth = 1;
1314          $sp_len = strlen($r);
1315          for ($j = $ws+1; $j < $sp_len; $j++) {
1316            switch ($r[$j]) {
1317              case '(': ++$paren_depth; break;
1318              case ')': --$paren_depth; break;
1319            }
1320            if ($paren_depth == 0)
1321              break;
1322          }
1323          if (!$remainder) {
1324            $as = substr($r, $ws+1, $j-$ws-1);
1325          }
1326          if ($j < $sp_len-1) {// if alt closed before end of string, trim it
1327            $ws = $j+1;
1328          }
1329          if (ctype_space($r[$ws])) {
1330            $ws++; // skip a trailing space
1331          }
1332        }
1333      }
1334      else {
1335        if ($images_only) {
1336          $us = '';
1337        }
1338      }
1339    }
1340    else {
1341      if ($images_only) {
1342        $us = '';
1343      }
1344    }
1345    if (!$remainder) {
1346      if ($us != '') {
1347        $u[count($u)] = strcat($us, ' ', $as);
1348      }
1349      else {
1350        return $u;
1351      }
1352    }
1353    if ($ws > 0) {
1354      $r = substr($r, $ws, strlen($r)-$ws);
1355    }
1356    else {
1357      $r = '';
1358    }
1359  }
1360  if (!$remainder) {
1361    return $u;
1362  }
1363  $r = trim($r);
1364  return ((strlen($r) > 0) ? $r : $s);
1365}
1366
1367function auto_space($s) {
1368// replace linebreaks with <br class="auto-break"/>
1369//  and one leading space with &nbsp;
1370// replace "  " with " &nbsp;"
1371// replace leading spaces (on a line or before spaces) with nbsp;
1372// TBI switch from str_ireplace to a line-by-line processor for auto_blocks
1373  if ($s[0] === ' ') {
1374    $s = strcat('&#xA0;', substr($s, 1, strlen($s)-1));
1375  }
1376  return str_ireplace(array("\r\n", "\r", "\n ", "\n", "  "),
1377                      array("\n", "\n", '<br class="auto-break"/>&#xA0;',
1378                            '<br class="auto-break"/>',
1379                            ' &#xA0;'),
1380                      $s);
1381}
1382
1383function auto_link_re() {
1384  return '/(?:\\^[0-9]{1,2}
1384)|(?:(?:(:?\\@|(?:(?:http|https|irc)?:\\/\\/))?(?:(?:(?:[a-zA-Z0-9ŽžÀ-ÿ][-a-zA-Z0-9ŽžÀ-ÿ]*\\.)+(?:(?:aero|app|arpa|asia|a[cdefgilmnoqrstuwxz])|(?:biz|blog|b[abdefghijmnorstvwyz])|(?:cafe|cat|cloud|club|coffee|com|coop|c[acdfghiklmnoruvxyz])|(?:design|dev|dog|d[ejkmoz])|(?:edu|e[cegrstu])|(?:fyi|f[ijkmor])|(?:garden|gov|g[abdefghilmnpqrstuwy])|h[kmnrtu]|(?:info|int|i[delmnoqrst])|j[emop]|k[eghimnrwyz]|(?:lol|l[abcikrstuvy])|(?:mil|museum|m[acdeghklmnopqrstuvwxyz])|(?:name|net|n[acefgilopruz])|(?:org|om|one)|(?:party|pro|pub|p[aefghklmnrstwy])|qa|(?:rocks|r[eouw])|(?:social|space|s[abcdeghijklmnortuvyz])|(?:tech|tel|travel|t[cdfghjklmnoprtvwz])|u[agkmsyz]|v[aceginu]|(?:world|wtf|w[fs])|xyz|y[etu]|(?:zone|z[amw])))|(?:(?:25[0-5]|2[0-4][0-9]|[0-1][0-9]{2}|[1-9][0-9]|[1-9])\\.(?:25[0-5]|2[0-4][0-9]|[0-1][0-9]{2}|[1-9][0-9]|[0-9])\\.(?:25[0-5]|2[0-4][0-9]|[0-1][0-9]{2}|[1-9][0-9]|[0-9])\\.(?:25[0-5]|2[0-4][0-9]|[0-1][0-9]{2}|[1-9][0-9]|[0-9])))(?:\\:\\d{1,5})?)(?:\\/(?:(?:[!#&-;=?-Z_a-z~])|(?:\\%[a-fA-F0-9]{2}))*)?)|(?:\\@[_a-zA-Z0-9]{1,17})(?=\\b|\\s|$)/';
1385  // ccTLD compressed regular expression clauses (re)created.
1386  // .mobi .jobs deliberately excluded to discourage layer violations
1387  // .security .trust also excluded to discourage phishing abuses
1388  // see http://flic.kr/p/2kmuSL for more on the problematic new gTLDs
1389  // part of $re derived from Android Open Source Project, Apache 2.0
1390  // with a bunch of subsequent fixes/improvements (e.g. ttk.me/t44H2)
1391  // thus auto_link_re is also Apache 2.0 licensed
1392  //  http://www.apache.org/licenses/LICENSE-2.0
1393  // - Tantek 2010-046 (moved to auto_link_re 2012-062)
1394}
1395
1396
1397// auto_link: param 1: text; 
1398//  optional: param 2: do embeds & more markup or not (false),
1399//            param 3: do auto_links or not (true)
1400//            param 4: do u-* photo/video upgrade (1st) image (false)
1401//            param 5: do footnotes with fragmentprefix ("")
1402// auto_link is idempotent, works on plain text or typical markup.
1403function auto_link(/*$t*/) {
1404  $isjs = js();
1405  $args = $isjs ? arguments : func_get_args();
1406  if (count($args) === 0) {
1407    return '';
1408  }
1409  $t = $args[0];
1410  $do_embed = (count($args) > 1) && ($args[1]!==false);
1411  $do_link = (count($args) < 3) || ($args[2]!==false);
1412  $do_u_media = (count($args) > 3) && ($args[3]!==false);
1413  $fnote_frag = (count($args) > 4) ? $args[4] : "";
1414  $doing_u_media = false; // do any number in a row
1415  $re = auto_link_re();
1416  $ms = preg_matches($re, $t);
1417  if (!$ms) {
1418    return $t;
1419  }
1420  
1421  $mlen = count($ms);
1422  $sp = preg_split($re, $t);
1423  $t = "";
1424  
1425  if (!js()) { $debug = $GLOBALS["debug"]; }
1426  if ($debug) { $t = strcat('$ms', var_dump($ms), '<br /> '); }
1427//  if ($debug) { $t = strcat($t, '$sp', var_dump($sp), '<br /> '); }
1428
1429
1430  $sp[0] = string($sp[0]); // force undefined to ""
1431  for ($i=0; $i<$mlen; $i++) {
1432    $mi = $ms[$i];
1433    $spliti = $sp[$i];
1434    $t = strcat($t, $spliti);
1435    $sp[$i+1] = string($sp[$i+1]); // force undefined to ""
1436    if (substr($sp[$i+1], 0, 1)=='/') { //regex omits end slash before </a
1437      $sp[$i+1] = substr($sp[$i+1], 1);
1438      $mi = strcat($mi, '/'); // explicitly include it in the match
1439    }
1440    $spe = substr($spliti, -2, 2);
1441
1442    if ($debug) { var_dump($spliti); var_dump($mi); }
1443
1444    // avoid 2x-linking, don't link CSS @-rules, attr values, asciibet
1445    if ((!$spe || !preg_match('/(?:\\=[\\"\\\']?|t;)/', $spe)) &&
1446        substr(trim($sp[$i+1]), 0, 3)!='</a' && 
1447        (!contains('@charset@font@font-face@import@media@namespace@page@supports@ABCDEFGHIJKLMNOPQ@',
1448                   strcat($mi, '@'))))
1449    {
1450      $afterlink = '';
1451      $afterchar = substr($mi, -1, 1);
1452      if (contains($mi, '(') && $afterchar!=')' &&
1453          substr($sp[$i+1], 0, 1)===')') {
1454        $mi = strcat($mi, ')');
1455        $afterchar = ')';
1456        $sp[$i+1] = substr($sp[$i+1], 1);
1457      }
1458      while (contains('.!?,:;"\')]}', $afterchar) && //trim end puncts
1459          ($afterchar!=')' || !contains($mi, '('))) { // allow a ()
1460          $afterlink = strcat($afterchar, $afterlink);
1461          $mi = substr($mi, 0, -1);
1462          $afterchar = substr($mi, -1, 1);
1463      }
1464      
1465      $fe = 0;
1466      if ($do_embed && strlen($mi) > 5) {
1467         $fe = strtolower(
1468                (substr($mi, -4, 1) === '.') ? substr($mi, -4, 4) 
1469                                             : substr($mi, -5, 5));
1470      }
1471      $wmi = web_address_to_uri($mi, true);
1472      $prot = protocol_of_uri($wmi);
1473      $hn = hostname_of_uri($wmi);
1474      $pa = path_of_uri($wmi);
1475      $ih = is_http_uri($wmi);
1476
1477      $ahref = '<span class="figure" style="text-align:left">';
1478      $enda = '</span>';
1479			if ($do_link) {
1480        $ahref = strcat('<a class="auto-link figure" href="',      
1481                        $wmi, '">');
1482        $enda = '</a>';
1483      }
1484
1485      if ($fe && 
1486          ($fe === '.jpeg' || $fe === '.jpg' || 
1487           $fe === '.png' || $fe === '.gif' || $fe === '.svg' || $fe === '.webp' ||
1488           $fe === '.mp4' )) // hack for IG mp4 for u-video
1489      {
1490        $alt = strcat('a ',
1491                      (offset('photo', $mi) != 0) ? 'photo' 
1492                                                  : substr($fe, 1),
1493                      '. ');
1494        $media_class = 'auto-embed';
1495        $poster = '';
1496        if (($i === 0 || $doing_u_media) && 
1497             // check first URL for u-photo upgrade, or sequential
1498            $do_u_media) {
1499          if ($fe === '.mp4') {
1500						$media_class = strcat($media_class, ' u-video');          
1501          }
1502          else {
1503						$media_class = strcat($media_class, ' u-photo');
1504          }
1505          $doing_u_media = true;
1506        }
1507        if ($i+1 < $mlen &&
1508            $afterlink === '' &&
1509            ((contains("\n\r ", $sp[$i+1][0]) && 
1510             (strlen($sp[$i+1]) == 1 || $sp[$i+1][1] == '('/*)*/)) 
1511             || $sp[$i+1] == '<br class="auto-break"/>')) {
1512          // if the non-URL after a photo/video is space or line-break
1513          // or if there's a (1 of "\n\r ")+"(" after, use as alt text til ")"
1514          // and there's a URL afterwards, link it
1515
1516          if ($sp[$i+1][1] == '('/*)*/) {
1517            // alt text found, balance for close paren, allow balanced parens in alt
1518            $alt = ''; // set empty alt by default since alt was explicitly set
1519            $paren_depth = 1;
1520            $sp_len = strlen($sp[$i+1]);
1521            for ($j = 2; $j < $sp_len; $j++) {
1522              switch ($sp[$i+1][$j]) {
1523                case '(': ++$paren_depth; break;
1524                case ')': --$paren_depth; break;
1525              }
1526              if ($paren_depth == 0)
1527                break;
1528            }
1529            $alt = substr($sp[$i+1], 2, $j-2);
1530            $sp[$i+1] = ($j < $sp_len-1) ? substr($sp[$i+1], $j+1, $sp_len-$j-1) : '';
1531          }
1532          if (contains("\n\r ", $sp[$i+1]) || $sp[$i+1] == '<br class="auto-break"/>') {
1533            $sp[$i+1] = ''; // consume any remaining single space or line-break
1534          }
1535
1536          $m1 = $ms[$i+1];
1537					$acm1 = substr($m1, -1, 1);
1538					if (contains($m1, '(') && $acm1!=')' &&
1539							substr($sp[$i+2], 0, 1)===')') {
1540						$m1 = strcat($m1, ')');
1541						$acm1 = ')';
1542						$sp[$i+2] = substr($sp[$i+2], 1);
1543					}
1544          while (contains('.!?,:;"\')]}', $acm1) && //trim end puncts
1545          ($acm1!=')' || !contains($m1, '('))) { // allow a ()
1546						$afterlink = strcat($acm1, $afterlink);
1547						$m1 = substr($m1, 0, -1);
1548						$acm1 = substr($m1, -1, 1);
1549          }
1550          $ms[$i+1] = $m1;
1551
1552          if ($afterlink==='' && $sp[$i+2] != '') { // fix the URL after if necessary
1553            if (substr($sp[$i+2], 0, 1) == '/') {
1554							// if regex pushed a trailing slash to the sp
1555							$sp[$i+2] = substr($sp[$i+2], 1, strlen($sp[$i+2]) - 1);
1556							$ms[$i+1] = strcat($ms[$i+1], '/'); // include in match
1557            }
1558						if (contains("\n\r ", substr($sp[$i+2], 0, 1))) {
1559							// consume blank space or linebreak after the link
1560							// TBI: look for a third space alt-text string
1561							// http://tantek.com/w/Markdown#Alttextforimages
1562							$sp[$i+2] = substr($sp[$i+2], 1, strlen($sp[$i+2]) - 1);
1563						}					
1564						if (substr($sp[$i+2], 0, 24)=='<br class="auto-break"/>')
1565						{
1566							// consume next auto_space linebreak (if called first)
1567							$sp[$i+2] = substr($sp[$i+2], 24, strlen($sp[$i+2])-24);
1568						}
1569					}
1570          // check second link for poster image of a video
1571          if ($fe == '.mp4') {
1572 					  $fe2 = strtolower((substr($ms[$i+1], -4, 1) === '.') 
1573 					                    ? substr($ms[$i+1], -4, 4)
1574 					                    : substr($ms[$i+1], -5, 5));
1575 					  if ($fe2 && 
1576                ($fe2 === '.jpeg' || $fe2 === '.jpg' || 
1577                 $fe2 === '.png' || $fe2 === '.gif' || $fe2 === '.webp')) {
1578              $poster = $ms[$i+1];
1579              // poster image found. now also check for a link after!
1580							if ($i+2<$mlen &&
1581									$afterlink === '' &&
1582									($sp[$i+2] == '' || contains("\n\r ", $sp[$i+2]) ||
1583									 $sp[$i+2] == '<br class="auto-break"/>')) {
1584          // if the non-URL after the poster is space or line-break
1585          // and there's a URL afterwards, link it
1586								$sp[$i+2] = ''; // consume single space or line-break
1587                // fix URL after if necessary
1588                if ($sp[$i+3] != '') {
1589									if (substr($sp[$i+3], 0, 1) == '/') {
1590										// if regex pushed a trailing slash to the sp
1591										$sp[$i+3] = substr($sp[$i+3], 1, strlen($sp[$i+3]) - 1);
1592										$ms[$i+2] = strcat($ms[$i+2], '/'); //add to match
1593									}
1594									if (contains("\n\r ", substr($sp[$i+3], 0, 1))) {
1595										// No need to look for a third space alt-text string, because posters can't alt
1596										// consume blank space or linebreak after the link
1597										$sp[$i+3] = substr($sp[$i+3], 1, 
1598										                   strlen($sp[$i+3]) - 1);
1599									}					
1600									if (substr($sp[$i+3], 0, 24) == 
1601									    '<br class="auto-break"/>')
1602									{
1603										// consume next auto_space linebreak (if called first)
1604										$sp[$i+3] = substr($sp[$i+3], 24,
1605										                   strlen($sp[$i+3])-24);
1606									}
1607								}
1608                $i++; // skip handling the poster separately
1609              }
1610            }
1611          }
1612          $poster_only = ($poster === $ms[$i+1]);
1613          // TBI: should poster be a (linked) fallback image?
1614					$a_class = "auto-link";
1615					if ($fe !== '.mp4') {$a_class=strcat($a_class, ' figure');}
1616					$ig_link = contains($ms[$i+1], 'instagram.com/p/');
1617					if (contains($media_class, 'u-') &&
1618					    ($ig_link ||
1619					     contains($ms[$i+1],
1620					              'commons.wikimedia.org/wiki/File:'))) {
1621					  $a_class = strcat($a_class, ' u-syndication');   
1622					}
1623					if (contains($media_class, 'u-photo') && 
1624					    contains($ms[$i+1], '4sqi.net/img/general/original/')) {
1625            // was: move u-photo to higher resolution original jpg URL
1626					  // $a_class = strcat($a_class, ' u-photo');
1627            // $media_class = 'auto-embed';
1628            // need to keep u-photo on img for the alt text to work!
1629            // Bridgy Publish syndicate higher resolution photo to Flickr
1630					  $a_class = strcat($a_class, ' u-bridgy-flickr-photo');
1631					}
1632					$ahref = strcat('<a class="', $a_class, '" href="',
1633					                $poster_only 
1634					                ? $wmi
1635					                : $ms[$i+1], '">');
1636          $i++; // skip handling the link separately
1637        }
1638        if ($fe === '.mp4') { // more hack
1639          if ($poster) { $poster = strcat('poster="', $poster, '" ');}
1640					$t = strcat($t, 
1641					            '<span class="figure"><video class="',
1642					            $media_class, '"',
1643					            ($ig_link ? ' loop="loop" ' : ' '), $poster,
1644											'controls="controls" src="', $wmi, '">', 
1645											$ahref, 'a video', $enda, '</video></span>', $afterlink);
1646        } else {
1647          $t = strcat($t, $ahref, '<img class="', $media_class, 
1648                      '" alt="', $alt, '" src="', $wmi, '"/>', 
1649                      $enda, $afterlink);
1650        }
1651      } else if ($fe && 
1652                 ($fe === '.mp4' || $fe === '.mov' || 
1653                  $fe === '.ogv' || $fe === '.webm'))
1654      {
1655        $t = strcat($t, $ahref, 
1656                    '<span class="figure"><video class="auto-embed" ',
1657                    'controls="controls" src="', $wmi, '">a video</video></span>',
1658                    $enda, $afterlink);
1659      } else if ($hn === 'vimeo.com' 
1660                     && ctype_digit(substr($pa, 1)))
1661      {
1662				if ($do_link) {
1663				  $t = strcat($t, '<a class="auto-link" href="',
1664				              'https:', relative_uri_hash($wmi),
1665                      '">', $mi, '</a> ');
1666				}
1667        if ($do_embed) {
1668          $t = strcat($t, '<iframe class="vimeo-player auto-embed figure" width="480" height="385" style="border:0" src="', 'https://player.vimeo.com/video/', 
1669                      substr($pa, 1), '"></iframe>', 
1670                      $afterlink);
1671        }
1672      } else if ($hn === 'youtu.be' ||
1673                (($hn === 'youtube.com' || $hn === 'www.youtube.com')
1674                 && ($yvid = offset('watch?v=', $mi)) !== 0))
1675      {
1676        if ($hn === 'youtu.be') {
1677          $yvid = substr($pa, 1);
1678        }
1679        else {
1680          $yvid = explode('&', substr($mi, $yvid+7));
1681          $yvid = $yvid[0];
1682        }
1683				if ($do_link) {
1684  				$t = strcat($t, '<a class="auto-link" href="',
1685  				            'https:', relative_uri_hash($wmi),
1686                      '">', $mi, '</a> ');
1687        }
1688        if ($do_embed) {
1689          $t = strcat($t, '<iframe class="youtube-player auto-embed figure" width="480" height="385" style="border:0"  src="', 'https://www.youtube.com/embed/', 
1690                      $yvid, '"></iframe>', 
1691                      $afterlink);
1692        }
1693      } else if ($mi[0] === '^' && $do_link) {
1694        // convert footnote to Unicode and hyperlink
1695				if ($debug) { var_dump($debug); var_dump($afterlink);  $t=strcat($t,'§ffa','§',string(strlen($afterlink)), '§'); } 
1696				// $afterlink should be $sp[$i+1] but that crashes somewhere
1697        if ($fnote_frag!='' && !contains('"\'', $afterlink[0])) { // if not quoted example
1698					if ($debug) { $t=strcat($t, '§', $fnote_frag, '§ff'); }
1699					$fnote_num = substr($mi, 1);
1700					$mi = nstr_to_usup($fnote_num); // convert number string to Unicode superscripts
1701					$fnote_exp_id = strcat($fnote_frag, '_note-', $fnote_num);
1702					$fnote_ref_id = strcat($fnote_frag, '_ref-', $fnote_num);
1703					if (contains("\r\n", substr($spliti, -1, 1)) ||
1704  	  			  substr($spliti, -5, 5) == '<br/>' ||
1705  	  			  substr($spliti, -6, 6) == '<br />' ||
1706					    substr($spliti, -14, 14) == '"auto-break"/>') { // before ^ is a linebreak
1707						// TBI? append ⮐ after note expansion hyperlinked to inline ref
1708						// create <a id=$fnote_exp_id href=#,$fnote_ref_id > uni-num </a>
1709						$mi = strcat('<a id="', $fnote_exp_id, '" href="#', $fnote_ref_id, '">', 
1710													$mi, '</a>');
1711					} else {
1712						// create <a id=$fnote_ref_id href=#,$fnote_exp_id > uni-num </a>
1713						$mi = strcat('<a id="', $fnote_ref_id, '" href="#', $fnote_exp_id, '">', 
1714													$mi, '</a>');
1715					}
1716        }
1717        $t = strcat($t, $mi, $afterlink);
1718      } else if ($do_link) {
1719				$extra_class = '';
1720        if ($mi[0] === '@') { 
1721          $wmi = substr($mi, 1); // $spliti
1722          // link @name@domainpath @domainpath@domainpath or @name
1723          if ($i<$mlen-1 && $ms[$i+1][0] == '@' && contains($ms[$i+1], '.') 
1724              && strlen($sp[$i+1]) == 0)
1725          { // if @-@ and second @ is @domain, then link them together
1726						if ($mi == $ms[$i+1]) { // @domain@domain
1727							$wmi = strcat('https://', $wmi);
1728						}
1729						else { // @something@domain
1730							$wmi = strcat('https://', substr($ms[$i+1], 1), '/', $mi);
1731							$mi = strcat($mi, $ms[$i+1]);
1732						}
1733						$ms[$i+1] = ''; // already linked the next link match in this link
1734					}
1735					else if (contains($mi, '.')) { // @domain
1736						$wmi = strcat('https://', $wmi);
1737          }
1738					else { // otherwise Twitter @username
1739						$wmi = strcat('https://twitter.com/', $wmi);
1740						$extra_class = ' h-cassis-username';
1741					}
1742        }
1743				$doing_u_media = false;
1744        $t = strcat($t, '<a class="auto-link', $extra_class, '" href="',
1745                    $wmi, '">', $mi, '</a>', 
1746                    $afterlink);
1747      } else {
1748        $doing_u_media = false;
1749        $t = strcat($t, $mi, $afterlink);
1750      }
1751    } else {
1752			$doing_u_media = false;
1753      $t = strcat($t, $mi);
1754    }
1755  }
1756  return strcat($t, $sp[$mlen]);
1757}
1758
1759// auto_embed: syntactic sugar for calling auto_link to produce embedding markup
1760// required param 1: text to auto link and embed 
1761function auto_embed($t) {
1762  return auto_link($t, true);
1763}
1764
1765
1766function get_auto_linked_urls($s) {
1767  // in: $s result of auto_link() applied to plain text
1768  // out: array of urls from hyperlinks in $s
1769  
1770  $s = explode('href="', $s);
1771  $irtn = count($s);
1772  if ($irtn < 2) { return array(); }
1773  $r = array();
1774  for ($i=1; $i<$irtn; $i++) {
1775    $r[$i-1] = substr($s[$i], 0, offset('"', $s[$i])-1);
1776  }
1777  return $r;
1778}
1779
1780
1781// returns array of URLs after literal "in-reply-to:" in text
1782function get_in_reply_to_urls($s) {
1783  $s = explode('in-reply-to: ', $s);
1784  $irtn = count($s);
1785  if ($irtn < 2) { return array(); }
1786  $r = array();
1787  $re = auto_link_re();
1788  for ($i=1; $i<$irtn; $i++) {
1789    // iterate through all strings after an 'in-reply-to: ' for URLs
1790    $ms = preg_matches($re, $s[$i]);
1791    $msn = count($ms);
1792    if ($ms) {
1793      $sp = preg_split($re, $s[$i]);
1794      $j = 0;
1795      $afterlink = '';
1796      while ($j<$msn && 
1797             $afterlink == '' &&
1798             ($sp[$j] == '' || ctype_space($sp[$j]))) {
1799        // iterate through space separated URLs and add them to $r
1800        $m = $ms[$j];
1801        if ($m[0] != '@') { // skip @-references
1802          $ac = substr($m, -1, 1);
1803          while (contains('.!?,:;"\')]}', $ac) && // trim punc @ end
1804              ($ac != ')' || !contains($m, '('))) { 
1805              // allow one paren pair
1806              // *** not sure twitter is this smart
1807              $afterlink = strcat($ac, $afterlink);
1808              $m = substr($m, 0, -1);
1809              $ac = substr($m, -1, 1);
1810          }
1811          if (substr($m, 0, 6) === 'irc://') { 
1812            // skip it. no known use of in-reply-to an IRC URL
1813          } else {
1814            $r[count($r)] = web_address_to_uri($m, true);
1815          }
1816        }
1817        $j++;
1818      }
1819    }
1820  } 
1821  return $r;
1822}
1823
1824/* Twitter POSSE support */
1825
1826function tw_text_proxy($t) {
1827  // replace URLs with https://j.mp/0011235813 to mimic Twitter's t.co
1828  // $t must be plain text
1829  $re = auto_link_re();
1830  $ms = preg_matches($re, $t);
1831  if (!$ms) {
1832    return $t;
1833  }
1834
1835  $mlen = count($ms);
1836  $sp = preg_split($re, $t);
1837  $t = "";
1838  $sp[0] = string($sp[0]); // force undefined to ""
1839  for ($i=0; $i<$mlen; $i++) {
1840    $mi = $ms[$i];
1841    $spliti = $sp[$i];
1842    $t = strcat($t, $spliti);
1843    $sp[$i+1] = string($sp[$i+1]); // force undefined to ""
1844    if (substr($sp[$i+1], 0, 1)=='/') { // regex omits '/' before </a
1845      $sp[$i+1] = substr($sp[$i+1], 1, strlen($sp[$i+1])-1);
1846      $mi = strcat($mi, '/'); // explicitly include in match
1847    }
1848    if ($i<$mlen-1 && $mi[0] == '@' && strlen($sp[$i+1]) == 0 && $ms[$i+1][0] == '@') {
1849      // found an @-@, combine it into one linkable thing.
1850      $mi = strcat($mi, $ms[$i+1]);
1851      $ms[$i+1] = '';
1852    }
1853    $spe = substr($spliti, -2, 2);
1854    // don't proxy @-names (and before 2018-024: plain ccTLDs)
1855    if ($mi !== '' && $mi[0] !== '@'
1856      //&& (substr($mi, -3, 1) !== '.' || substr_count($mi, '.') > 1)
1857        ) {
1858      $afterlink = '';
1859      $afterchar = substr($mi, -1, 1);
1860      while (contains('.!?,:;"\')]}', $afterchar) && // trim punc @ end
1861          ($afterchar !== ')' || !contains($mi, '('))) { 
1862          // allow one paren pair
1863          // *** not sure twitter is this smart
1864          $afterlink = strcat($afterchar, $afterlink);
1865          $mi = substr($mi, 0, -1);
1866          $afterchar = substr($mi, -1, 1);
1867      }
1868      
1869      $prot = protocol_of_uri($mi);
1870      $proxy_url = '';
1871      if ($prot === 'irc:') {
1872        $proxy_url = $mi; // Twitter doesn't tco irc: URLs
1873      }
1874      else { /* 'https:', 'http:' or presumed for schemeless URLs */ 
1875        $proxy_url = 'https://j.mp/0011235813';
1876      }
1877      $t = strcat($t, $proxy_url, $afterlink);
1878    }
1879    else {
1880      $t = strcat($t, $mi);
1881    }
1882  }
1883  return strcat($t, $sp[$mlen]);
1884}
1885
1886
1887function note_length_check($note, $maxlen, $username) {
1888// checks to see if $note fits in $maxlen characters.
1889// if $username is non-empty, checks to see if a RT'd $note fits in $maxlen
1890// 0 - bad params or other precondition failure error
1891// 200 - exactly fits max characters with RT if username provided
1892// 206 - less than max chars with RT if username provided
1893// 207 - more than RT safe length, but less than tweet max
1894// 208 - tweet max length but with RT would be over
1895// 413 - (entity too large) over max tweet length
1896// strlen('RT @: ') == 6.
1897  if ($maxlen < 1) return 0;
1898  
1899  $note_size_check_u = $username ? 6 + strlen(string($username)) : 0;
1900  $note_size_check_n = strlen(string($note)) + $note_size_check_u;
1901  
1902  if ($note_size_check_n == $maxlen)                      return 200;
1903  if ($note_size_check_n < $maxlen)                       return 206;
1904  if ($note_size_check_n - $note_size_check_u < $maxlen)  return 207;
1905  if ($note_size_check_n - $note_size_check_u == $maxlen) return 208;
1906  return 413;
1907}
1908
1909function tw_length_check($t, $maxlen, $username) {
1910  return note_length_check(tw_text_proxy($t), 
1911                           $maxlen, $username);
1912}
1913
1914function tw_url_to_status_id($u) {
1915// $u - tweet permalink url
1916// returns tweet status id string; 0 if not a tweet permalink.
1917  if (!$u) return 0;
1918  $u = explode("/", string($u)); // https:,,twitter.com,t,status,nnn
1919  if (($u[2] != "twitter.com" && $u[2] != "mobile.twitter.com") || 
1920      $u[4] != "status"      ||
1921      !ctype_digit($u[5])) {
1922    return 0;
1923  }
1924  return $u[5];
1925}
1926
1927function tw_url_to_username($u) {
1928// $u - tweet permalink url
1929// returns twitter username; 0 if not a tweet permalink.
1930  if (!$u) return 0;
1931  $u = explode("/", string($u)); // https:,,twitter.com,t,status,nnn
1932  if ($u[2] != "twitter.com" || 
1933      $u[4] != "status"      ||
1934      !ctype_digit($u[5])) {
1935    return 0;
1936  }
1937  return $u[3];
1938}
1939
1940function fb_url_to_event_id($u) {
1941// $u - fb event permalink url
1942// returns fb event id string; 0 if not a fb event permalink.
1943  if (!$u) return 0;
1944  $u = explode("/", string($u)); // https:,,fb.com,events,nnn
1945  if (($u[2] != "fb.com" && $u[2] != "facebook.com" && 
1946       $u[2] != "www.facebook.com") || 
1947      $u[3] != "events"      ||
1948      !ctype_digit($u[4])) {
1949    return 0;
1950  }
1951  return $u[4];
1952}
1953
1954function is_slash_at_post($u) {
1955// $u - Mastodon or other ActivityPub supporting site post permalink URL
1956// returns whether or not the URL has one of the following syntaxes:
1957// domain/@user@domain/number
1958// domain/@user/number
1959// false positives should be harmless so this function can be shorter/quicker
1960  if (!$u) return false;
1961  $u = explode('/', string($u)); // https:,,domain,@user(@domain),nnn
1962  if (count($u) != 5) return false;
1963  if (substr($u[3], 0, 1) != '@') return false;
1964  if (!ctype_digit($u[4])) return false;
1965  return true;
1966}
1967
1968
1969/* end Falcon */
1970
1971
1972/* ------------------------------------------------------------------ */
1973
1974/* end cassis.js */
1975// ?> -->

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.