2 * Copyright (c) 1989 The Regents of the University of California.
5 * This code is derived from software contributed to Berkeley by
8 * Redistribution and use in source and binary forms, with or without
9 * modification, are permitted provided that the following conditions
11 * 1. Redistributions of source code must retain the above copyright
12 * notice, this list of conditions and the following disclaimer.
13 * 2. Redistributions in binary form must reproduce the above copyright
14 * notice, this list of conditions and the following disclaimer in the
15 * documentation and/or other materials provided with the distribution.
16 * 3. Neither the name of the University nor the names of its contributors
17 * may be used to endorse or promote products derived from this software
18 * without specific prior written permission.
20 * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
21 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22 * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
23 * ARE DISCLAIMED. IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
24 * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25 * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
26 * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
27 * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
28 * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
29 * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
32 #if defined(LIBC_SCCS) && !defined(lint)
33 static char sccsid[] = "@(#)glob.c 5.12 (Berkeley) 6/24/91";
34 #endif /* LIBC_SCCS and not lint */
36 * Glob: the interface is a superset of the one defined in POSIX 1003.2,
39 * The [!...] convention to negate a range is supported (SysV, Posix, ksh).
41 * Optional extra services, controlled by flags not defined by POSIX:
44 * Escaping convention: \ inhibits any special meaning the following
45 * character might have (except \ at end of string is retained).
47 * Set in gl_flags if pattern contained a globbing character.
49 * Use ^ instead of ! for "not".
51 * Number of matches in the current invocation of glob.
55 #pragma warning(disable:4244)
56 #endif /* WINNT_NATIVE */
64 #define mblen(_s,_n) mbrlen((_s),(_n),NULL)
75 #define S_ISDIR(a) (((a) & S_IFMT) == S_IFDIR)
78 #if !defined(S_ISLNK) && defined(S_IFLNK)
79 #define S_ISLNK(a) (((a) & S_IFMT) == S_IFLNK)
82 #if !defined(S_ISLNK) && !defined(lstat)
86 typedef unsigned short Char;
88 static int glob1 (Char *, glob_t *, int);
89 static int glob2 (struct strbuf *, const Char *, glob_t *, int);
90 static int glob3 (struct strbuf *, const Char *, const Char *,
91 const Char *, glob_t *, int);
92 static void globextend (const char *, glob_t *);
93 static int match (const char *, const Char *, const Char *,
95 static int compare (const void *, const void *);
96 static DIR *Opendir (const char *);
98 static int Lstat (const char *, struct stat *);
100 static int Stat (const char *, struct stat *sb);
101 static Char *Strchr (Char *, int);
103 static void qprintf (const char *, const Char *);
119 #define UNDERSCORE '_'
121 #define M_META 0x8000
122 #define M_PROTECT 0x4000
123 #define M_MASK 0xffff
124 #define M_ASCII 0x00ff
126 #define LCHAR(c) ((c)&M_ASCII)
127 #define META(c) ((c)|M_META)
128 #define M_ALL META('*')
129 #define M_END META(']')
130 #define M_NOT META('!')
131 #define M_ALTNOT META('^')
132 #define M_ONE META('?')
133 #define M_RNG META('-')
134 #define M_SET META('[')
135 #define ismeta(c) (((c)&M_META) != 0)
138 globcharcoll(__Char c1, __Char c2, int cs)
140 #if defined(NLS) && defined(LC_COLLATE) && defined(HAVE_STRCOLL)
141 # if defined(WIDE_STRINGS)
142 wchar_t s1[2], s2[2];
151 /* This should not be here, but I'll rather leave it in than engage in
152 a LC_COLLATE flamewar about a shell I don't use... */
153 if (iswlower(c1) && iswupper(c2))
155 if (iswupper(c1) && iswlower(c2))
161 s1[1] = s2[1] = '\0';
162 return wcscoll(s1, s2);
163 # else /* not WIDE_STRINGS */
169 * From kevin lyda <kevin@suberic.net>:
170 * strcoll does not guarantee case sorting, so we pre-process now:
173 c1 = islower(c1) ? c1 : tolower(c1);
174 c2 = islower(c2) ? c2 : tolower(c2);
176 if (islower(c1) && isupper(c2))
178 if (isupper(c1) && islower(c2))
183 s1[1] = s2[1] = '\0';
184 return strcoll(s1, s2);
192 * Need to dodge two kernel bugs:
193 * opendir("") != opendir(".")
194 * NAMEI_BUG: on plain files trailing slashes are ignored in some kernels.
195 * POSIX specifies that they should be ignored in directories.
199 Opendir(const char *str)
201 #if defined(hpux) || defined(__hpux)
206 return (opendir("."));
207 #if defined(hpux) || defined(__hpux)
209 * Opendir on some device files hangs, so avoid it
211 if (stat(str, &st) == -1 || !S_ISDIR(st.st_mode))
219 Lstat(const char *fn, struct stat *sb)
225 if (*fn != 0 && strend(fn)[-1] == '/' && !S_ISDIR(sb->st_mode))
227 # endif /* NAMEI_BUG */
235 Stat(const char *fn, struct stat *sb)
241 if (*fn != 0 && strend(fn)[-1] == '/' && !S_ISDIR(sb->st_mode))
243 #endif /* NAMEI_BUG */
248 Strchr(Char *str, int ch)
259 qprintf(const char *pre, const Char *s)
265 xprintf("%c", *p & 0xff);
266 xprintf("\n%s", pre);
268 xprintf("%c", *p & M_PROTECT ? '"' : ' ');
269 xprintf("\n%s", pre);
271 xprintf("%c", *p & M_META ? '_' : ' ');
277 compare(const void *p, const void *q)
279 #if defined(NLS) && defined(HAVE_STRCOLL)
280 return (strcoll(*(char *const *) p, *(char *const *) q));
282 return (strcmp(*(char *const *) p, *(char *const *) q));
283 #endif /* NLS && HAVE_STRCOLL */
287 * The main glob() routine: compiles the pattern (optionally processing
288 * quotes), calls glob1() to do the real pattern matching, and finally
289 * sorts the list (unless unsorted operation is requested). Returns 0
290 * if things went well, nonzero if errors occurred. It is not an error
291 * to find no matches.
294 glob(const char *pattern, int flags, int (*errfunc) (const char *, int),
298 Char *bufnext, m_not;
299 const unsigned char *patnext;
301 Char *qpatnext, *patbuf;
304 patnext = (const unsigned char *) pattern;
305 if (!(flags & GLOB_APPEND)) {
307 pglob->gl_pathv = NULL;
308 if (!(flags & GLOB_DOOFFS))
311 pglob->gl_flags = flags & ~GLOB_MAGCHAR;
312 pglob->gl_errfunc = errfunc;
313 oldpathc = pglob->gl_pathc;
314 pglob->gl_matchc = 0;
316 if (pglob->gl_flags & GLOB_ALTNOT) {
325 patbuf = xmalloc((strlen(pattern) + 1) * sizeof(*patbuf));
328 no_match = *patnext == not;
332 if (flags & GLOB_QUOTE) {
333 /* Protect the quoted characters */
334 while ((c = *patnext++) != EOS) {
338 len = mblen((const char *)(patnext - 1), MB_LEN_MAX);
340 TCSH_IGNORE(mblen(NULL, 0));
342 *bufnext++ = (Char) c;
344 *bufnext++ = (Char) (*patnext++ | M_PROTECT);
346 #endif /* WIDE_STRINGS */
348 if ((c = *patnext++) == EOS) {
352 *bufnext++ = (Char) (c | M_PROTECT);
355 *bufnext++ = (Char) c;
359 while ((c = *patnext++) != EOS)
360 *bufnext++ = (Char) c;
365 while ((c = *qpatnext++) != EOS) {
371 if (*qpatnext == EOS ||
372 Strchr(qpatnext + 1, RBRACKET) == NULL) {
373 *bufnext++ = LBRACKET;
378 pglob->gl_flags |= GLOB_MAGCHAR;
384 *bufnext++ = LCHAR(c);
385 if (*qpatnext == RANGE &&
386 (c = qpatnext[1]) != RBRACKET) {
388 *bufnext++ = LCHAR(c);
391 } while ((c = *qpatnext++) != RBRACKET);
395 pglob->gl_flags |= GLOB_MAGCHAR;
399 pglob->gl_flags |= GLOB_MAGCHAR;
400 /* collapse adjacent stars to one [or three if globstar],
401 * to avoid exponential behavior
403 if (bufnext == patbuf || bufnext[-1] != M_ALL ||
404 ((flags & GLOB_STAR) != 0 &&
405 (bufnext - 1 == patbuf || bufnext[-2] != M_ALL ||
406 bufnext - 2 == patbuf || bufnext[-3] != M_ALL)))
410 *bufnext++ = LCHAR(c);
416 qprintf("patbuf=", patbuf);
419 if ((err = glob1(patbuf, pglob, no_match)) != 0) {
425 * If there was no match we are going to append the pattern
426 * if GLOB_NOCHECK was specified or if GLOB_NOMAGIC was specified
427 * and the pattern did not contain any magic characters
428 * GLOB_NOMAGIC is there just for compatibility with csh.
430 if (pglob->gl_pathc == oldpathc &&
431 ((flags & GLOB_NOCHECK) ||
432 ((flags & GLOB_NOMAGIC) && !(pglob->gl_flags & GLOB_MAGCHAR)))) {
433 if (!(flags & GLOB_QUOTE))
434 globextend(pattern, pglob);
439 /* copy pattern, interpreting quotes */
440 copy = xmalloc(strlen(pattern) + 1);
443 while (*src != EOS) {
444 /* Don't interpret quotes. The spec does not say we should do */
452 globextend(copy, pglob);
458 else if (!(flags & GLOB_NOSORT) && (pglob->gl_pathc != oldpathc))
459 qsort(pglob->gl_pathv + pglob->gl_offs + oldpathc,
460 pglob->gl_pathc - oldpathc, sizeof(char *), compare);
466 glob1(Char *pattern, glob_t *pglob, int no_match)
468 struct strbuf pathbuf = strbuf_INIT;
472 * a null pathname is invalid -- POSIX 1003.1 sect. 2.4.
476 err = glob2(&pathbuf, pattern, pglob, no_match);
482 * functions glob2 and glob3 are mutually recursive; there is one level
483 * of recursion for each segment in the pattern that contains one or
484 * more meta characters.
487 glob2(struct strbuf *pathbuf, const Char *pattern, glob_t *pglob, int no_match)
495 * loop over pattern segments until end of pattern or until segment with
496 * meta character found.
500 if (*pattern == EOS) { /* end of pattern? */
501 strbuf_terminate(pathbuf);
503 if (Lstat(pathbuf->s, &sbuf))
506 if (((pglob->gl_flags & GLOB_MARK) &&
507 pathbuf->s[pathbuf->len - 1] != SEP) &&
508 (S_ISDIR(sbuf.st_mode)
510 || (S_ISLNK(sbuf.st_mode) &&
511 (Stat(pathbuf->s, &sbuf) == 0) &&
512 S_ISDIR(sbuf.st_mode))
515 strbuf_append1(pathbuf, SEP);
516 strbuf_terminate(pathbuf);
519 globextend(pathbuf->s, pglob);
523 /* find end of next segment, tentatively copy to pathbuf */
525 orig_len = pathbuf->len;
526 while (*p != EOS && *p != SEP) {
529 strbuf_append1(pathbuf, *p++);
532 if (!anymeta) { /* no expansion, do next segment */
534 while (*pattern == SEP)
535 strbuf_append1(pathbuf, *pattern++);
537 else { /* need expansion, recurse */
538 pathbuf->len = orig_len;
539 return (glob3(pathbuf, pattern, p, pattern, pglob, no_match));
546 One_Char_mbtowc(__Char *pwc, const Char *s, size_t n)
549 char buf[MB_LEN_MAX], *p;
554 while (p < buf + n && (*p++ = LCHAR(*s++)) != 0)
556 return one_mbtowc(pwc, buf, n);
564 glob3(struct strbuf *pathbuf, const Char *pattern, const Char *restpattern,
565 const Char *pglobstar, glob_t *pglob, int no_match)
571 Char m_not = (pglob->gl_flags & GLOB_ALTNOT) ? M_ALTNOT : M_NOT;
574 int chase_symlinks = 0;
575 const Char *termstar = NULL;
577 strbuf_terminate(pathbuf);
578 orig_len = pathbuf->len;
581 while (pglobstar < restpattern) {
583 size_t width = One_Char_mbtowc(&wc, pglobstar, MB_LEN_MAX);
584 if ((pglobstar[0] & M_MASK) == M_ALL &&
585 (pglobstar[width] & M_MASK) == M_ALL) {
587 chase_symlinks = (pglobstar[2 * width] & M_MASK) == M_ALL;
588 termstar = pglobstar + (2 + chase_symlinks) * width;
595 err = pglobstar==pattern && termstar==restpattern ?
596 *restpattern == EOS ?
597 glob2(pathbuf, restpattern - 1, pglob, no_match) :
598 glob2(pathbuf, restpattern + 1, pglob, no_match) :
599 glob3(pathbuf, pattern, restpattern, termstar, pglob, no_match);
602 pathbuf->len = orig_len;
603 strbuf_terminate(pathbuf);
606 if (*pathbuf->s && (Lstat(pathbuf->s, &sbuf) || !S_ISDIR(sbuf.st_mode)
608 && ((globstar && !chase_symlinks) || !S_ISLNK(sbuf.st_mode))
613 if (!(dirp = Opendir(pathbuf->s))) {
614 /* todo: don't call for ENOENT or ENOTDIR? */
615 if ((pglob->gl_errfunc && (*pglob->gl_errfunc) (pathbuf->s, errno)) ||
616 (pglob->gl_flags & GLOB_ERR))
622 /* search directory for matching names */
623 while ((dp = readdir(dirp)) != NULL) {
624 /* initial DOT must be matched literally */
625 if (dp->d_name[0] == DOT && *pattern != DOT)
626 if (!(pglob->gl_flags & GLOB_DOT) || !dp->d_name[1] ||
627 (dp->d_name[1] == DOT && !dp->d_name[2]))
628 continue; /*unless globdot and not . or .. */
629 pathbuf->len = orig_len;
630 strbuf_append(pathbuf, dp->d_name);
631 strbuf_terminate(pathbuf);
635 if (!chase_symlinks &&
636 (Lstat(pathbuf->s, &sbuf) || S_ISLNK(sbuf.st_mode)))
639 if (match(pathbuf->s + orig_len, pattern, termstar,
640 (int)m_not) == no_match)
642 strbuf_append1(pathbuf, SEP);
643 strbuf_terminate(pathbuf);
644 if ((err = glob2(pathbuf, pglobstar, pglob, no_match)) != 0)
647 if (match(pathbuf->s + orig_len, pattern, restpattern,
648 (int) m_not) == no_match)
650 if ((err = glob2(pathbuf, restpattern, pglob, no_match)) != 0)
654 /* todo: check error from readdir? */
661 * Extend the gl_pathv member of a glob_t structure to accomodate a new item,
662 * add the new item, and update gl_pathc.
664 * This assumes the BSD realloc, which only copies the block when its size
665 * crosses a power-of-two boundary; for v7 realloc, this would cause quadratic
668 * Return 0 if new item added, error code if memory couldn't be allocated.
670 * Invariant of the glob_t structure:
671 * Either gl_pathc is zero and gl_pathv is NULL; or gl_pathc > 0 and
672 * gl_pathv points to (gl_offs + gl_pathc + 1) items.
675 globextend(const char *path, glob_t *pglob)
681 newsize = sizeof(*pathv) * (2 + pglob->gl_pathc + pglob->gl_offs);
682 pathv = xrealloc(pglob->gl_pathv, newsize);
684 if (pglob->gl_pathv == NULL && pglob->gl_offs > 0) {
685 /* first time around -- clear initial gl_offs items */
686 pathv += pglob->gl_offs;
687 for (i = pglob->gl_offs; --i >= 0;)
690 pglob->gl_pathv = pathv;
692 pathv[pglob->gl_offs + pglob->gl_pathc++] = strsave(path);
693 pathv[pglob->gl_offs + pglob->gl_pathc] = NULL;
697 * pattern matching function for filenames.
700 match(const char *name, const Char *pat, const Char *patend, int m_not)
702 int ok, negate_range;
704 const char *nameNext, *nameStart, *nameEnd;
708 nameStart = nameNext = name;
711 while (pat < patend || *name) {
715 c = *pat; /* Only for M_MASK bits */
719 pwk = One_Char_mbtowc(&wc, pat, MB_LEN_MAX);
720 lwk = one_mbtowc(&wk, name, MB_LEN_MAX);
721 switch (c & M_MASK) {
723 while ((*(pat + pwk) & M_MASK) == M_ALL) {
725 pwk = One_Char_mbtowc(&wc, pat, MB_LEN_MAX);
728 nameNext = name + lwk;
742 pwk = One_Char_mbtowc(&wc, pat, MB_LEN_MAX);
744 if ((negate_range = ((*pat & M_MASK) == m_not)) != 0) {
746 pwk = One_Char_mbtowc(&wc, pat, MB_LEN_MAX);
749 while ((*pat & M_MASK) != M_END) {
750 if ((*pat & M_MASK) == M_RNG) {
754 pwk = One_Char_mbtowc(&wc2, pat, MB_LEN_MAX);
755 if (globcharcoll(wc1, wk, 0) <= 0 &&
756 globcharcoll(wk, wc2, 0) <= 0)
762 pwk = One_Char_mbtowc(&wc, pat, MB_LEN_MAX);
765 pwk = One_Char_mbtowc(&wc, pat, MB_LEN_MAX);
766 if (ok == negate_range)
770 if (*name == EOS || samecase(wk) != samecase(wc))
776 if (nameNext != nameStart
777 && (nameEnd == NULL || nameNext <= nameEnd)) {
787 /* free allocated data belonging to a glob_t structure */
789 globfree(glob_t *pglob)
794 if (pglob->gl_pathv != NULL) {
795 pp = pglob->gl_pathv + pglob->gl_offs;
796 for (i = pglob->gl_pathc; i--; ++pp)
798 xfree(*pp), *pp = NULL;
799 xfree(pglob->gl_pathv), pglob->gl_pathv = NULL;