Can I use a mask to iterate files in a directory with Boost?

I want to iterate over all files in a directory matching something like somefiles*.txt.

Does boost::filesystem have something built in to do that, or do I need a regex or something against each leaf()?


Solution 1:

EDIT: As noted in the comments, the code below is valid for versions of boost::filesystem prior to v3. For v3, refer to the suggestions in the comments.


boost::filesystem does not have wildcard search, you have to filter files yourself.

This is a code sample extracting the content of a directory with a boost::filesystem's directory_iterator and filtering it with boost::regex:

const std::string target_path( "/my/directory/" );
const boost::regex my_filter( "somefiles.*\.txt" );

std::vector< std::string > all_matching_files;

boost::filesystem::directory_iterator end_itr; // Default ctor yields past-the-end
for( boost::filesystem::directory_iterator i( target_path ); i != end_itr; ++i )
{
    // Skip if not a file
    if( !boost::filesystem::is_regular_file( i->status() ) ) continue;

    boost::smatch what;

    // Skip if no match for V2:
    if( !boost::regex_match( i->leaf(), what, my_filter ) ) continue;
    // For V3:
    //if( !boost::regex_match( i->path().filename().string(), what, my_filter ) ) continue;

    // File matches, store it
    all_matching_files.push_back( i->leaf() );
}

(If you are looking for a ready-to-use class with builtin directory filtering, have a look at Qt's QDir.)

Solution 2:

There is a Boost Range Adaptors way:

#define BOOST_RANGE_ENABLE_CONCEPT_ASSERT 0
#include <boost/filesystem.hpp>
#include <boost/range/adaptors.hpp>

namespace bfs = boost::filesystem;
namespace ba = boost::adaptors;

const std::string target_path( "/my/directory/" );
const boost::regex my_filter( "somefiles.*\.txt" );
boost::smatch what;

for (auto &entry: boost::make_iterator_range(bfs::directory_iterator(target_path), {})
| ba::filtered(static_cast<bool (*)(const bfs::path &)>(&bfs::is_regular_file))
| ba::filtered([&](const bfs::path &path){ return boost::regex_match(path.filename().string(), what, my_filter); })
)
{
  // There are only files matching defined pattern "somefiles*.txt".
  std::cout << entry.path().filename() << std::endl;
}

Solution 3:

My solution is essentially the same as Julien-L, but encapsulated in the include file it is nicer to use. Implemented using boost::filesystem v3. I guess that something like that is not included in the boost::filesystem directly because it would introduce dependency on boost::regex.

#include "FilteredDirectoryIterator.h"
std::vector< std::string > all_matching_files;
std::for_each(
        FilteredDirectoryIterator("/my/directory","somefiles.*\.txt"),
        FilteredDirectoryIterator(),
        [&all_matching_files](const FilteredDirectoryIterator::value_type &dirEntry){
                all_matching_files.push_back(dirEntry.path());
            }
        );

alternatively use FilteredRecursiveDirectoryIterator for recursive sub directories search:

#include "FilteredDirectoryIterator.h"
std::vector< std::string > all_matching_files;
std::for_each(
        FilteredRecursiveDirectoryIterator("/my/directory","somefiles.*\.txt"),
        FilteredRecursiveDirectoryIterator(),
        [&all_matching_files](const FilteredRecursiveDirectoryIterator::value_type &dirEntry){
                all_matching_files.push_back(dirEntry.path());
            }
        );

FilteredDirectoryIterator.h

#ifndef TOOLS_BOOST_FILESYSTEM_FILTEREDDIRECTORYITERATOR_H_
#define TOOLS_BOOST_FILESYSTEM_FILTEREDDIRECTORYITERATOR_H_

#include "boost/filesystem.hpp"
#include "boost/regex.hpp"
#include <functional>

template <class NonFilteredIterator = boost::filesystem::directory_iterator>
class FilteredDirectoryIteratorTmpl
:   public std::iterator<
    std::input_iterator_tag, typename NonFilteredIterator::value_type
    >
{
private:
    typedef std::string string;
    typedef boost::filesystem::path path;
    typedef
        std::function<
            bool(const typename NonFilteredIterator::value_type &dirEntry)
            >
        FilterFunction;

    NonFilteredIterator it;

    NonFilteredIterator end;

    const FilterFunction filter;

public:

    FilteredDirectoryIteratorTmpl();

    FilteredDirectoryIteratorTmpl(
        const path &iteratedDir, const string &regexMask
        );

    FilteredDirectoryIteratorTmpl(
        const path &iteratedDir, const boost::regex &mask
        );

    FilteredDirectoryIteratorTmpl(
        const path &iteratedDir,
        const FilterFunction &filter
        );

    //preincrement
    FilteredDirectoryIteratorTmpl<NonFilteredIterator>& operator++() {
        for(++it;it!=end && !filter(*it);++it);
        return *this;
    };

    //postincrement
    FilteredDirectoryIteratorTmpl<NonFilteredIterator> operator++(int) {
        for(++it;it!=end && !filter(*it);++it);
        return FilteredDirectoryIteratorTmpl<NonFilteredIterator>(it,filter);
    };
    const boost::filesystem::directory_entry &operator*() {return *it;};
    bool operator!=(const FilteredDirectoryIteratorTmpl<NonFilteredIterator>& other)
    {
        return it!=other.it;
    };
    bool operator==(const FilteredDirectoryIteratorTmpl<NonFilteredIterator>& other)
    {
        return it==other.it;
    };
};

typedef
    FilteredDirectoryIteratorTmpl<boost::filesystem::directory_iterator>
    FilteredDirectoryIterator;

typedef
    FilteredDirectoryIteratorTmpl<boost::filesystem::recursive_directory_iterator>
    FilteredRecursiveDirectoryIterator;

template <class NonFilteredIterator>
FilteredDirectoryIteratorTmpl<NonFilteredIterator>::FilteredDirectoryIteratorTmpl()
:   it(),
    filter(
        [](const boost::filesystem::directory_entry& /*dirEntry*/){return true;}
        )
{

}

template <class NonFilteredIterator>
FilteredDirectoryIteratorTmpl<NonFilteredIterator>::FilteredDirectoryIteratorTmpl(
    const path &iteratedDir,const string &regexMask
    )
:   FilteredDirectoryIteratorTmpl(iteratedDir, boost::regex(regexMask))
{
}

template <class NonFilteredIterator>
FilteredDirectoryIteratorTmpl<NonFilteredIterator>::FilteredDirectoryIteratorTmpl(
    const path &iteratedDir,const boost::regex &regexMask
    )
:   it(NonFilteredIterator(iteratedDir)),
    filter(
        [regexMask](const boost::filesystem::directory_entry& dirEntry){
            using std::endl;
            // return false to skip dirEntry if no match
            const string filename = dirEntry.path().filename().native();
            return boost::regex_match(filename, regexMask);
        }
        )
{
    if (it!=end && !filter(*it)) ++(*this);
}

template <class NonFilteredIterator>
FilteredDirectoryIteratorTmpl<NonFilteredIterator>::FilteredDirectoryIteratorTmpl(
    const path &iteratedDir, const FilterFunction &filter
    )
:   it(NonFilteredIterator(iteratedDir)),
    filter(filter)
{
    if (it!=end && !filter(*it)) ++(*this);
}

#endif

Solution 4:

I believe the directory_iterators will only provide all files in a directory. It up to you to filter them as necessary.