Merge pull request #527 from Ana06/v1-6-2

changelog: v1.6.2
This release backports a fix to capa 1.6: The Windows binary was built with Python 3.9 which doesn't support Windows 7.
2025-12-07 05:10:36 -08:00 · 2021-04-13 17:21:31 +02:00 · 2021-04-13 12:18:50 +02:00 · 2021-04-13 12:16:19 +02:00 · 2021-04-13 12:08:30 +02:00 · 2021-04-13 12:06:05 +02:00
57 changed files with 4118 additions and 2298 deletions
--- a/.gitattributes
+++ b/.gitattributes
@@ -0,0 +1,9 @@
+# Set the default behavior, in case people don't have core.autocrlf set.
+* text=auto
+
+# Explicitly declare text files you want to always be normalized and converted
+# to native line endings on checkout.
+*.py text
+*.yml text
+*.md text
+*.txt text
--- a/.github/CODE_OF_CONDUCT.md
+++ b/.github/CODE_OF_CONDUCT.md
@@ -1,46 +1,46 @@
-# Contributor Covenant Code of Conduct
-
-## Our Pledge
-
-In the interest of fostering an open and welcoming environment, we as contributors and maintainers pledge to making participation in our project and our community a harassment-free experience for everyone, regardless of age, body size, disability, ethnicity, gender identity and expression, level of experience, nationality, personal appearance, race, religion, or sexual identity and orientation.
-
-## Our Standards
-
-Examples of behavior that contributes to creating a positive environment include:
-
-* Using welcoming and inclusive language
-* Being respectful of differing viewpoints and experiences
-* Gracefully accepting constructive criticism
-* Focusing on what is best for the community
-* Showing empathy towards other community members
-
-Examples of unacceptable behavior by participants include:
-
-* The use of sexualized language or imagery and unwelcome sexual attention or advances
-* Trolling, insulting/derogatory comments, and personal or political attacks
-* Public or private harassment
-* Publishing others' private information, such as a physical or electronic address, without explicit permission
-* Other conduct which could reasonably be considered inappropriate in a professional setting
-
-## Our Responsibilities
-
-Project maintainers are responsible for clarifying the standards of acceptable behavior and are expected to take appropriate and fair corrective action in response to any instances of unacceptable behavior.
-
-Project maintainers have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct, or to ban temporarily or permanently any contributor for other behaviors that they deem inappropriate, threatening, offensive, or harmful.
-
-## Scope
-
-This Code of Conduct applies both within project spaces and in public spaces when an individual is representing the project or its community. Examples of representing a project or community include using an official project e-mail address, posting via an official social media account, or acting as an appointed representative at an online or offline event. Representation of a project may be further defined and clarified by project maintainers.
-
-## Enforcement
-
-Instances of abusive, harassing, or otherwise unacceptable behavior may be reported by contacting the project team. All complaints will be reviewed and investigated and will result in a response that is deemed necessary and appropriate to the circumstances. The project team is obligated to maintain confidentiality with regard to the reporter of an incident. Further details of specific enforcement policies may be posted separately.
-
-Project maintainers who do not follow or enforce the Code of Conduct in good faith may face temporary or permanent repercussions as determined by other members of the project's leadership.
-
-## Attribution
-
-This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4, available at [https://contributor-covenant.org/version/1/4][version]
-
-[homepage]: https://contributor-covenant.org
-[version]: https://contributor-covenant.org/version/1/4/
+# Contributor Covenant Code of Conduct
+
+## Our Pledge
+
+In the interest of fostering an open and welcoming environment, we as contributors and maintainers pledge to making participation in our project and our community a harassment-free experience for everyone, regardless of age, body size, disability, ethnicity, gender identity and expression, level of experience, nationality, personal appearance, race, religion, or sexual identity and orientation.
+
+## Our Standards
+
+Examples of behavior that contributes to creating a positive environment include:
+
+* Using welcoming and inclusive language
+* Being respectful of differing viewpoints and experiences
+* Gracefully accepting constructive criticism
+* Focusing on what is best for the community
+* Showing empathy towards other community members
+
+Examples of unacceptable behavior by participants include:
+
+* The use of sexualized language or imagery and unwelcome sexual attention or advances
+* Trolling, insulting/derogatory comments, and personal or political attacks
+* Public or private harassment
+* Publishing others' private information, such as a physical or electronic address, without explicit permission
+* Other conduct which could reasonably be considered inappropriate in a professional setting
+
+## Our Responsibilities
+
+Project maintainers are responsible for clarifying the standards of acceptable behavior and are expected to take appropriate and fair corrective action in response to any instances of unacceptable behavior.
+
+Project maintainers have the right and responsibility to remove, edit, or reject comments, commits, code, wiki edits, issues, and other contributions that are not aligned to this Code of Conduct, or to ban temporarily or permanently any contributor for other behaviors that they deem inappropriate, threatening, offensive, or harmful.
+
+## Scope
+
+This Code of Conduct applies both within project spaces and in public spaces when an individual is representing the project or its community. Examples of representing a project or community include using an official project e-mail address, posting via an official social media account, or acting as an appointed representative at an online or offline event. Representation of a project may be further defined and clarified by project maintainers.
+
+## Enforcement
+
+Instances of abusive, harassing, or otherwise unacceptable behavior may be reported by contacting the project team. All complaints will be reviewed and investigated and will result in a response that is deemed necessary and appropriate to the circumstances. The project team is obligated to maintain confidentiality with regard to the reporter of an incident. Further details of specific enforcement policies may be posted separately.
+
+Project maintainers who do not follow or enforce the Code of Conduct in good faith may face temporary or permanent repercussions as determined by other members of the project's leadership.
+
+## Attribution
+
+This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4, available at [https://contributor-covenant.org/version/1/4][version]
+
+[homepage]: https://contributor-covenant.org
+[version]: https://contributor-covenant.org/version/1/4/
--- a/.github/CONTRIBUTING.md
+++ b/.github/CONTRIBUTING.md
@@ -1,197 +1,197 @@
-# Contributing to Capa
-
-First off, thanks for taking the time to contribute!
-
-The following is a set of guidelines for contributing to capa and its packages, which are hosted in the [FireEye Organization](https://github.com/fireeye) on GitHub. These are mostly guidelines, not rules. Use your best judgment, and feel free to propose changes to this document in a pull request.
-
-#### Table Of Contents
-
-[Code of Conduct](#code-of-conduct)
-
-[What should I know before I get started?](#what-should-i-know-before-i-get-started)
-  * [Capa and its Repositories](#capa-and-its-repositories)
-  * [Capa Design Decisions](#design-decisions)
-
-[How Can I Contribute?](#how-can-i-contribute)
-  * [Reporting Bugs](#reporting-bugs)
-  * [Suggesting Enhancements](#suggesting-enhancements)
-  * [Your First Code Contribution](#your-first-code-contribution)
-  * [Pull Requests](#pull-requests)
-
-[Styleguides](#styleguides)
-  * [Git Commit Messages](#git-commit-messages)
-  * [Python Styleguide](#python-styleguide)
-  * [Rules Styleguide](#rules-styleguide)
-
-## Code of Conduct
-
-This project and everyone participating in it is governed by the [Capa Code of Conduct](CODE_OF_CONDUCT.md). By participating, you are expected to uphold this code. Please report unacceptable behavior to the maintainers.
-
-## What should I know before I get started?
-
-### Capa and its repositories
-
-We host the capa project as three Github repositories:
-  - [capa](https://github.com/fireeye/capa)
-  - [capa-rules](https://github.com/fireeye/capa-rules)
-  - [capa-testfiles](https://github.com/fireeye/capa-testfiles)
-  
-The command line tools, logic engine, and other Python source code are found in the `capa` repository.
-This is the repository to fork when you want to enhance the features, performance, or user interface of capa.
-Do *not* push rules directly to this repository, instead...
-
-The standard rules contributed by the community are found in the `capa-rules` repository.
-When you have an idea for a new rule, you should open a PR against `capa-rules`.
-We keep `capa` and `capa-rules` separate to distinguish where ideas, bugs, and discussions should happen.
-If you're writing yaml it probably goes in `capa-rules` and if you're writing Python it probably goes in `capa`.
-Also, we encourage users to develop their own rule repositories, so we treat our default set of rules in the same way.
-
-Test fixtures, such as malware samples and analysis workspaces, are found in the `capa-testfiles` repository.
-These are files you'll need in order to run the linter (in `--thorough` mode) and full test suites;
- however, they take up a lot of space (1GB+), so by keeping `capa-testfiles` separate,
- a shallow checkout of `capa` and `capa-rules` doesn't take much bandwidth.
-
-### Design Decisions
-
-When we make a significant decision in how we maintain the project and what we can or cannot support,
- we will document it in the [capa issues tracker](https://github.com/fireeye/capa/issues).
-This is the best place review our discussions about what/how/why we do things in the project.
-If you have a question, check to see if it is documented there.
-If it is *not* documented there, or you can't find an answer, please open a issue.
-We'll link to existing issues when appropriate to keep discussions in one place.
-
-## How Can I Contribute?
-
-### Reporting Bugs
-
-This section guides you through submitting a bug report for capa.
-Following these guidelines helps maintainers and the community understand your report, reproduce the behavior, and find related reports.
-
-Before creating bug reports, please check [this list](#before-submitting-a-bug-report)
- as you might find out that you don't need to create one.
-When you are creating a bug report, please [include as many details as possible](#how-do-i-submit-a-good-bug-report).
-Fill out [the required template](./ISSUE_TEMPLATE/bug_report.md),
- the information it asks for helps us resolve issues faster.
-
-> **Note:** If you find a **Closed** issue that seems like it is the same thing that you're experiencing, open a new issue and include a link to the original issue in the body of your new one.
-
-#### Before Submitting A Bug Report
-
-* **Determine [which repository the problem should be reported in](#capa-and-its-repositories)**.
-* **Perform a [cursory search](https://github.com/fireeye/capa/issues?q=is%3Aissue)** to see if the problem has already been reported. If it has **and the issue is still open**, add a comment to the existing issue instead of opening a new one.
-
-#### How Do I Submit A (Good) Bug Report?
-
-Bugs are tracked as [GitHub issues](https://guides.github.com/features/issues/).
-After you've determined [which repository](#capa-and-its-repositories) your bug is related to,
- create an issue on that repository and provide the following information by filling in 
- [the template](./ISSUE_TEMPLATE/bug_report.md).
-
-Explain the problem and include additional details to help maintainers reproduce the problem:
-
-* **Use a clear and descriptive title** for the issue to identify the problem.
-* **Describe the exact steps which reproduce the problem** in as many details as possible. For example, start by explaining how you started capa, e.g. which command exactly you used in the terminal, or how you started capa otherwise.
-* **Provide specific examples to demonstrate the steps**. Include links to files or GitHub projects, or copy/pasteable snippets, which you use in those examples. If you're providing snippets in the issue, use [Markdown code blocks](https://help.github.com/articles/markdown-basics/#multiple-lines).
-* **Describe the behavior you observed after following the steps** and point out what exactly is the problem with that behavior.
-* **Explain which behavior you expected to see instead and why.**
-* **Include screenshots and animated GIFs** which show you following the described steps and clearly demonstrate the problem. You can use [this tool](https://www.cockos.com/licecap/) to record GIFs on macOS and Windows, and [this tool](https://github.com/colinkeenan/silentcast) or [this tool](https://github.com/GNOME/byzanz) on Linux.
-* **If you're reporting that capa crashed**, include the stack trace from the terminal. Include the stack trace in the issue in a [code block](https://help.github.com/articles/markdown-basics/#multiple-lines), a [file attachment](https://help.github.com/articles/file-attachments-on-issues-and-pull-requests/), or put it in a [gist](https://gist.github.com/) and provide link to that gist.
-* **If the problem wasn't triggered by a specific action**, describe what you were doing before the problem happened and share more information using the guidelines below.
-
-Provide more context by answering these questions:
-
-* **Did the problem start happening recently** (e.g. after updating to a new version of capa) or was this always a problem?
-* If the problem started happening recently, **can you reproduce the problem in an older version of capa?** What's the most recent version in which the problem doesn't happen? You can download older versions of capa from [the releases page](https://github.com/fireeye/capa/releases).
-* **Can you reliably reproduce the issue?** If not, provide details about how often the problem happens and under which conditions it normally happens.
-* If the problem is related to working with files (e.g. opening and editing files), **does the problem happen for all files and projects or only some?** Does the problem happen only when working with local or remote files (e.g. on network drives), with files of a specific type (e.g. only JavaScript or Python files), with large files or files with very long lines, or with files in a specific encoding? Is there anything else special about the files you are using?
-
-Include details about your configuration and environment:
-
-* **Which version of capa are you using?** You can get the exact version by running `capa --version` in your terminal.
-* **What's the name and version of the OS you're using**?
-
-### Suggesting Enhancements
-
-This section guides you through submitting an enhancement suggestion for capa, including completely new features and minor improvements to existing functionality. Following these guidelines helps maintainers and the community understand your suggestion and find related suggestions.
-
-Before creating enhancement suggestions, please check [this list](#before-submitting-an-enhancement-suggestion) as you might find out that you don't need to create one. When you are creating an enhancement suggestion, please [include as many details as possible](#how-do-i-submit-a-good-enhancement-suggestion). Fill in [the template](./ISSUE_TEMPLATE/feature_request.md), including the steps that you imagine you would take if the feature you're requesting existed.
-
-#### Before Submitting An Enhancement Suggestion
-
-* **Determine [which repository the enhancement should be suggested in](#capa-and-its-repositories).**
-* **Perform a [cursory search](https://github.com/fireeye/capa/issues?q=is%3Aissue)** to see if the enhancement has already been suggested. If it has, add a comment to the existing issue instead of opening a new one.
-
-#### How Do I Submit A (Good) Enhancement Suggestion?
-
-Enhancement suggestions are tracked as [GitHub issues](https://guides.github.com/features/issues/). After you've determined [which repository](#capa-and-its-repositories) your enhancement suggestion is related to, create an issue on that repository and provide the following information:
-
-* **Use a clear and descriptive title** for the issue to identify the suggestion.
-* **Provide a step-by-step description of the suggested enhancement** in as many details as possible.
-* **Provide specific examples to demonstrate the steps**. Include copy/pasteable snippets which you use in those examples, as [Markdown code blocks](https://help.github.com/articles/markdown-basics/#multiple-lines).
-* **Describe the current behavior** and **explain which behavior you expected to see instead** and why.
-* **Include screenshots and animated GIFs** which help you demonstrate the steps or point out the part of capa which the suggestion is related to. You can use [this tool](https://www.cockos.com/licecap/) to record GIFs on macOS and Windows, and [this tool](https://github.com/colinkeenan/silentcast) or [this tool](https://github.com/GNOME/byzanz) on Linux.
-* **Explain why this enhancement would be useful** to most capa users and isn't something that can or should be implemented as an external tool that uses capa as a library.
-* **Specify which version of capa you're using.** You can get the exact version by running `capa --version` in your terminal.
-* **Specify the name and version of the OS you're using.**
-
-### Your First Code Contribution
-
-Unsure where to begin contributing to capa? You can start by looking through these `good-first-issue` and `rule-idea` issues:
-
-* [good-first-issue](https://github.com/fireeye/capa/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22) - issues which should only require a few lines of code, and a test or two.
-* [rule-idea](https://github.com/fireeye/capa-rules/issues?q=is%3Aissue+is%3Aopen+label%3A%22rule+idea%22) - issues that describe potential new rule ideas.
-
-Both issue lists are sorted by total number of comments. While not perfect, number of comments is a reasonable proxy for impact a given change will have.
-
-#### Local development
-
-capa and all its resources can be developed locally. 
-For instructions on how to do this, see the "Method 3" section of the [installation guide](https://github.com/fireeye/capa/blob/master/doc/installation.md).
-
-### Pull Requests
-
-The process described here has several goals:
-
- Maintain capa's quality
- Fix problems that are important to users
- Engage the community in working toward the best possible capa
- Enable a sustainable system for capa's maintainers to review contributions
-
-Please follow these steps to have your contribution considered by the maintainers:
-
-1. Follow all instructions in [the template](PULL_REQUEST_TEMPLATE.md)
-2. Follow the [styleguides](#styleguides)
-3. After you submit your pull request, verify that all [status checks](https://help.github.com/articles/about-status-checks/) are passing <details><summary>What if the status checks are failing? </summary>If a status check is failing, and you believe that the failure is unrelated to your change, please leave a comment on the pull request explaining why you believe the failure is unrelated. A maintainer will re-run the status check for you. If we conclude that the failure was a false positive, then we will open an issue to track that problem with our status check suite.</details>
-
-While the prerequisites above must be satisfied prior to having your pull request reviewed, the reviewer(s) may ask you to complete additional design work, tests, or other changes before your pull request can be ultimately accepted.
-
-## Styleguides
-
-### Git Commit Messages
-
-* Use the present tense ("Add feature" not "Added feature")
-* Use the imperative mood ("Move cursor to..." not "Moves cursor to...")
-* Prefix the first line with the component in question ("rules: ..." or "render: ...")
-* Reference issues and pull requests liberally after the first line
-
-### Python Styleguide
-
-All Python code must adhere to the style guide used by capa: 
-
-  1. [PEP8](https://www.python.org/dev/peps/pep-0008/), with clarifications from
-  2. [Willi's style guide](https://docs.google.com/document/d/1iRpeg-w4DtibwytUyC_dDT7IGhNGBP25-nQfuBa-Fyk/edit?usp=sharing), formatted with
-  3. [isort](https://pypi.org/project/isort/) (with line width 120 and ordered by line length), and formatted with
-  4. [black](https://github.com/psf/black) (with line width 120), and formatted with
-  5. [dos2unix](https://linux.die.net/man/1/dos2unix)
-  
-Our CI pipeline will reformat and enforce the Python styleguide.
- 
-### Rules Styleguide
-
-All (non-nursery) capa rules must:
-
-  1. pass the [linter](https://github.com/fireeye/capa/blob/master/scripts/lint.py), and
-  2. be formatted with [capafmt](https://github.com/fireeye/capa/blob/master/scripts/capafmt.py)
-  
-This ensures that all rules meet the same minimum level of quality and are structured in a consistent way.
-Our CI pipeline will reformat and enforce the capa rules styleguide.
+# Contributing to Capa
+
+First off, thanks for taking the time to contribute!
+
+The following is a set of guidelines for contributing to capa and its packages, which are hosted in the [FireEye Organization](https://github.com/fireeye) on GitHub. These are mostly guidelines, not rules. Use your best judgment, and feel free to propose changes to this document in a pull request.
+
+#### Table Of Contents
+
+[Code of Conduct](#code-of-conduct)
+
+[What should I know before I get started?](#what-should-i-know-before-i-get-started)
+  * [Capa and its Repositories](#capa-and-its-repositories)
+  * [Capa Design Decisions](#design-decisions)
+
+[How Can I Contribute?](#how-can-i-contribute)
+  * [Reporting Bugs](#reporting-bugs)
+  * [Suggesting Enhancements](#suggesting-enhancements)
+  * [Your First Code Contribution](#your-first-code-contribution)
+  * [Pull Requests](#pull-requests)
+
+[Styleguides](#styleguides)
+  * [Git Commit Messages](#git-commit-messages)
+  * [Python Styleguide](#python-styleguide)
+  * [Rules Styleguide](#rules-styleguide)
+
+## Code of Conduct
+
+This project and everyone participating in it is governed by the [Capa Code of Conduct](CODE_OF_CONDUCT.md). By participating, you are expected to uphold this code. Please report unacceptable behavior to the maintainers.
+
+## What should I know before I get started?
+
+### Capa and its repositories
+
+We host the capa project as three Github repositories:
+  - [capa](https://github.com/fireeye/capa)
+  - [capa-rules](https://github.com/fireeye/capa-rules)
+  - [capa-testfiles](https://github.com/fireeye/capa-testfiles)
+  
+The command line tools, logic engine, and other Python source code are found in the `capa` repository.
+This is the repository to fork when you want to enhance the features, performance, or user interface of capa.
+Do *not* push rules directly to this repository, instead...
+
+The standard rules contributed by the community are found in the `capa-rules` repository.
+When you have an idea for a new rule, you should open a PR against `capa-rules`.
+We keep `capa` and `capa-rules` separate to distinguish where ideas, bugs, and discussions should happen.
+If you're writing yaml it probably goes in `capa-rules` and if you're writing Python it probably goes in `capa`.
+Also, we encourage users to develop their own rule repositories, so we treat our default set of rules in the same way.
+
+Test fixtures, such as malware samples and analysis workspaces, are found in the `capa-testfiles` repository.
+These are files you'll need in order to run the linter (in `--thorough` mode) and full test suites;
+ however, they take up a lot of space (1GB+), so by keeping `capa-testfiles` separate,
+ a shallow checkout of `capa` and `capa-rules` doesn't take much bandwidth.
+
+### Design Decisions
+
+When we make a significant decision in how we maintain the project and what we can or cannot support,
+ we will document it in the [capa issues tracker](https://github.com/fireeye/capa/issues).
+This is the best place review our discussions about what/how/why we do things in the project.
+If you have a question, check to see if it is documented there.
+If it is *not* documented there, or you can't find an answer, please open a issue.
+We'll link to existing issues when appropriate to keep discussions in one place.
+
+## How Can I Contribute?
+
+### Reporting Bugs
+
+This section guides you through submitting a bug report for capa.
+Following these guidelines helps maintainers and the community understand your report, reproduce the behavior, and find related reports.
+
+Before creating bug reports, please check [this list](#before-submitting-a-bug-report)
+ as you might find out that you don't need to create one.
+When you are creating a bug report, please [include as many details as possible](#how-do-i-submit-a-good-bug-report).
+Fill out [the required template](./ISSUE_TEMPLATE/bug_report.md),
+ the information it asks for helps us resolve issues faster.
+
+> **Note:** If you find a **Closed** issue that seems like it is the same thing that you're experiencing, open a new issue and include a link to the original issue in the body of your new one.
+
+#### Before Submitting A Bug Report
+
+* **Determine [which repository the problem should be reported in](#capa-and-its-repositories)**.
+* **Perform a [cursory search](https://github.com/fireeye/capa/issues?q=is%3Aissue)** to see if the problem has already been reported. If it has **and the issue is still open**, add a comment to the existing issue instead of opening a new one.
+
+#### How Do I Submit A (Good) Bug Report?
+
+Bugs are tracked as [GitHub issues](https://guides.github.com/features/issues/).
+After you've determined [which repository](#capa-and-its-repositories) your bug is related to,
+ create an issue on that repository and provide the following information by filling in 
+ [the template](./ISSUE_TEMPLATE/bug_report.md).
+
+Explain the problem and include additional details to help maintainers reproduce the problem:
+
+* **Use a clear and descriptive title** for the issue to identify the problem.
+* **Describe the exact steps which reproduce the problem** in as many details as possible. For example, start by explaining how you started capa, e.g. which command exactly you used in the terminal, or how you started capa otherwise.
+* **Provide specific examples to demonstrate the steps**. Include links to files or GitHub projects, or copy/pasteable snippets, which you use in those examples. If you're providing snippets in the issue, use [Markdown code blocks](https://help.github.com/articles/markdown-basics/#multiple-lines).
+* **Describe the behavior you observed after following the steps** and point out what exactly is the problem with that behavior.
+* **Explain which behavior you expected to see instead and why.**
+* **Include screenshots and animated GIFs** which show you following the described steps and clearly demonstrate the problem. You can use [this tool](https://www.cockos.com/licecap/) to record GIFs on macOS and Windows, and [this tool](https://github.com/colinkeenan/silentcast) or [this tool](https://github.com/GNOME/byzanz) on Linux.
+* **If you're reporting that capa crashed**, include the stack trace from the terminal. Include the stack trace in the issue in a [code block](https://help.github.com/articles/markdown-basics/#multiple-lines), a [file attachment](https://help.github.com/articles/file-attachments-on-issues-and-pull-requests/), or put it in a [gist](https://gist.github.com/) and provide link to that gist.
+* **If the problem wasn't triggered by a specific action**, describe what you were doing before the problem happened and share more information using the guidelines below.
+
+Provide more context by answering these questions:
+
+* **Did the problem start happening recently** (e.g. after updating to a new version of capa) or was this always a problem?
+* If the problem started happening recently, **can you reproduce the problem in an older version of capa?** What's the most recent version in which the problem doesn't happen? You can download older versions of capa from [the releases page](https://github.com/fireeye/capa/releases).
+* **Can you reliably reproduce the issue?** If not, provide details about how often the problem happens and under which conditions it normally happens.
+* If the problem is related to working with files (e.g. opening and editing files), **does the problem happen for all files and projects or only some?** Does the problem happen only when working with local or remote files (e.g. on network drives), with files of a specific type (e.g. only JavaScript or Python files), with large files or files with very long lines, or with files in a specific encoding? Is there anything else special about the files you are using?
+
+Include details about your configuration and environment:
+
+* **Which version of capa are you using?** You can get the exact version by running `capa --version` in your terminal.
+* **What's the name and version of the OS you're using**?
+
+### Suggesting Enhancements
+
+This section guides you through submitting an enhancement suggestion for capa, including completely new features and minor improvements to existing functionality. Following these guidelines helps maintainers and the community understand your suggestion and find related suggestions.
+
+Before creating enhancement suggestions, please check [this list](#before-submitting-an-enhancement-suggestion) as you might find out that you don't need to create one. When you are creating an enhancement suggestion, please [include as many details as possible](#how-do-i-submit-a-good-enhancement-suggestion). Fill in [the template](./ISSUE_TEMPLATE/feature_request.md), including the steps that you imagine you would take if the feature you're requesting existed.
+
+#### Before Submitting An Enhancement Suggestion
+
+* **Determine [which repository the enhancement should be suggested in](#capa-and-its-repositories).**
+* **Perform a [cursory search](https://github.com/fireeye/capa/issues?q=is%3Aissue)** to see if the enhancement has already been suggested. If it has, add a comment to the existing issue instead of opening a new one.
+
+#### How Do I Submit A (Good) Enhancement Suggestion?
+
+Enhancement suggestions are tracked as [GitHub issues](https://guides.github.com/features/issues/). After you've determined [which repository](#capa-and-its-repositories) your enhancement suggestion is related to, create an issue on that repository and provide the following information:
+
+* **Use a clear and descriptive title** for the issue to identify the suggestion.
+* **Provide a step-by-step description of the suggested enhancement** in as many details as possible.
+* **Provide specific examples to demonstrate the steps**. Include copy/pasteable snippets which you use in those examples, as [Markdown code blocks](https://help.github.com/articles/markdown-basics/#multiple-lines).
+* **Describe the current behavior** and **explain which behavior you expected to see instead** and why.
+* **Include screenshots and animated GIFs** which help you demonstrate the steps or point out the part of capa which the suggestion is related to. You can use [this tool](https://www.cockos.com/licecap/) to record GIFs on macOS and Windows, and [this tool](https://github.com/colinkeenan/silentcast) or [this tool](https://github.com/GNOME/byzanz) on Linux.
+* **Explain why this enhancement would be useful** to most capa users and isn't something that can or should be implemented as an external tool that uses capa as a library.
+* **Specify which version of capa you're using.** You can get the exact version by running `capa --version` in your terminal.
+* **Specify the name and version of the OS you're using.**
+
+### Your First Code Contribution
+
+Unsure where to begin contributing to capa? You can start by looking through these `good-first-issue` and `rule-idea` issues:
+
+* [good-first-issue](https://github.com/fireeye/capa/issues?q=is%3Aissue+is%3Aopen+label%3A%22good+first+issue%22) - issues which should only require a few lines of code, and a test or two.
+* [rule-idea](https://github.com/fireeye/capa-rules/issues?q=is%3Aissue+is%3Aopen+label%3A%22rule+idea%22) - issues that describe potential new rule ideas.
+
+Both issue lists are sorted by total number of comments. While not perfect, number of comments is a reasonable proxy for impact a given change will have.
+
+#### Local development
+
+capa and all its resources can be developed locally. 
+For instructions on how to do this, see the "Method 3" section of the [installation guide](https://github.com/fireeye/capa/blob/master/doc/installation.md).
+
+### Pull Requests
+
+The process described here has several goals:
+
+- Maintain capa's quality
+- Fix problems that are important to users
+- Engage the community in working toward the best possible capa
+- Enable a sustainable system for capa's maintainers to review contributions
+
+Please follow these steps to have your contribution considered by the maintainers:
+
+1. Follow all instructions in [the template](PULL_REQUEST_TEMPLATE.md)
+2. Follow the [styleguides](#styleguides)
+3. After you submit your pull request, verify that all [status checks](https://help.github.com/articles/about-status-checks/) are passing <details><summary>What if the status checks are failing? </summary>If a status check is failing, and you believe that the failure is unrelated to your change, please leave a comment on the pull request explaining why you believe the failure is unrelated. A maintainer will re-run the status check for you. If we conclude that the failure was a false positive, then we will open an issue to track that problem with our status check suite.</details>
+
+While the prerequisites above must be satisfied prior to having your pull request reviewed, the reviewer(s) may ask you to complete additional design work, tests, or other changes before your pull request can be ultimately accepted.
+
+## Styleguides
+
+### Git Commit Messages
+
+* Use the present tense ("Add feature" not "Added feature")
+* Use the imperative mood ("Move cursor to..." not "Moves cursor to...")
+* Prefix the first line with the component in question ("rules: ..." or "render: ...")
+* Reference issues and pull requests liberally after the first line
+
+### Python Styleguide
+
+All Python code must adhere to the style guide used by capa: 
+
+  1. [PEP8](https://www.python.org/dev/peps/pep-0008/), with clarifications from
+  2. [Willi's style guide](https://docs.google.com/document/d/1iRpeg-w4DtibwytUyC_dDT7IGhNGBP25-nQfuBa-Fyk/edit?usp=sharing), formatted with
+  3. [isort](https://pypi.org/project/isort/) (with line width 120 and ordered by line length), and formatted with
+  4. [black](https://github.com/psf/black) (with line width 120), and formatted with
+  5. [dos2unix](https://linux.die.net/man/1/dos2unix)
+  
+Our CI pipeline will reformat and enforce the Python styleguide.
+ 
+### Rules Styleguide
+
+All (non-nursery) capa rules must:
+
+  1. pass the [linter](https://github.com/fireeye/capa/blob/master/scripts/lint.py), and
+  2. be formatted with [capafmt](https://github.com/fireeye/capa/blob/master/scripts/capafmt.py)
+  
+This ensures that all rules meet the same minimum level of quality and are structured in a consistent way.
+Our CI pipeline will reformat and enforce the capa rules styleguide.
--- a/.github/ISSUE_TEMPLATE/bug_report.md
+++ b/.github/ISSUE_TEMPLATE/bug_report.md
@@ -1,47 +1,47 @@
---
-name: Bug report
-about: Create a report to help us improve
-
---
-<!--
-# Is your bug report related to capa rules (for example a false positive)?
-We use sybmodules to separate code, rules and test data. If your issue is related to capa rules, please report it at https://github.com/fireeye/capa-rules/issues.
-
-# Have you checked that your issue isn't already filed?
-Please search if there is a similar issue at https://github.com/fireeye/capa/issues. If there is already a similar issue, please add more details there instead of opening a new one.
-
-# Have you read capa's Code of Conduct?
-By filing an Issue, you are expected to comply with it, including treating everyone with respect: https://github.com/fireeye/capa/blob/master/.github/CODE_OF_CONDUCT.md
-
-# Have you read capa's CONTRIBUTING guide?
-It contains helpful information about how to contribute to capa. Check https://github.com/fireeye/capa/blob/master/.github/CONTRIBUTING.md#reporting-bugs
-->
-
-### Description
-
-<!-- Description of the issue -->
-
-### Steps to Reproduce
-
-<!-- 1. First Step -->
-<!-- 2. Second Step -->
-<!-- 3. and so on… -->
-
-**Expected behavior:**
-
-<!-- What you expect to happen -->
-
-**Actual behavior:**
-
-<!-- What actually happens -->
-
-### Versions
-
-<!-- You can get this information from copy and pasting the output of `capa --version` from the command line.
- Please specify the component you're using (e.g. standalone tool or IDA Pro integration) and your Python version.
- Also, please include the OS and what version of the OS you're running. -->
-
-### Additional Information
-
-<!-- Any additional information, configuration or data that might be necessary to reproduce the issue. -->
-
+---
+name: Bug report
+about: Create a report to help us improve
+
+---
+<!--
+# Is your bug report related to capa rules (for example a false positive)?
+We use submodules to separate code, rules and test data. If your issue is related to capa rules, please report it at https://github.com/fireeye/capa-rules/issues.
+
+# Have you checked that your issue isn't already filed?
+Please search if there is a similar issue at https://github.com/fireeye/capa/issues. If there is already a similar issue, please add more details there instead of opening a new one.
+
+# Have you read capa's Code of Conduct?
+By filing an Issue, you are expected to comply with it, including treating everyone with respect: https://github.com/fireeye/capa/blob/master/.github/CODE_OF_CONDUCT.md
+
+# Have you read capa's CONTRIBUTING guide?
+It contains helpful information about how to contribute to capa. Check https://github.com/fireeye/capa/blob/master/.github/CONTRIBUTING.md#reporting-bugs
+-->
+
+### Description
+
+<!-- Description of the issue -->
+
+### Steps to Reproduce
+
+<!-- 1. First Step -->
+<!-- 2. Second Step -->
+<!-- 3. and so on… -->
+
+**Expected behavior:**
+
+<!-- What you expect to happen -->
+
+**Actual behavior:**
+
+<!-- What actually happens -->
+
+### Versions
+
+<!-- You can get this information from copy and pasting the output of `capa --version` from the command line.
+ Please specify the component you're using (e.g. standalone tool or IDA Pro integration) and your Python version.
+ Also, please include the OS and what version of the OS you're running. -->
+
+### Additional Information
+
+<!-- Any additional information, configuration or data that might be necessary to reproduce the issue. -->
+
--- a/.github/ISSUE_TEMPLATE/feature_request.md
+++ b/.github/ISSUE_TEMPLATE/feature_request.md
@@ -1,35 +1,35 @@
---
-name: Feature request
-about: Suggest an idea for capa
-
---
-<!--
-# Is your issue related to capa rules (for example an idea for a new rule)?
-We use sybmodules to separate code, rules and test data. If your issue is related to capa rules, please report it at https://github.com/fireeye/capa-rules/issues.
-
-# Have you checked that your issue isn't already filed?
-Please search if there is a similar issue at https://github.com/fireeye/capa/issues. If there is already a similar issue, please add more details there instead of opening a new one.
-
-# Have you read capa's Code of Conduct?
-By filing an Issue, you are expected to comply with it, including treating everyone with respect: https://github.com/fireeye/capa/blob/master/.github/CODE_OF_CONDUCT.md
-
-# Have you read capa's CONTRIBUTING guide?
-It contains helpful information about how to contribute to capa. Check https://github.com/fireeye/capa/blob/master/.github/CONTRIBUTING.md#suggesting-enhancements
-->
-
-### Summary
-
-<!-- One paragraph explanation of the feature. -->
-
-### Motivation
-
-<!-- Why are we doing this? What use cases does it support? What is the expected outcome? -->
-
-### Describe alternatives you've considered
-
-<!-- A clear and concise description of the alternative solutions you've considered. -->
-
-## Additional context
-
-<!-- Add any other context or screenshots about the feature request here. -->
-
+---
+name: Feature request
+about: Suggest an idea for capa
+
+---
+<!--
+# Is your issue related to capa rules (for example an idea for a new rule)?
+We use submodules to separate code, rules and test data. If your issue is related to capa rules, please report it at https://github.com/fireeye/capa-rules/issues.
+
+# Have you checked that your issue isn't already filed?
+Please search if there is a similar issue at https://github.com/fireeye/capa/issues. If there is already a similar issue, please add more details there instead of opening a new one.
+
+# Have you read capa's Code of Conduct?
+By filing an Issue, you are expected to comply with it, including treating everyone with respect: https://github.com/fireeye/capa/blob/master/.github/CODE_OF_CONDUCT.md
+
+# Have you read capa's CONTRIBUTING guide?
+It contains helpful information about how to contribute to capa. Check https://github.com/fireeye/capa/blob/master/.github/CONTRIBUTING.md#suggesting-enhancements
+-->
+
+### Summary
+
+<!-- One paragraph explanation of the feature. -->
+
+### Motivation
+
+<!-- Why are we doing this? What use cases does it support? What is the expected outcome? -->
+
+### Describe alternatives you've considered
+
+<!-- A clear and concise description of the alternative solutions you've considered. -->
+
+## Additional context
+
+<!-- Add any other context or screenshots about the feature request here. -->
+
--- a/.github/pull_request_template.md
+++ b/.github/pull_request_template.md
@@ -0,0 +1,32 @@
+<!--
+Thank you for contributing to capa! :heart:
+
+IMPORTANT NOTE
+It's most important that you submit your improvements. So even if you don't use this complete template we look forward to collaborating!
+
+Please read capa's CONTRIBUTING guide if you haven't done so already.
+It contains helpful information about how to contribute to capa. Check https://github.com/fireeye/capa/blob/master/.github/CONTRIBUTING.md
+
+PR template based on https://embeddedartistry.com/blog/2017/08/04/a-github-pull-request-template-for-your-projects/
+-->
+
+### Description
+
+<!-- Please describe the changes in this PR. Including your motivation and context helps us to review. -->
+
+closes # (issue)
+
+### Type of change
+
+Please update the [CHANGELOG.md](/CHANGELOG.md)
+
+- [ ] Bug fix (non-breaking change which fixes an issue)
+- [ ] New feature (non-breaking change which adds functionality)
+- [ ] Breaking change (fix or feature that would cause existing functionality to not work as expected)
+- [ ] This change requires a documentation update
+  - [ ] I have made the corresponding changes to the documentation
+
+### Tests
+
+- [ ] I have added tests that prove my fix is effective or that my feature works
+- [ ] No new tests needed
--- a/.github/pyinstaller/hooks/hook-smda.py
+++ b/.github/pyinstaller/hooks/hook-smda.py
@@ -0,0 +1,5 @@
+# Copyright (C) 2020 FireEye, Inc. All Rights Reserved.
+import PyInstaller.utils.hooks
+
+# ref: https://groups.google.com/g/pyinstaller/c/amWi0-66uZI/m/miPoKfWjBAAJ
+binaries = PyInstaller.utils.hooks.collect_dynamic_libs("capstone")
--- a/.github/pyinstaller/hooks/hook-vivisect.py
+++ b/.github/pyinstaller/hooks/hook-vivisect.py
@@ -13,3 +13,144 @@ from PyInstaller.utils.hooks import copy_metadata
 #
 # ref: https://github.com/pyinstaller/pyinstaller/issues/1713#issuecomment-162682084
 datas = copy_metadata("vivisect")
+
+excludedimports = [
+    # viv gui requires these heavy libraries,
+    # but viv as a library doesn't.
+    # they shouldn't be installed in our configuration,
+    # but we'll ensure they don't slip in here (such as on developers' systems).
+    "PyQt5",
+    "qt5",
+    "pyqtwebengine",
+    # the above are imported by these viv modules.
+    # so really, we'd want to exclude these submodules of viv.
+    # but i dont think this works.
+    "vqt",
+    "vdb.qt",
+    "envi.qt",
+    # unused by capa
+    "pyasn1",
+]
+
+hiddenimports = [
+    # vivisect does manual/runtime importing of its modules,
+    # so declare the things that could be imported here.
+    "vivisect",
+    "vivisect.analysis",
+    "vivisect.analysis.amd64",
+    "vivisect.analysis.amd64",
+    "vivisect.analysis.amd64.emulation",
+    "vivisect.analysis.amd64.golang",
+    "vivisect.analysis.crypto",
+    "vivisect.analysis.crypto",
+    "vivisect.analysis.crypto.constants",
+    "vivisect.analysis.elf",
+    "vivisect.analysis.elf",
+    "vivisect.analysis.elf.elfplt",
+    "vivisect.analysis.elf.libc_start_main",
+    "vivisect.analysis.generic",
+    "vivisect.analysis.generic",
+    "vivisect.analysis.generic.codeblocks",
+    "vivisect.analysis.generic.emucode",
+    "vivisect.analysis.generic.entrypoints",
+    "vivisect.analysis.generic.funcentries",
+    "vivisect.analysis.generic.impapi",
+    "vivisect.analysis.generic.mkpointers",
+    "vivisect.analysis.generic.pointers",
+    "vivisect.analysis.generic.pointertables",
+    "vivisect.analysis.generic.relocations",
+    "vivisect.analysis.generic.strconst",
+    "vivisect.analysis.generic.switchcase",
+    "vivisect.analysis.generic.thunks",
+    "vivisect.analysis.generic.noret",
+    "vivisect.analysis.i386",
+    "vivisect.analysis.i386",
+    "vivisect.analysis.i386.calling",
+    "vivisect.analysis.i386.golang",
+    "vivisect.analysis.i386.importcalls",
+    "vivisect.analysis.i386.instrhook",
+    "vivisect.analysis.i386.thunk_bx",
+    "vivisect.analysis.ms",
+    "vivisect.analysis.ms",
+    "vivisect.analysis.ms.hotpatch",
+    "vivisect.analysis.ms.localhints",
+    "vivisect.analysis.ms.msvc",
+    "vivisect.analysis.ms.msvcfunc",
+    "vivisect.analysis.ms.vftables",
+    "vivisect.analysis.pe",
+    "vivisect.impapi.posix.amd64",
+    "vivisect.impapi.posix.i386",
+    "vivisect.impapi.windows",
+    "vivisect.impapi.windows.amd64",
+    "vivisect.impapi.windows.i386",
+    "vivisect.impapi.winkern.i386",
+    "vivisect.impapi.winkern.amd64",
+    "vivisect.parsers.blob",
+    "vivisect.parsers.elf",
+    "vivisect.parsers.ihex",
+    "vivisect.parsers.macho",
+    "vivisect.parsers.pe",
+    "vivisect.storage",
+    "vivisect.storage.basicfile",
+    "vstruct.constants",
+    "vstruct.constants.ntstatus",
+    "vstruct.defs",
+    "vstruct.defs.arm7",
+    "vstruct.defs.bmp",
+    "vstruct.defs.dns",
+    "vstruct.defs.elf",
+    "vstruct.defs.gif",
+    "vstruct.defs.ihex",
+    "vstruct.defs.inet",
+    "vstruct.defs.java",
+    "vstruct.defs.kdcom",
+    "vstruct.defs.macho",
+    "vstruct.defs.macho.const",
+    "vstruct.defs.macho.fat",
+    "vstruct.defs.macho.loader",
+    "vstruct.defs.macho.stabs",
+    "vstruct.defs.minidump",
+    "vstruct.defs.pcap",
+    "vstruct.defs.pe",
+    "vstruct.defs.pptp",
+    "vstruct.defs.rar",
+    "vstruct.defs.swf",
+    "vstruct.defs.win32",
+    "vstruct.defs.windows",
+    "vstruct.defs.windows.win_5_1_i386",
+    "vstruct.defs.windows.win_5_1_i386.ntdll",
+    "vstruct.defs.windows.win_5_1_i386.ntoskrnl",
+    "vstruct.defs.windows.win_5_1_i386.win32k",
+    "vstruct.defs.windows.win_5_2_i386",
+    "vstruct.defs.windows.win_5_2_i386.ntdll",
+    "vstruct.defs.windows.win_5_2_i386.ntoskrnl",
+    "vstruct.defs.windows.win_5_2_i386.win32k",
+    "vstruct.defs.windows.win_6_1_amd64",
+    "vstruct.defs.windows.win_6_1_amd64.ntdll",
+    "vstruct.defs.windows.win_6_1_amd64.ntoskrnl",
+    "vstruct.defs.windows.win_6_1_amd64.win32k",
+    "vstruct.defs.windows.win_6_1_i386",
+    "vstruct.defs.windows.win_6_1_i386.ntdll",
+    "vstruct.defs.windows.win_6_1_i386.ntoskrnl",
+    "vstruct.defs.windows.win_6_1_i386.win32k",
+    "vstruct.defs.windows.win_6_1_wow64",
+    "vstruct.defs.windows.win_6_1_wow64.ntdll",
+    "vstruct.defs.windows.win_6_2_amd64",
+    "vstruct.defs.windows.win_6_2_amd64.ntdll",
+    "vstruct.defs.windows.win_6_2_amd64.ntoskrnl",
+    "vstruct.defs.windows.win_6_2_amd64.win32k",
+    "vstruct.defs.windows.win_6_2_i386",
+    "vstruct.defs.windows.win_6_2_i386.ntdll",
+    "vstruct.defs.windows.win_6_2_i386.ntoskrnl",
+    "vstruct.defs.windows.win_6_2_i386.win32k",
+    "vstruct.defs.windows.win_6_2_wow64",
+    "vstruct.defs.windows.win_6_2_wow64.ntdll",
+    "vstruct.defs.windows.win_6_3_amd64",
+    "vstruct.defs.windows.win_6_3_amd64.ntdll",
+    "vstruct.defs.windows.win_6_3_amd64.ntoskrnl",
+    "vstruct.defs.windows.win_6_3_i386",
+    "vstruct.defs.windows.win_6_3_i386.ntdll",
+    "vstruct.defs.windows.win_6_3_i386.ntoskrnl",
+    "vstruct.defs.windows.win_6_3_wow64",
+    "vstruct.defs.windows.win_6_3_wow64.ntdll",
+]
--- a/.github/pyinstaller/pyinstaller.spec
+++ b/.github/pyinstaller/pyinstaller.spec
@@ -16,9 +16,10 @@ with open('./capa/version.py', 'wb') as f:
    #                 - commits since
    #                   g------- git hash fragment
    version = (subprocess.check_output(["git", "describe", "--always", "--tags", "--long"])
+               .decode("utf-8")
               .strip()
               .replace("tags/", ""))
-    f.write("__version__ = '%s'" % version)
+    f.write(("__version__ = '%s'" % version).encode("utf-8"))

 a = Analysis(
    # when invoking pyinstaller from the project root,
@@ -41,128 +42,6 @@ a = Analysis(
        # ref: https://stackoverflow.com/a/62278462/87207
        (os.path.dirname(wcwidth.__file__), 'wcwidth')
    ],
-    hiddenimports=[
-        # vivisect does manual/runtime importing of its modules,
-        # so declare the things that could be imported here.
-        "vivisect",
-        "vivisect.analysis",
-        "vivisect.analysis.amd64",
-        "vivisect.analysis.amd64",
-        "vivisect.analysis.amd64.emulation",
-        "vivisect.analysis.amd64.golang",
-        "vivisect.analysis.crypto",
-        "vivisect.analysis.crypto",
-        "vivisect.analysis.crypto.constants",
-        "vivisect.analysis.elf",
-        "vivisect.analysis.elf",
-        "vivisect.analysis.elf.elfplt",
-        "vivisect.analysis.elf.libc_start_main",
-        "vivisect.analysis.generic",
-        "vivisect.analysis.generic",
-        "vivisect.analysis.generic.codeblocks",
-        "vivisect.analysis.generic.emucode",
-        "vivisect.analysis.generic.entrypoints",
-        "vivisect.analysis.generic.funcentries",
-        "vivisect.analysis.generic.impapi",
-        "vivisect.analysis.generic.mkpointers",
-        "vivisect.analysis.generic.pointers",
-        "vivisect.analysis.generic.pointertables",
-        "vivisect.analysis.generic.relocations",
-        "vivisect.analysis.generic.strconst",
-        "vivisect.analysis.generic.switchcase",
-        "vivisect.analysis.generic.thunks",
-        "vivisect.analysis.i386",
-        "vivisect.analysis.i386",
-        "vivisect.analysis.i386.calling",
-        "vivisect.analysis.i386.golang",
-        "vivisect.analysis.i386.importcalls",
-        "vivisect.analysis.i386.instrhook",
-        "vivisect.analysis.i386.thunk_bx",
-        "vivisect.analysis.ms",
-        "vivisect.analysis.ms",
-        "vivisect.analysis.ms.hotpatch",
-        "vivisect.analysis.ms.localhints",
-        "vivisect.analysis.ms.msvc",
-        "vivisect.analysis.ms.msvcfunc",
-        "vivisect.analysis.ms.vftables",
-        "vivisect.analysis.pe",
-        "vivisect.impapi.posix.amd64",
-        "vivisect.impapi.posix.i386",
-        "vivisect.impapi.windows",
-        "vivisect.impapi.windows.amd64",
-        "vivisect.impapi.windows.i386",
-        "vivisect.impapi.winkern.i386",
-        "vivisect.impapi.winkern.amd64",
-        "vivisect.parsers.blob",
-        "vivisect.parsers.elf",
-        "vivisect.parsers.ihex",
-        "vivisect.parsers.macho",
-        "vivisect.parsers.pe",
-        "vivisect.parsers.utils",
-        "vivisect.storage",
-        "vivisect.storage.basicfile",
-        "vstruct.constants",
-        "vstruct.constants.ntstatus",
-        "vstruct.defs",
-        "vstruct.defs.arm7",
-        "vstruct.defs.bmp",
-        "vstruct.defs.dns",
-        "vstruct.defs.elf",
-        "vstruct.defs.gif",
-        "vstruct.defs.ihex",
-        "vstruct.defs.inet",
-        "vstruct.defs.java",
-        "vstruct.defs.kdcom",
-        "vstruct.defs.macho",
-        "vstruct.defs.macho.const",
-        "vstruct.defs.macho.fat",
-        "vstruct.defs.macho.loader",
-        "vstruct.defs.macho.stabs",
-        "vstruct.defs.minidump",
-        "vstruct.defs.pcap",
-        "vstruct.defs.pe",
-        "vstruct.defs.pptp",
-        "vstruct.defs.rar",
-        "vstruct.defs.swf",
-        "vstruct.defs.win32",
-        "vstruct.defs.windows",
-        "vstruct.defs.windows.win_5_1_i386",
-        "vstruct.defs.windows.win_5_1_i386.ntdll",
-        "vstruct.defs.windows.win_5_1_i386.ntoskrnl",
-        "vstruct.defs.windows.win_5_1_i386.win32k",
-        "vstruct.defs.windows.win_5_2_i386",
-        "vstruct.defs.windows.win_5_2_i386.ntdll",
-        "vstruct.defs.windows.win_5_2_i386.ntoskrnl",
-        "vstruct.defs.windows.win_5_2_i386.win32k",
-        "vstruct.defs.windows.win_6_1_amd64",
-        "vstruct.defs.windows.win_6_1_amd64.ntdll",
-        "vstruct.defs.windows.win_6_1_amd64.ntoskrnl",
-        "vstruct.defs.windows.win_6_1_amd64.win32k",
-        "vstruct.defs.windows.win_6_1_i386",
-        "vstruct.defs.windows.win_6_1_i386.ntdll",
-        "vstruct.defs.windows.win_6_1_i386.ntoskrnl",
-        "vstruct.defs.windows.win_6_1_i386.win32k",
-        "vstruct.defs.windows.win_6_1_wow64",
-        "vstruct.defs.windows.win_6_1_wow64.ntdll",
-        "vstruct.defs.windows.win_6_2_amd64",
-        "vstruct.defs.windows.win_6_2_amd64.ntdll",
-        "vstruct.defs.windows.win_6_2_amd64.ntoskrnl",
-        "vstruct.defs.windows.win_6_2_amd64.win32k",
-        "vstruct.defs.windows.win_6_2_i386",
-        "vstruct.defs.windows.win_6_2_i386.ntdll",
-        "vstruct.defs.windows.win_6_2_i386.ntoskrnl",
-        "vstruct.defs.windows.win_6_2_i386.win32k",
-        "vstruct.defs.windows.win_6_2_wow64",
-        "vstruct.defs.windows.win_6_2_wow64.ntdll",
-        "vstruct.defs.windows.win_6_3_amd64",
-        "vstruct.defs.windows.win_6_3_amd64.ntdll",
-        "vstruct.defs.windows.win_6_3_amd64.ntoskrnl",
-        "vstruct.defs.windows.win_6_3_i386",
-        "vstruct.defs.windows.win_6_3_i386.ntdll",
-        "vstruct.defs.windows.win_6_3_i386.ntoskrnl",
-        "vstruct.defs.windows.win_6_3_wow64",
-        "vstruct.defs.windows.win_6_3_wow64.ntdll",
-    ],
    # when invoking pyinstaller from the project root,
    # this gets run from the project root.
    hookspath=['.github/pyinstaller/hooks'],
@@ -180,6 +59,25 @@ a = Analysis(
        # since we don't spawn a notebook, we can safely remove these.
        "IPython",
        "ipywidgets",
+
+        # these are pulled in by networkx
+        # but we don't need to compute the strongly connected components.
+        "numpy",
+        "scipy",
+        "matplotlib",
+        "pandas",
+        "pytest",
+
+        # deps from viv that we don't use.
+        # this duplicates the entries in `hook-vivisect`,
+        # but works better this way.
+        "vqt",
+        "vdb.qt",
+        "envi.qt",
+        "PyQt5",
+        "qt5",
+        "pyqtwebengine",
+        "pyasn1"
    ])

 a.binaries = a.binaries - TOC([
--- a/.github/workflows/build.yml
+++ b/.github/workflows/build.yml
@@ -1,82 +1,77 @@
-name: build
-
-on:
-  release:
-    types: [edited, published]
-
-jobs:
-  build:
-    name: PyInstaller for ${{ matrix.os }}
-    runs-on: ${{ matrix.os }}
-    strategy:
-      matrix:
-        include:
-          - os: ubuntu-16.04
-            # use old linux so that the shared library versioning is more portable
-            artifact_name: capa
-            asset_name: linux
-          - os: windows-latest
-            artifact_name: capa.exe
-            asset_name: windows
-          - os: macos-latest
-            artifact_name: capa
-            asset_name: macos
-    steps:
-      - name: Checkout capa
-        uses: actions/checkout@v2
-        with:
-          submodules: true
-      - name: Set up Python 2.7
-        uses: actions/setup-python@v2
-        with:
-          python-version: 2.7
-      - if: matrix.os == 'ubuntu-latest'
-        run: sudo apt-get install -y libyaml-dev
-      - if: matrix.os == 'windows-latest'
-        run: |
-          choco install vcredist2008
-          choco install --ignore-dependencies vcpython27
-      - name: Install PyInstaller
-        # pyinstaller 4 doesn't support Python 2.7
-        run: pip install 'pyinstaller==3.*'
-      - name: Install capa
-        run: pip install -e .
-      - name: Build standalone executable
-        run: pyinstaller .github/pyinstaller/pyinstaller.spec
-      - name: Does it run?
-        run: dist/capa "tests/data/Practical Malware Analysis Lab 01-01.dll_"
-      - uses: actions/upload-artifact@v2
-        with:
-          name: ${{ matrix.asset_name }}
-          path: dist/${{ matrix.artifact_name }}
-
-  zip:
-    name: zip ${{ matrix.asset_name }}
-    runs-on: ubuntu-latest
-    needs: build
-    strategy:
-      matrix:
-        include:
-          - asset_name: linux
-            artifact_name: capa
-          - asset_name: windows
-            artifact_name: capa.exe
-          - asset_name: macos
-            artifact_name: capa
-    steps:
-      - name: Download ${{ matrix.asset_name }}
-        uses: actions/download-artifact@v2
-        with:
-          name: ${{ matrix.asset_name }}
-      - name: Set executable flag
-        run: chmod +x ${{ matrix.artifact_name }}
-      - name: Set zip name
-        run: echo "zip_name=capa-${GITHUB_REF#refs/tags/}-${{ matrix.asset_name }}.zip" >> $GITHUB_ENV
-      - name: Zip ${{ matrix.artifact_name }} into ${{ env.zip_name }}
-        run: zip ${{ env.zip_name }} ${{ matrix.artifact_name }}
-      - name: Upload ${{ env.zip_name }} to GH Release
-        uses: svenstaro/upload-release-action@v2
-        with:
-          repo_token: ${{ secrets.GITHUB_TOKEN}}
-          file: ${{ env.zip_name }}
-          tag: ${{ github.ref }}
+name: build
+
+on:
+  release:
+    types: [edited, published]
+
+jobs:
+  build:
+    name: PyInstaller for ${{ matrix.os }}
+    runs-on: ${{ matrix.os }}
+    strategy:
+      matrix:
+        include:
+          - os: ubuntu-16.04
+            # use old linux so that the shared library versioning is more portable
+            artifact_name: capa
+            asset_name: linux
+          - os: windows-2019
+            artifact_name: capa.exe
+            asset_name: windows
+          - os: macos-10.15
+            artifact_name: capa
+            asset_name: macos
+    steps:
+      - name: Checkout capa
+        uses: actions/checkout@v2
+        with:
+          submodules: true
+      - name: Set up Python 3.8
+        uses: actions/setup-python@v2
+        with:
+          python-version: 3.8
+      - if: matrix.os == 'ubuntu-16.04'
+        run: sudo apt-get install -y libyaml-dev
+      - name: Install PyInstaller
+        run: pip install 'pyinstaller==4.2'
+      - name: Install capa
+        run: pip install -e .
+      - name: Build standalone executable
+        run: pyinstaller .github/pyinstaller/pyinstaller.spec
+      - name: Does it run?
+        run: dist/capa "tests/data/Practical Malware Analysis Lab 01-01.dll_"
+      - uses: actions/upload-artifact@v2
+        with:
+          name: ${{ matrix.asset_name }}
+          path: dist/${{ matrix.artifact_name }}
+
+  zip:
+    name: zip ${{ matrix.asset_name }}
+    runs-on: ubuntu-20.04
+    needs: build
+    strategy:
+      matrix:
+        include:
+          - asset_name: linux
+            artifact_name: capa
+          - asset_name: windows
+            artifact_name: capa.exe
+          - asset_name: macos
+            artifact_name: capa
+    steps:
+      - name: Download ${{ matrix.asset_name }}
+        uses: actions/download-artifact@v2
+        with:
+          name: ${{ matrix.asset_name }}
+      - name: Set executable flag
+        run: chmod +x ${{ matrix.artifact_name }}
+      - name: Set zip name
+        run: echo "zip_name=capa-${GITHUB_REF#refs/tags/}-${{ matrix.asset_name }}.zip" >> $GITHUB_ENV
+      - name: Zip ${{ matrix.artifact_name }} into ${{ env.zip_name }}
+        run: zip ${{ env.zip_name }} ${{ matrix.artifact_name }}
+      - name: Upload ${{ env.zip_name }} to GH Release
+        uses: svenstaro/upload-release-action@v2
+        with:
+          repo_token: ${{ secrets.GITHUB_TOKEN}}
+          file: ${{ env.zip_name }}
+          tag: ${{ github.ref }}
--- a/.github/workflows/publish.yml
+++ b/.github/workflows/publish.yml
@@ -1,29 +1,29 @@
-# This workflows will upload a Python Package using Twine when a release is created
-# For more information see: https://help.github.com/en/actions/language-and-framework-guides/using-python-with-github-actions#publishing-to-package-registries
-
-name: publish to pypi
-
-on:
-  release:
-    types: [published]
-
-jobs:
-  deploy:
-    runs-on: ubuntu-latest
-    steps:
-      - uses: actions/checkout@v2
-      - name: Set up Python
-        uses: actions/setup-python@v2
-        with:
-          python-version: '2.7'
-      - name: Install dependencies
-        run: |
-          python -m pip install --upgrade pip
-          pip install setuptools wheel twine
-      - name: Build and publish
-        env:
-          TWINE_USERNAME: ${{ secrets.PYPI_USERNAME }}
-          TWINE_PASSWORD: ${{ secrets.PYPI_PASSWORD }}
-        run: |
-          python setup.py sdist bdist_wheel
-          twine upload --skip-existing dist/*
+# This workflows will upload a Python Package using Twine when a release is created
+# For more information see: https://help.github.com/en/actions/language-and-framework-guides/using-python-with-github-actions#publishing-to-package-registries
+
+name: publish to pypi
+
+on:
+  release:
+    types: [published]
+
+jobs:
+  deploy:
+    runs-on: ubuntu-20.04
+    steps:
+      - uses: actions/checkout@v2
+      - name: Set up Python
+        uses: actions/setup-python@v2
+        with:
+          python-version: '2.7'
+      - name: Install dependencies
+        run: |
+          python -m pip install --upgrade pip
+          pip install setuptools wheel twine
+      - name: Build and publish
+        env:
+          TWINE_USERNAME: ${{ secrets.PYPI_USERNAME }}
+          TWINE_PASSWORD: ${{ secrets.PYPI_PASSWORD }}
+        run: |
+          python setup.py sdist bdist_wheel
+          twine upload --skip-existing dist/*
--- a/.github/workflows/tag.yml
+++ b/.github/workflows/tag.yml
@@ -0,0 +1,24 @@
+name: tag
+
+on:
+  release:
+    types: [published]
+
+jobs:
+  tag:
+    name: Tag capa rules
+    runs-on: ubuntu-20.04
+    steps:
+    - name: Checkout capa-rules
+      uses: actions/checkout@v2
+      with:
+        repository: fireeye/capa-rules
+        token: ${{ secrets.CAPA_TOKEN }}
+    - name: Tag capa-rules
+      run: git tag ${{ github.event.release.tag_name }}
+    - name: Push tag to capa-rules
+      uses: ad-m/github-push-action@master
+      with:
+        repository: fireeye/capa-rules
+        github_token: ${{ secrets.CAPA_TOKEN }}
+        tags: true
--- a/.github/workflows/tests.yml
+++ b/.github/workflows/tests.yml
@@ -2,13 +2,13 @@ name: CI

 on:
  push:
-    branches: [ master ]
+    branches: [ master, master-py2 ]
  pull_request:
-    branches: [ master ]
+    branches: [ master, master-py2 ]

 jobs:
  code_style:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
    - name: Checkout capa
      uses: actions/checkout@v2
@@ -24,7 +24,7 @@ jobs:
      run: black -l 120 --check .

  rule_linter:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
    - name: Checkout capa with rules submodule
      uses: actions/checkout@v2
@@ -41,30 +41,39 @@ jobs:
      run: python scripts/lint.py rules/

  tests:
-    name: Tests in ${{ matrix.python }}
-    runs-on: ubuntu-latest
+    name: Tests in ${{ matrix.python-version }} on ${{ matrix.os }}
+    runs-on: ${{ matrix.os }}
    needs: [code_style, rule_linter]
    strategy:
      fail-fast: false
      matrix:
+        os: [ubuntu-20.04, windows-2019, macos-10.15]
+        # across all operating systems
+        python-version: [3.6, 3.9]
        include:
-          - python: 2.7
-          - python: 3.7
-          - python: 3.8
-          - python: 3.9.1
+          # on Ubuntu run these as well
+          - os: ubuntu-20.04
+            python-version: 2.7
+          - os: ubuntu-20.04
+            python-version: 3.7
+          - os: ubuntu-20.04
+            python-version: 3.8
    steps:
    - name: Checkout capa with submodules
      uses: actions/checkout@v2
      with:
        submodules: true
-    - name: Set up Python ${{ matrix.python }}
+    - name: Set up Python ${{ matrix.python-version }}
      uses: actions/setup-python@v2
      with:
-        python-version: ${{ matrix.python }}
+        python-version: ${{ matrix.python-version }}
    - name: Install pyyaml
+      if: matrix.os == 'ubuntu-20.04'
      run: sudo apt-get install -y libyaml-dev
+    - name: Install Microsoft Visual C++ 9.0
+      if: matrix.os == 'windows-2019' && matrix.python-version == '2.7'
+      run: choco install vcpython27
    - name: Install capa
      run: pip install -e .[dev]
    - name: Run tests
      run: pytest tests/
-
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
--- a/README.md
+++ b/README.md
@@ -1,7 +1,10 @@
 ![capa](.github/logo.png)

+[![PyPI - Python Version](https://img.shields.io/pypi/pyversions/flare-capa)](https://pypi.org/project/flare-capa)
+[![Last release](https://img.shields.io/github/v/release/fireeye/capa)](https://github.com/fireeye/capa/releases)
+[![Number of rules](https://img.shields.io/badge/rules-485-blue.svg)](https://github.com/fireeye/capa-rules)
 [![CI status](https://github.com/fireeye/capa/workflows/CI/badge.svg)](https://github.com/fireeye/capa/actions?query=workflow%3ACI+event%3Apush+branch%3Amaster)
-[![Number of rules](https://img.shields.io/badge/rules-458-blue.svg)](https://github.com/fireeye/capa-rules)
+[![Downloads](https://img.shields.io/github/downloads/fireeye/capa/total)](https://github.com/fireeye/capa/releases)
 [![License](https://img.shields.io/badge/license-Apache--2.0-green.svg)](LICENSE.txt)

 capa detects capabilities in executable files.
@@ -146,11 +149,10 @@ rule:
 The [github.com/fireeye/capa-rules](https://github.com/fireeye/capa-rules) repository contains hundreds of standard library rules that are distributed with capa.
 Please learn to write rules and contribute new entries as you find interesting techniques in malware.

-If you use IDA Pro, then you use can use the [capa explorer IDA plugin](capa/ida/plugin/).
-capa explorer lets you quickly identify and navigate to interesting areas of a program and dissect capa rule matches at
-the assembly level.
+If you use IDA Pro, then you can use the [capa explorer](capa/ida/plugin/) plugin.
+capa explorer helps you identify interesting areas of a program and build new capa rules using features extracted directly from your IDA Pro database.

-![capa + IDA Pro integration](doc/img/ida_plugin_intro.gif)
+![capa + IDA Pro integration](doc/img/explorer_expanded.png)

 # further information
 ## capa
--- a/capa/features/init.py
+++ b/capa/features/init.py
@@ -38,6 +38,20 @@ def hex_string(h):
    return " ".join(h[i : i + 2] for i in range(0, len(h), 2)).upper()


+def escape_string(s):
+    """escape special characters"""
+    s = repr(s)
+    if not s.startswith(('"', "'")):
+        # u'hello\r\nworld' -> hello\\r\\nworld
+        s = s[2:-1]
+    else:
+        # 'hello\r\nworld' -> hello\\r\\nworld
+        s = s[1:-1]
+    s = s.replace("\\'", "'")  # repr() may escape "'" in some edge cases, remove
+    s = s.replace('"', '\\"')  # repr() does not escape '"', add
+    return s
+
+
 class Feature(object):
    def __init__(self, value, arch=None, description=None):
        """
--- a/capa/features/extractors/ida/helpers.py
+++ b/capa/features/extractors/ida/helpers.py
@@ -351,6 +351,10 @@ def find_data_reference_from_insn(insn, max_depth=10):
            # break if circular reference
            break

+        if not idaapi.is_mapped(data_refs[0]):
+            # break if address is not mapped
+            break
+
        depth += 1
        if depth > max_depth:
            # break if max depth
--- a/capa/features/extractors/viv/init.py
+++ b/capa/features/extractors/viv/init.py
@@ -8,11 +8,7 @@

 import types

-import file
-import insn
-import function
 import viv_utils
-import basicblock

 import capa.features.extractors
 import capa.features.extractors.viv.file
@@ -42,7 +38,7 @@ def add_va_int_cast(o):
    this bit of skullduggery lets use cast viv-utils objects as ints.
    the correct way of doing this is to update viv-utils (or subclass the objects here).
    """
-    setattr(o, "__int__", types.MethodType(get_va, o, type(o)))
+    setattr(o, "__int__", types.MethodType(get_va, o))
    return o


--- a/capa/features/extractors/viv/basicblock.py
+++ b/capa/features/extractors/viv/basicblock.py
@@ -125,11 +125,16 @@ def get_printable_len(oper):


 def is_printable_ascii(chars):
-    return all(ord(c) < 127 and c in string.printable for c in chars)
+    try:
+        chars_str = chars.decode("ascii")
+    except UnicodeDecodeError:
+        return False
+    else:
+        return all(c in string.printable for c in chars_str)


 def is_printable_utf16le(chars):
-    if all(c == "\x00" for c in chars[1::2]):
+    if all(c == b"\x00" for c in chars[1::2]):
        return is_printable_ascii(chars[::2])


--- a/capa/features/extractors/viv/insn.py
+++ b/capa/features/extractors/viv/insn.py
@@ -239,7 +239,7 @@ def read_bytes(vw, va):
    """
    segm = vw.getSegment(va)
    if not segm:
-        raise envi.SegmentationViolation()
+        raise envi.SegmentationViolation(va)

    segm_end = segm[0] + segm[1]
    try:
@@ -499,6 +499,10 @@ def extract_insn_cross_section_cflow(f, bb, insn):
    inspect the instruction for a CALL or JMP that crosses section boundaries.
    """
    for va, flags in insn.getBranches():
+        if va is None:
+            # va may be none for dynamic branches that haven't been resolved, such as `jmp eax`.
+            continue
+
        if flags & envi.BR_FALL:
            continue

--- a/capa/features/freeze.py
+++ b/capa/features/freeze.py
@@ -264,6 +264,15 @@ def main(argv=None):
    parser.add_argument(
        "-f", "--format", choices=[f[0] for f in formats], default="auto", help="Select sample format, %s" % format_help
    )
+    if sys.version_info >= (3, 0):
+        parser.add_argument(
+            "-b",
+            "--backend",
+            type=str,
+            help="select the backend to use",
+            choices=(capa.main.BACKEND_VIV, capa.main.BACKEND_SMDA),
+            default=capa.main.BACKEND_VIV,
+        )
    args = parser.parse_args(args=argv)

    if args.quiet:
@@ -276,7 +285,8 @@ def main(argv=None):
        logging.basicConfig(level=logging.INFO)
        logging.getLogger().setLevel(logging.INFO)

-    extractor = capa.main.get_extractor(args.sample, args.format)
+    backend = args.backend if sys.version_info > (3, 0) else capa.main.BACKEND_VIV
+    extractor = capa.main.get_extractor(args.sample, args.format, backend)
    with open(args.output, "wb") as f:
        f.write(dump(extractor))

--- a/capa/ida/helpers/init.py
+++ b/capa/ida/helpers/init.py
@@ -82,14 +82,26 @@ def get_func_start_ea(ea):
    return f if f is None else f.start_ea


-def collect_metadata():
+def get_file_md5():
+    """ """
    md5 = idautils.GetInputFileMD5()
    if not isinstance(md5, six.string_types):
        md5 = capa.features.bytes_to_str(md5)
+    return md5

+
+def get_file_sha256():
+    """ """
    sha256 = idaapi.retrieve_input_file_sha256()
    if not isinstance(sha256, six.string_types):
        sha256 = capa.features.bytes_to_str(sha256)
+    return sha256
+
+
+def collect_metadata():
+    """ """
+    md5 = get_file_md5()
+    sha256 = get_file_sha256()

    return {
        "timestamp": datetime.datetime.now().isoformat(),
--- a/capa/ida/plugin/README.md
+++ b/capa/ida/plugin/README.md
@@ -1,49 +1,35 @@
 ![capa explorer](../../../.github/capa-explorer-logo.png)

-capa explorer is an IDA Pro plugin written in Python that integrates the FLARE team's open-source framework, capa, with IDA. capa is a framework that uses a well-defined collection of rules to 
+capa explorer is an IDAPython plugin that integrates the FLARE team's open-source framework, capa, with IDA Pro. capa is a framework that uses a well-defined collection of rules to 
 identify capabilities in a program. You can run capa against a PE file or shellcode and it tells you what it thinks the program can do. For example, it might suggest that 
-the program is a backdoor, can install services, or relies on HTTP to communicate. You can use capa explorer to run capa directly on an IDA database without requiring access
-to the source binary. Once a database has been analyzed, capa explorer can be used to quickly identify and navigate to interesting areas of a program 
-and dissect capa rule matches at the assembly level.
+the program is a backdoor, can install services, or relies on HTTP to communicate. capa explorer runs capa directly against your IDA Pro database (IDB) without requiring access
+to the original binary file. Once a database has been analyzed, capa explorer helps you identify interesting areas of a program and build new capa rules using features extracted from your IDB.

 We love using capa explorer during malware analysis because it teaches us what parts of a program suggest a behavior. As we click on rows, capa explorer jumps directly 
-to important addresses in the IDA Pro database and highlights key features in the Disassembly view so they stand out visually. To illustrate, we use capa explorer to 
+to important addresses in the IDB and highlights key features in the Disassembly view so they stand out visually. To illustrate, we use capa explorer to 
 analyze Lab 14-02 from [Practical Malware Analysis](https://nostarch.com/malware) (PMA) available [here](https://practicalmalwareanalysis.com/labs/). Our goal is to understand 
 the program's functionality.

 After loading Lab 14-02 into IDA and analyzing the database with capa explorer, we see that capa detected a rule match for `self delete via COMSPEC environment variable`:

-![](../../../doc/img/ida_plugin_example_1.png)
+![](../../../doc/img/explorer_condensed.png)

-We can use capa explorer to navigate the IDA Disassembly view directly to the suspect function and get an assembly-level breakdown of why capa matched `self delete via COMSPEC environment variable` 
-for this particular function.
+We can use capa explorer to navigate our Disassembly view directly to the suspect function and get an assembly-level breakdown of why capa matched `self delete via COMSPEC environment variable`.

-![](../../../doc/img/ida_plugin_example_2.png)
+![](../../../doc/img/explorer_expanded.png)

 Using the `Rule Information` and `Details` columns capa explorer shows us that the suspect function matched `self delete via COMSPEC environment variable` because it contains capa rule matches for `create process`, `get COMSPEC environment variable`,
-and `query environment variable`, references to the strings `COMSPEC`, ` > nul`, and `/c del`, and calls to the Windows API functions `GetEnvironmentVariableA` and `ShellExecuteEx`.
+and `query environment variable`, references to the strings `COMSPEC`, ` > nul`, and `/c del `, and calls to the Windows API functions `GetEnvironmentVariableA` and `ShellExecuteEx`.
+
+capa explorer also helps you build new capa rules. To start select the `Rule Generator` tab, navigate to a function in your Disassembly view,
+and click `Analyze`. capa explorer will extract features from the function and display them in the `Features` pane. You can add features listed in this pane to the `Editor` pane
+by either double-clicking a feature or using multi-select + right-click to add multiple features at once. The `Preview` and `Editor` panes help edit your rule. Use the `Preview` pane
+to modify the rule text directly and the `Editor` pane to construct and rearrange your hierarchy of statements and features. When you finish a rule you can save it directly to a file by clicking `Save`.
+
+![](../../../doc/img/rulegen_expanded.png)

 For more information on the FLARE team's open-source framework, capa, check out the overview in our first [blog](https://www.fireeye.com/blog/threat-research/2020/07/capa-automatically-identify-malware-capabilities.html).

-## Features
-
-![](../../../doc/img/ida_plugin_intro.gif)
-
-* Display capa results in an interactive tree view of rule matches and their locations in the current database
-* Search for keywords or phrases found in the `Rule Information`, `Address`, or `Details` columns
-* Display rule source content when a user hovers their cursor over a rule match
-* Double-click `Address` column to view associated feature in the IDA Disassembly view
-* Limit tree view results to the function currently displayed in the IDA Disassembly view; update results as a user navigates to different functions
-* Export results as formatted JSON by navigating to `File > Export results...`
-* Remember a user's capa rules directory for future runs; change capa rules directory by navigating to `Rules > Change rules directory...`
-* Automatically re-analyze database when user performs a program rebase
-* Automatically update results when IDA is used to rename a function
-* Select one or more checkboxes to highlight the associated addresses in the IDA Disassembly view
-* Right-click a function match to rename it; the new function name is propagated to the current IDA database
-* Right-click to copy a result by column or by row
-* Sort results by column
-* Reset tree view and IDA Disassembly view highlighting by clicking `Reset`
-
 ## Getting Started

 ### Requirements
@@ -56,7 +42,7 @@ If you encounter issues with your specific setup, please open a new [Issue](http

 ### Supported File Types

-capa explorer is limited to the file types supported by capa, which includes:
+capa explorer is limited to the file types supported by capa, which include:

 * Windows 32-bit and 64-bit PE files
 * Windows 32-bit and 64-bit shellcode
@@ -74,38 +60,48 @@ You can install capa explorer using the following steps:

 ### Usage

-1. Run IDA and analyze a supported file type (select the `Manual Load` and `Load Resources` options in IDA for best results)
+1. Open IDA and analyze a supported file type (select the `Manual Load` and `Load Resources` options in IDA for best results)
 2. Open capa explorer in IDA by navigating to `Edit > Plugins > FLARE capa explorer` or using the keyboard shortcut `Alt+F5`
-3. Click the `Analyze` button
+3. Select the `Program Analysis` tab
+4. Click the `Analyze` button

 When running capa explorer for the first time you are prompted to select a file directory containing capa rules. The plugin conveniently
-remembers your selection for future runs; you can change this selection by navigating to `Rules > Change rules directory...`. We recommend 
+remembers your selection for future runs; you can change this selection and other default settings by clicking `Settings`. We recommend 
 downloading and using the [standard collection of capa rules](https://github.com/fireeye/capa-rules) when getting started with the plugin.

-#### Tips
+#### Tips for Program Analysis

 * Start analysis by clicking the `Analyze` button
-* Reset the plugin user interface and remove highlighting from IDA disassembly view by clicking the `Reset` button
-* Change your capa rules directory by navigating to `Rules > Change rules directory...` from the plugin menu
+* Reset the plugin user interface and remove highlighting from your Disassembly view by clicking the `Reset` button
+* Change your capa rules directory and other default settings by clicking `Settings`
 * Hover your cursor over a rule match to view the source content of the rule
-* Double-click the `Address` column to navigate the IDA Disassembly view to the associated feature
+* Double-click the `Address` column to navigate your Disassembly view to the address of the associated feature
 * Double-click a result in the `Rule Information` column to expand its children
-* Select a checkbox in the `Rule Information` column to highlight the address of the associated feature in the IDA Dissasembly view
+* Select a checkbox in the `Rule Information` column to highlight the address of the associated feature in your Dissasembly view
+
+#### Tips for Rule Generator
+
+* Navigate to a function in your Disassembly view and click`Analyze` to get started
+* Double-click or use multi-select + right-click to add features from the `Features` pane to the `Editor` pane
+* Right-click features in the `Editor` pane to make context-specific modifications
+* Drag-and-drop (single click + multi-select support) features in the `Editor` pane to construct your hierarchy of statements and features
+* Right-click anywhere in the `Editor` pane not on a feature to remove all features
+* Add descriptions or comments to a feature by editing the corresponding column in the `Editor` pane
+* Directly edit rule text and metadata fields using the `Preview` pane
+* Change the default rule author and default rule scope displayed in the `Preview` pane by clicking `Settings`

 ## Development

-Because capa explorer is packaged with capa you will need to install capa locally for development.
-
-You can install capa locally by following the steps outlined in `Method 3: Inspecting the capa source code` of the [capa 
+capa explorer is packaged with capa so you will need to install capa locally for development. You can install capa locally by following the steps outlined in `Method 3: Inspecting the capa source code` of the [capa 
 installation guide](https://github.com/fireeye/capa/blob/master/doc/installation.md#method-3-inspecting-the-capa-source-code). Once installed, copy [capa_explorer.py](https://raw.githubusercontent.com/fireeye/capa/master/capa/ida/plugin/capa_explorer.py) 
-to your IDA plugins directory to run the plugin in IDA.
+to your plugins directory to install capa explorer in IDA.

 ### Components

 capa explorer consists of two main components:

-* An IDA [feature extractor](https://github.com/fireeye/capa/tree/master/capa/features/extractors/ida) built on top of IDA's binary analysis engine
-  * This component uses IDAPython to extract [capa features](https://github.com/fireeye/capa-rules/blob/master/doc/format.md#extracted-features) from the IDA database such as strings, 
+* An [feature extractor](https://github.com/fireeye/capa/tree/master/capa/features/extractors/ida) built on top of IDA's binary analysis engine
+  * This component uses IDAPython to extract [capa features](https://github.com/fireeye/capa-rules/blob/master/doc/format.md#extracted-features) from your IDBs such as strings, 
 disassembly, and control flow; these extracted features are used by capa to find feature combinations that result in a rule match
 * An [interactive user interface](https://github.com/fireeye/capa/tree/master/capa/ida/plugin) for displaying and exploring capa rule matches
-  * This component integrates the IDA feature extractor and capa, providing an interactive user interface to dissect rule matches found by capa using features extracted by the IDA feature extractor
+  * This component integrates the feature extractor and capa, providing an interactive user interface to dissect rule matches found by capa using features extracted directly from your IDBs
--- a/capa/ida/plugin/form.py
+++ b/capa/ida/plugin/form.py
--- a/capa/ida/plugin/item.py
+++ b/capa/ida/plugin/item.py
@@ -35,20 +35,19 @@ def location_to_hex(location):
 class CapaExplorerDataItem(object):
    """store data for CapaExplorerDataModel"""

-    def __init__(self, parent, data):
+    def __init__(self, parent, data, can_check=True):
        """initialize item"""
        self.pred = parent
        self._data = data
        self.children = []
        self._checked = False
+        self._can_check = can_check

        # default state for item
-        self.flags = (
-            QtCore.Qt.ItemIsEnabled
-            | QtCore.Qt.ItemIsSelectable
-            | QtCore.Qt.ItemIsTristate
-            | QtCore.Qt.ItemIsUserCheckable
-        )
+        self.flags = QtCore.Qt.ItemIsEnabled | QtCore.Qt.ItemIsSelectable
+
+        if self._can_check:
+            self.flags = self.flags | QtCore.Qt.ItemIsUserCheckable | QtCore.Qt.ItemIsTristate

        if self.pred:
            self.pred.appendChild(self)
@@ -70,6 +69,10 @@ class CapaExplorerDataItem(object):
        """
        self._checked = checked

+    def canCheck(self):
+        """ """
+        return self._can_check
+
    def isChecked(self):
        """get item is checked"""
        return self._checked
@@ -165,7 +168,7 @@ class CapaExplorerRuleItem(CapaExplorerDataItem):

    fmt = "%s (%d matches)"

-    def __init__(self, parent, name, namespace, count, source):
+    def __init__(self, parent, name, namespace, count, source, can_check=True):
        """initialize item

        @param parent: parent node
@@ -175,7 +178,7 @@ class CapaExplorerRuleItem(CapaExplorerDataItem):
        @param source: rule source (tooltip)
        """
        display = self.fmt % (name, count) if count > 1 else name
-        super(CapaExplorerRuleItem, self).__init__(parent, [display, "", namespace])
+        super(CapaExplorerRuleItem, self).__init__(parent, [display, "", namespace], can_check)
        self._source = source

    @property
@@ -208,14 +211,14 @@ class CapaExplorerFunctionItem(CapaExplorerDataItem):

    fmt = "function(%s)"

-    def __init__(self, parent, location):
+    def __init__(self, parent, location, can_check=True):
        """initialize item

        @param parent: parent node
        @param location: virtual address of function as seen by IDA
        """
        super(CapaExplorerFunctionItem, self).__init__(
-            parent, [self.fmt % idaapi.get_name(location), location_to_hex(location), ""]
+            parent, [self.fmt % idaapi.get_name(location), location_to_hex(location), ""], can_check
        )

    @property
--- a/capa/ida/plugin/model.py
+++ b/capa/ida/plugin/model.py
@@ -6,7 +6,7 @@
 #  is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 # See the License for the specific language governing permissions and limitations under the License.

-from collections import deque
+from collections import deque, defaultdict

 import idc
 import idaapi
@@ -110,6 +110,8 @@ class CapaExplorerDataModel(QtCore.QAbstractItemModel):

        if role == QtCore.Qt.CheckStateRole and column == CapaExplorerDataModel.COLUMN_INDEX_RULE_INFORMATION:
            # inform view how to display content of checkbox - un/checked
+            if not item.canCheck():
+                return None
            return QtCore.Qt.Checked if item.isChecked() else QtCore.Qt.Unchecked

        if role == QtCore.Qt.FontRole and column in (
@@ -424,14 +426,28 @@ class CapaExplorerDataModel(QtCore.QAbstractItemModel):
        for child in match.get("children", []):
            self.render_capa_doc_match(parent2, child, doc)

-    def render_capa_doc(self, doc):
-        """render capa features specified in doc
-
-        @param doc: capa result doc
-        """
-        # inform model that changes are about to occur
-        self.beginResetModel()
+    def render_capa_doc_by_function(self, doc):
+        """ """
+        matches_by_function = {}
+        for rule in rutils.capability_rules(doc):
+            for ea in rule["matches"].keys():
+                ea = capa.ida.helpers.get_func_start_ea(ea)
+                if ea is None:
+                    # file scope, skip for rendering in this mode
+                    continue
+                if None is matches_by_function.get(ea, None):
+                    matches_by_function[ea] = CapaExplorerFunctionItem(self.root_node, ea, can_check=False)
+                CapaExplorerRuleItem(
+                    matches_by_function[ea],
+                    rule["meta"]["name"],
+                    rule["meta"].get("namespace"),
+                    len(rule["matches"]),
+                    rule["source"],
+                    can_check=False,
+                )

+    def render_capa_doc_by_program(self, doc):
+        """ """
        for rule in rutils.capability_rules(doc):
            rule_name = rule["meta"]["name"]
            rule_namespace = rule["meta"].get("namespace")
@@ -451,6 +467,19 @@ class CapaExplorerDataModel(QtCore.QAbstractItemModel):

                self.render_capa_doc_match(parent2, match, doc)

+    def render_capa_doc(self, doc, by_function):
+        """render capa features specified in doc
+
+        @param doc: capa result doc
+        """
+        # inform model that changes are about to occur
+        self.beginResetModel()
+
+        if by_function:
+            self.render_capa_doc_by_function(doc)
+        else:
+            self.render_capa_doc_by_program(doc)
+
        # inform model changes have ended
        self.endResetModel()

@@ -459,13 +488,17 @@ class CapaExplorerDataModel(QtCore.QAbstractItemModel):

        @param feature: capa feature read from doc
        """
-        if feature[feature["type"]]:
+        key = feature["type"]
+        value = feature[feature["type"]]
+        if value:
+            if key == "string":
+                value = '"%s"' % capa.features.escape_string(value)
            if feature.get("description", ""):
-                return "%s(%s = %s)" % (feature["type"], feature[feature["type"]], feature["description"])
+                return "%s(%s = %s)" % (key, value, feature["description"])
            else:
-                return "%s(%s)" % (feature["type"], feature[feature["type"]])
+                return "%s(%s)" % (key, value)
        else:
-            return "%s" % feature["type"]
+            return "%s" % key

    def render_capa_doc_feature_node(self, parent, feature, locations, doc):
        """process capa doc feature node
@@ -522,7 +555,9 @@ class CapaExplorerDataModel(QtCore.QAbstractItemModel):
            )

        if feature["type"] == "regex":
-            return CapaExplorerStringViewItem(parent, display, location, feature["match"])
+            return CapaExplorerStringViewItem(
+                parent, display, location, '"%s"' % capa.features.escape_string(feature["match"])
+            )

        if feature["type"] == "basicblock":
            return CapaExplorerBlockItem(parent, location)
@@ -547,7 +582,9 @@ class CapaExplorerDataModel(QtCore.QAbstractItemModel):

        if feature["type"] in ("string",):
            # display string preview
-            return CapaExplorerStringViewItem(parent, display, location, feature[feature["type"]])
+            return CapaExplorerStringViewItem(
+                parent, display, location, '"%s"' % capa.features.escape_string(feature[feature["type"]])
+            )

        if feature["type"] in ("import", "export"):
            # display no preview
--- a/capa/ida/plugin/view.py
+++ b/capa/ida/plugin/view.py
@@ -5,15 +5,936 @@
 # Unless required by applicable law or agreed to in writing, software distributed under the License
 #  is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 # See the License for the specific language governing permissions and limitations under the License.
+import re
+from collections import Counter

 import idc
-from PyQt5 import QtCore, QtWidgets
+from PyQt5 import QtGui, QtCore, QtWidgets

+import capa.rules
+import capa.engine
+import capa.ida.helpers
+import capa.features.basicblock
 from capa.ida.plugin.item import CapaExplorerFunctionItem
 from capa.ida.plugin.model import CapaExplorerDataModel

 MAX_SECTION_SIZE = 750

+# default colors used in views
+COLOR_GREEN_RGB = (79, 121, 66)
+COLOR_BLUE_RGB = (37, 147, 215)
+
+
+def calc_level_by_indent(line, prev_level=0):
+    """ """
+    if not len(line.strip()):
+        # blank line, which may occur for comments so we simply use the last level
+        return prev_level
+    stripped = line.lstrip()
+    if stripped.startswith("description"):
+        # need to adjust two spaces when encountering string description
+        line = line[2:]
+    # calc line level based on preceding whitespace
+    return len(line) - len(stripped)
+
+
+def parse_feature_for_node(feature):
+    """ """
+    description = ""
+    comment = ""
+
+    if feature.startswith("- count"):
+        # count is weird, we need to handle special
+        # first, we need to grab the comment, if exists
+        # next, we need to check for an embedded description
+        feature, _, comment = feature.partition("#")
+        m = re.search(r"- count\(([a-zA-Z]+)\((.+)\s+=\s+(.+)\)\):\s*(.+)", feature)
+        if m:
+            # reconstruct count without description
+            feature, value, description, count = m.groups()
+            feature = "- count(%s(%s)): %s" % (feature, value, count)
+    elif not feature.startswith("#"):
+        feature, _, comment = feature.partition("#")
+        feature, _, description = feature.partition("=")
+
+    return map(lambda o: o.strip(), (feature, description, comment))
+
+
+def parse_node_for_feature(feature, description, comment, depth):
+    """ """
+    depth = (depth * 2) + 4
+    display = ""
+
+    if feature.startswith("#"):
+        display += "%s%s\n" % (" " * depth, feature)
+    elif description:
+        if feature.startswith(("- and", "- or", "- optional", "- basic block", "- not")):
+            display += "%s%s" % (" " * depth, feature)
+            if comment:
+                display += " # %s" % comment
+            display += "\n%s- description: %s\n" % (" " * (depth + 2), description)
+        elif feature.startswith("- string"):
+            display += "%s%s" % (" " * depth, feature)
+            if comment:
+                display += " # %s" % comment
+            display += "\n%sdescription: %s\n" % (" " * (depth + 2), description)
+        elif feature.startswith("- count"):
+            # count is weird, we need to format description based on feature type, so we parse with regex
+            # assume format - count(<feature_name>(<feature_value>)): <count>
+            m = re.search(r"- count\(([a-zA-Z]+)\((.+)\)\): (.+)", feature)
+            if m:
+                name, value, count = m.groups()
+                if name in ("string",):
+                    display += "%s%s" % (" " * depth, feature)
+                    if comment:
+                        display += " # %s" % comment
+                    display += "\n%sdescription: %s\n" % (" " * (depth + 2), description)
+                else:
+                    display += "%s- count(%s(%s = %s)): %s" % (
+                        " " * depth,
+                        name,
+                        value,
+                        description,
+                        count,
+                    )
+                    if comment:
+                        display += " # %s\n" % comment
+        else:
+            display += "%s%s = %s" % (" " * depth, feature, description)
+            if comment:
+                display += " # %s\n" % comment
+    else:
+        display += "%s%s" % (" " * depth, feature)
+        if comment:
+            display += " # %s\n" % comment
+
+    return display if display.endswith("\n") else display + "\n"
+
+
+def yaml_to_nodes(s):
+    level = 0
+    for line in s.splitlines():
+        feature, description, comment = parse_feature_for_node(line.strip())
+
+        o = QtWidgets.QTreeWidgetItem(None)
+
+        # set node attributes
+        setattr(o, "capa_level", calc_level_by_indent(line, level))
+
+        if feature.startswith(("- and:", "- or:", "- not:", "- basic block:", "- optional:")):
+            setattr(o, "capa_type", CapaExplorerRulgenEditor.get_node_type_expression())
+        elif feature.startswith("#"):
+            setattr(o, "capa_type", CapaExplorerRulgenEditor.get_node_type_comment())
+        else:
+            setattr(o, "capa_type", CapaExplorerRulgenEditor.get_node_type_feature())
+
+        # set node text
+        for (i, v) in enumerate((feature, description, comment)):
+            o.setText(i, v)
+
+        yield o
+
+
+def iterate_tree(o):
+    """ """
+    itr = QtWidgets.QTreeWidgetItemIterator(o)
+    while itr.value():
+        yield itr.value()
+        itr += 1
+
+
+def calc_item_depth(o):
+    """ """
+    depth = 0
+    while True:
+        if not o.parent():
+            break
+        depth += 1
+        o = o.parent()
+    return depth
+
+
+def build_action(o, display, data, slot):
+    """ """
+    action = QtWidgets.QAction(display, o)
+
+    action.setData(data)
+    action.triggered.connect(lambda checked: slot(action))
+
+    return action
+
+
+def build_context_menu(o, actions):
+    """ """
+    menu = QtWidgets.QMenu()
+
+    for action in actions:
+        if isinstance(action, QtWidgets.QMenu):
+            menu.addMenu(action)
+        else:
+            menu.addAction(build_action(o, *action))
+
+    return menu
+
+
+class CapaExplorerRulgenPreview(QtWidgets.QTextEdit):
+
+    INDENT = " " * 2
+
+    def __init__(self, parent=None):
+        """ """
+        super(CapaExplorerRulgenPreview, self).__init__(parent)
+
+        self.setFont(QtGui.QFont("Courier", weight=QtGui.QFont.Bold))
+        self.setLineWrapMode(QtWidgets.QTextEdit.NoWrap)
+        self.setHorizontalScrollBarPolicy(QtCore.Qt.ScrollBarAsNeeded)
+
+    def reset_view(self):
+        """ """
+        self.clear()
+
+    def load_preview_meta(self, ea, author, scope):
+        """ """
+        metadata_default = [
+            "# generated using capa explorer for IDA Pro",
+            "rule:",
+            "  meta:",
+            "    name: <insert_name>",
+            "    namespace: <insert_namespace>",
+            "    author: %s" % author,
+            "    scope: %s" % scope,
+            "    references: <insert_references>",
+            "    examples:",
+            "      - %s:0x%X" % (capa.ida.helpers.get_file_md5().upper(), ea)
+            if ea
+            else "      - %s" % (capa.ida.helpers.get_file_md5().upper()),
+            "  features:",
+        ]
+        self.setText("\n".join(metadata_default))
+
+    def keyPressEvent(self, e):
+        """intercept key press events"""
+        if e.key() in (QtCore.Qt.Key_Tab, QtCore.Qt.Key_Backtab):
+            # apparently it's not easy to implement tabs as spaces, or multi-line tab or SHIFT + Tab
+            # so we need to implement it ourselves so we can retain properly formatted capa rules
+            # when a user uses the Tab key
+            if self.textCursor().selection().isEmpty():
+                # single line, only worry about Tab
+                if e.key() == QtCore.Qt.Key_Tab:
+                    self.insertPlainText(self.INDENT)
+            else:
+                # multi-line tab or SHIFT + Tab
+                cur = self.textCursor()
+                select_start_ppos = cur.selectionStart()
+                select_end_ppos = cur.selectionEnd()
+
+                scroll_ppos = self.verticalScrollBar().sliderPosition()
+
+                # determine lineno for first selected line, and column
+                cur.setPosition(select_start_ppos)
+                start_lineno = self.count_previous_lines_from_block(cur.block())
+                start_lineco = cur.columnNumber()
+
+                # determine lineno for last selected line
+                cur.setPosition(select_end_ppos)
+                end_lineno = self.count_previous_lines_from_block(cur.block())
+
+                # now we need to indent or dedent the selected lines. for now, we read the text, modify
+                # the lines between start_lineno and end_lineno accordingly, and then reset the view
+                # this might not be the best solution, but it avoids messing around with cursor positions
+                # to determine the beginning of lines
+
+                plain = self.toPlainText().splitlines()
+
+                if e.key() == QtCore.Qt.Key_Tab:
+                    # user Tab, indent selected lines
+                    lines_modified = end_lineno - start_lineno
+                    first_modified = True
+                    change = [self.INDENT + line for line in plain[start_lineno : end_lineno + 1]]
+                else:
+                    # user SHIFT + Tab, dedent selected lines
+                    lines_modified = 0
+                    first_modified = False
+                    change = []
+                    for (lineno, line) in enumerate(plain[start_lineno : end_lineno + 1]):
+                        if line.startswith(self.INDENT):
+                            if lineno == 0:
+                                # keep track if first line is modified, so we can properly display
+                                # the text selection later
+                                first_modified = True
+                            lines_modified += 1
+                            line = line[len(self.INDENT) :]
+                        change.append(line)
+
+                # apply modifications, and reset view
+                plain[start_lineno : end_lineno + 1] = change
+                self.setPlainText("\n".join(plain) + "\n")
+
+                # now we need to properly adjust the selection positions, so users don't have to
+                # re-select when indenting or dedenting the same lines repeatedly
+                if e.key() == QtCore.Qt.Key_Tab:
+                    # user Tab, increase increment selection positions
+                    select_start_ppos += len(self.INDENT)
+                    select_end_ppos += (lines_modified * len(self.INDENT)) + len(self.INDENT)
+                elif lines_modified:
+                    # user SHIFT + Tab, decrease selection positions
+                    if start_lineco not in (0, 1) and first_modified:
+                        # only decrease start position if not in first column
+                        select_start_ppos -= len(self.INDENT)
+                    select_end_ppos -= lines_modified * len(self.INDENT)
+
+                # apply updated selection and restore previous scroll position
+                self.set_selection(select_start_ppos, select_end_ppos, len(self.toPlainText()))
+                self.verticalScrollBar().setSliderPosition(scroll_ppos)
+        else:
+            super(CapaExplorerRulgenPreview, self).keyPressEvent(e)
+
+    def count_previous_lines_from_block(self, block):
+        """calculate number of lines preceding block"""
+        count = 0
+        while True:
+            block = block.previous()
+            if not block.isValid():
+                break
+            count += block.lineCount()
+        return count
+
+    def set_selection(self, start, end, max):
+        """set text selection"""
+        cursor = self.textCursor()
+        cursor.setPosition(start)
+        cursor.setPosition(end if end < max else max, QtGui.QTextCursor.KeepAnchor)
+        self.setTextCursor(cursor)
+
+
+class CapaExplorerRulgenEditor(QtWidgets.QTreeWidget):
+
+    updated = QtCore.pyqtSignal()
+
+    def __init__(self, preview, parent=None):
+        """ """
+        super(CapaExplorerRulgenEditor, self).__init__(parent)
+
+        self.preview = preview
+
+        self.setHeaderLabels(["Feature", "Description", "Comment"])
+        self.header().setSectionResizeMode(QtWidgets.QHeaderView.ResizeToContents)
+        self.header().setStretchLastSection(False)
+        self.setExpandsOnDoubleClick(False)
+        self.setEditTriggers(QtWidgets.QAbstractItemView.NoEditTriggers)
+        self.setContextMenuPolicy(QtCore.Qt.CustomContextMenu)
+        self.setSelectionMode(QtWidgets.QAbstractItemView.ExtendedSelection)
+        self.setStyleSheet("QTreeView::item {padding-right: 15 px;padding-bottom: 2 px;}")
+
+        # enable drag and drop
+        self.setDragEnabled(True)
+        self.setAcceptDrops(True)
+        self.setDragDropMode(QtWidgets.QAbstractItemView.InternalMove)
+
+        # connect slots
+        self.itemChanged.connect(self.slot_item_changed)
+        self.customContextMenuRequested.connect(self.slot_custom_context_menu_requested)
+        self.itemDoubleClicked.connect(self.slot_item_double_clicked)
+
+        self.root = None
+        self.reset_view()
+
+        self.is_editing = False
+
+    @staticmethod
+    def get_column_feature_index():
+        """ """
+        return 0
+
+    @staticmethod
+    def get_column_description_index():
+        """ """
+        return 1
+
+    @staticmethod
+    def get_column_comment_index():
+        """ """
+        return 2
+
+    @staticmethod
+    def get_node_type_expression():
+        """ """
+        return 0
+
+    @staticmethod
+    def get_node_type_feature():
+        """ """
+        return 1
+
+    @staticmethod
+    def get_node_type_comment():
+        """ """
+        return 2
+
+    def dragMoveEvent(self, e):
+        """ """
+        super(CapaExplorerRulgenEditor, self).dragMoveEvent(e)
+
+    def dragEventEnter(self, e):
+        """ """
+        super(CapaExplorerRulgenEditor, self).dragEventEnter(e)
+
+    def dropEvent(self, e):
+        """ """
+        if not self.indexAt(e.pos()).isValid():
+            return
+
+        super(CapaExplorerRulgenEditor, self).dropEvent(e)
+
+        # self.prune_expressions()
+        self.update_preview()
+        self.expandAll()
+
+    def reset_view(self):
+        """ """
+        self.root = None
+        self.clear()
+
+    def slot_item_changed(self, item, column):
+        """ """
+        if self.is_editing:
+            self.update_preview()
+            self.is_editing = False
+
+    def slot_remove_selected(self, action):
+        """ """
+        for o in self.selectedItems():
+            if o == self.root:
+                self.takeTopLevelItem(self.indexOfTopLevelItem(o))
+                self.root = None
+                continue
+            o.parent().removeChild(o)
+
+    def slot_nest_features(self, action):
+        """ """
+        # create a new parent under root node, by default; new node added last position in tree
+        new_parent = self.new_expression_node(self.root, (action.data()[0], ""))
+
+        if "basic block" in action.data()[0]:
+            # add default child expression when nesting under basic block
+            new_parent.setExpanded(True)
+            new_parent = self.new_expression_node(new_parent, ("- or:", ""))
+
+        for o in self.get_features(selected=True):
+            # take child from its parent by index, add to new parent
+            new_parent.addChild(o.parent().takeChild(o.parent().indexOfChild(o)))
+
+        # ensure new parent expanded
+        new_parent.setExpanded(True)
+
+    def slot_edit_expression(self, action):
+        """ """
+        expression, o = action.data()
+        if "basic block" in expression and "basic block" not in o.text(
+            CapaExplorerRulgenEditor.get_column_feature_index()
+        ):
+            # current expression is "basic block", and not changing to "basic block" expression
+            children = o.takeChildren()
+            new_parent = self.new_expression_node(o, ("- or:", ""))
+            for child in children:
+                new_parent.addChild(child)
+            new_parent.setExpanded(True)
+        o.setText(CapaExplorerRulgenEditor.get_column_feature_index(), expression)
+
+    def slot_clear_all(self, action):
+        """ """
+        self.reset_view()
+
+    def slot_custom_context_menu_requested(self, pos):
+        """ """
+        if not self.indexAt(pos).isValid():
+            # user selected invalid index
+            self.load_custom_context_menu_invalid_index(pos)
+        elif self.itemAt(pos).capa_type == CapaExplorerRulgenEditor.get_node_type_expression():
+            # user selected expression node
+            self.load_custom_context_menu_expression(pos)
+        else:
+            # user selected feature node
+            self.load_custom_context_menu_feature(pos)
+
+        self.update_preview()
+
+    def slot_item_double_clicked(self, o, column):
+        """ """
+        if column in (
+            CapaExplorerRulgenEditor.get_column_comment_index(),
+            CapaExplorerRulgenEditor.get_column_description_index(),
+        ):
+            o.setFlags(o.flags() | QtCore.Qt.ItemIsEditable)
+            self.editItem(o, column)
+            o.setFlags(o.flags() & ~QtCore.Qt.ItemIsEditable)
+            self.is_editing = True
+
+    def update_preview(self):
+        """ """
+        rule_text = self.preview.toPlainText()
+
+        if -1 != rule_text.find("features:"):
+            rule_text = rule_text[: rule_text.find("features:") + len("features:")]
+            rule_text += "\n"
+        else:
+            rule_text = rule_text.rstrip()
+            rule_text += "\n  features:\n"
+
+        for o in iterate_tree(self):
+            feature, description, comment = map(lambda o: o.strip(), tuple(o.text(i) for i in range(3)))
+            rule_text += parse_node_for_feature(feature, description, comment, calc_item_depth(o))
+
+        # FIXME we avoid circular update by disabling signals when updating
+        # the preview. Preferably we would refactor the code to avoid this
+        # in the first place
+        self.preview.blockSignals(True)
+        self.preview.setPlainText(rule_text)
+        self.preview.blockSignals(False)
+
+        # emit signal so views can update
+        self.updated.emit()
+
+    def load_custom_context_menu_invalid_index(self, pos):
+        """ """
+        actions = (("Remove all", (), self.slot_clear_all),)
+
+        menu = build_context_menu(self.parent(), actions)
+        menu.exec_(self.viewport().mapToGlobal(pos))
+
+    def load_custom_context_menu_feature(self, pos):
+        """ """
+        actions = (("Remove selection", (), self.slot_remove_selected),)
+
+        sub_actions = (
+            ("and", ("- and:",), self.slot_nest_features),
+            ("or", ("- or:",), self.slot_nest_features),
+            ("not", ("- not:",), self.slot_nest_features),
+            ("optional", ("- optional:",), self.slot_nest_features),
+            ("basic block", ("- basic block:",), self.slot_nest_features),
+        )
+
+        # build submenu with modify actions
+        sub_menu = build_context_menu(self.parent(), sub_actions)
+        sub_menu.setTitle("Nest feature%s" % ("" if len(tuple(self.get_features(selected=True))) == 1 else "s"))
+
+        # build main menu with submenu + main actions
+        menu = build_context_menu(self.parent(), (sub_menu,) + actions)
+
+        menu.exec_(self.viewport().mapToGlobal(pos))
+
+    def load_custom_context_menu_expression(self, pos):
+        """ """
+        actions = (("Remove expression", (), self.slot_remove_selected),)
+
+        sub_actions = (
+            ("and", ("- and:", self.itemAt(pos)), self.slot_edit_expression),
+            ("or", ("- or:", self.itemAt(pos)), self.slot_edit_expression),
+            ("not", ("- not:", self.itemAt(pos)), self.slot_edit_expression),
+            ("optional", ("- optional:", self.itemAt(pos)), self.slot_edit_expression),
+            ("basic block", ("- basic block:", self.itemAt(pos)), self.slot_edit_expression),
+        )
+
+        # build submenu with modify actions
+        sub_menu = build_context_menu(self.parent(), sub_actions)
+        sub_menu.setTitle("Modify")
+
+        # build main menu with submenu + main actions
+        menu = build_context_menu(self.parent(), (sub_menu,) + actions)
+
+        menu.exec_(self.viewport().mapToGlobal(pos))
+
+    def style_expression_node(self, o):
+        """ """
+        font = QtGui.QFont()
+        font.setBold(True)
+
+        o.setFont(CapaExplorerRulgenEditor.get_column_feature_index(), font)
+
+    def style_feature_node(self, o):
+        """ """
+        font = QtGui.QFont()
+        brush = QtGui.QBrush()
+
+        font.setFamily("Courier")
+        font.setWeight(QtGui.QFont.Medium)
+        brush.setColor(QtGui.QColor(*COLOR_GREEN_RGB))
+
+        o.setFont(CapaExplorerRulgenEditor.get_column_feature_index(), font)
+        o.setForeground(CapaExplorerRulgenEditor.get_column_feature_index(), brush)
+
+    def style_comment_node(self, o):
+        """ """
+        font = QtGui.QFont()
+        font.setBold(True)
+        font.setFamily("Courier")
+
+        o.setFont(CapaExplorerRulgenEditor.get_column_feature_index(), font)
+
+    def set_expression_node(self, o):
+        """ """
+        setattr(o, "capa_type", CapaExplorerRulgenEditor.get_node_type_expression())
+        self.style_expression_node(o)
+
+    def set_feature_node(self, o):
+        """ """
+        setattr(o, "capa_type", CapaExplorerRulgenEditor.get_node_type_feature())
+        o.setFlags(o.flags() & ~QtCore.Qt.ItemIsDropEnabled)
+        self.style_feature_node(o)
+
+    def set_comment_node(self, o):
+        """ """
+        setattr(o, "capa_type", CapaExplorerRulgenEditor.get_node_type_comment())
+        o.setFlags(o.flags() & ~QtCore.Qt.ItemIsDropEnabled)
+
+        self.style_comment_node(o)
+
+    def new_expression_node(self, parent, values=()):
+        """ """
+        o = QtWidgets.QTreeWidgetItem(parent)
+        self.set_expression_node(o)
+        for (i, v) in enumerate(values):
+            o.setText(i, v)
+        return o
+
+    def new_feature_node(self, parent, values=()):
+        """ """
+        o = QtWidgets.QTreeWidgetItem(parent)
+        self.set_feature_node(o)
+        for (i, v) in enumerate(values):
+            o.setText(i, v)
+        return o
+
+    def new_comment_node(self, parent, values=()):
+        """ """
+        o = QtWidgets.QTreeWidgetItem(parent)
+        self.set_comment_node(o)
+        for (i, v) in enumerate(values):
+            o.setText(i, v)
+        return o
+
+    def update_features(self, features):
+        """ """
+        if not self.root:
+            # root node does not exist, create default node, set expanded
+            self.root = self.new_expression_node(self, ("- or:", ""))
+
+        # build feature counts
+        counted = list(zip(Counter(features).keys(), Counter(features).values()))
+
+        # single features
+        for (k, v) in filter(lambda t: t[1] == 1, counted):
+            if isinstance(k, (capa.features.String,)):
+                value = '"%s"' % capa.features.escape_string(k.get_value_str())
+            else:
+                value = k.get_value_str()
+            self.new_feature_node(self.root, ("- %s: %s" % (k.name.lower(), value), ""))
+
+        # n > 1 features
+        for (k, v) in filter(lambda t: t[1] > 1, counted):
+            if k.value:
+                if isinstance(k, (capa.features.String,)):
+                    value = '"%s"' % capa.features.escape_string(k.get_value_str())
+                else:
+                    value = k.get_value_str()
+                display = "- count(%s(%s)): %d" % (k.name.lower(), value, v)
+            else:
+                display = "- count(%s): %d" % (k.name.lower(), v)
+            self.new_feature_node(self.root, (display, ""))
+
+        self.expandAll()
+        self.update_preview()
+
+    def load_features_from_yaml(self, rule_text, update_preview=False):
+        """ """
+
+        def add_node(parent, node):
+            if node.text(0).startswith("description:"):
+                if parent.childCount():
+                    parent.child(parent.childCount() - 1).setText(1, node.text(0).lstrip("description:").lstrip())
+                else:
+                    parent.setText(1, node.text(0).lstrip("description:").lstrip())
+            elif node.text(0).startswith("- description:"):
+                parent.setText(1, node.text(0).lstrip("- description:").lstrip())
+            else:
+                parent.addChild(node)
+
+        def build(parent, nodes):
+            if nodes:
+                child_lvl = nodes[0].capa_level
+                while nodes:
+                    node = nodes.pop(0)
+                    if node.capa_level == child_lvl:
+                        add_node(parent, node)
+                    elif node.capa_level > child_lvl:
+                        nodes.insert(0, node)
+                        build(parent.child(parent.childCount() - 1), nodes)
+                    else:
+                        parent = parent.parent() if parent.parent() else parent
+                        add_node(parent, node)
+
+        self.reset_view()
+
+        # check for lack of features block
+        if -1 == rule_text.find("features:"):
+            return
+
+        rule_features = rule_text[rule_text.find("features:") + len("features:") :].strip()
+        rule_nodes = list(yaml_to_nodes(rule_features))
+
+        # check for lack of nodes
+        if not rule_nodes:
+            return
+
+        for o in rule_nodes:
+            (self.set_expression_node, self.set_feature_node, self.set_comment_node)[o.capa_type](o)
+
+        self.root = rule_nodes.pop(0)
+        self.addTopLevelItem(self.root)
+
+        if update_preview:
+            self.preview.blockSignals(True)
+            self.preview.setPlainText(rule_text)
+            self.preview.blockSignals(False)
+
+        build(self.root, rule_nodes)
+
+        self.expandAll()
+
+    def get_features(self, selected=False, ignore=()):
+        """ """
+        for feature in filter(
+            lambda o: o.capa_type
+            in (CapaExplorerRulgenEditor.get_node_type_feature(), CapaExplorerRulgenEditor.get_node_type_comment()),
+            tuple(iterate_tree(self)),
+        ):
+            if feature in ignore:
+                continue
+            if selected and not feature.isSelected():
+                continue
+            yield feature
+
+    def get_expressions(self, selected=False, ignore=()):
+        """ """
+        for expression in filter(
+            lambda o: o.capa_type == CapaExplorerRulgenEditor.get_node_type_expression(), tuple(iterate_tree(self))
+        ):
+            if expression in ignore:
+                continue
+            if selected and not expression.isSelected():
+                continue
+            yield expression
+
+
+class CapaExplorerRulegenFeatures(QtWidgets.QTreeWidget):
+    def __init__(self, editor, parent=None):
+        """ """
+        super(CapaExplorerRulegenFeatures, self).__init__(parent)
+
+        self.parent_items = {}
+        self.editor = editor
+
+        self.setHeaderLabels(["Feature", "Virtual Address"])
+        self.header().setSectionResizeMode(QtWidgets.QHeaderView.ResizeToContents)
+        self.setStyleSheet("QTreeView::item {padding-right: 15 px;padding-bottom: 2 px;}")
+
+        self.setExpandsOnDoubleClick(False)
+        self.setContextMenuPolicy(QtCore.Qt.CustomContextMenu)
+        self.setSelectionMode(QtWidgets.QAbstractItemView.ExtendedSelection)
+
+        # connect slots
+        self.itemDoubleClicked.connect(self.slot_item_double_clicked)
+        self.customContextMenuRequested.connect(self.slot_custom_context_menu_requested)
+
+        self.reset_view()
+
+    @staticmethod
+    def get_column_feature_index():
+        """ """
+        return 0
+
+    @staticmethod
+    def get_column_address_index():
+        """ """
+        return 1
+
+    @staticmethod
+    def get_node_type_parent():
+        """ """
+        return 0
+
+    @staticmethod
+    def get_node_type_leaf():
+        """ """
+        return 1
+
+    def reset_view(self):
+        """ """
+        self.clear()
+
+    def slot_add_selected_features(self, action):
+        """ """
+        selected = [item.data(0, 0x100) for item in self.selectedItems()]
+        if selected:
+            self.editor.update_features(selected)
+
+    def slot_custom_context_menu_requested(self, pos):
+        """ """
+        actions = []
+        action_add_features_fmt = ""
+
+        selected_items_count = len(self.selectedItems())
+        if selected_items_count == 0:
+            return
+
+        if selected_items_count == 1:
+            action_add_features_fmt = "Add feature"
+        else:
+            action_add_features_fmt = "Add %d features" % selected_items_count
+
+        actions.append((action_add_features_fmt, (), self.slot_add_selected_features))
+
+        menu = build_context_menu(self.parent(), actions)
+        menu.exec_(self.viewport().mapToGlobal(pos))
+
+    def slot_item_double_clicked(self, o, column):
+        """ """
+        if column == CapaExplorerRulegenFeatures.get_column_address_index() and o.text(column):
+            idc.jumpto(int(o.text(column), 0x10))
+        elif o.capa_type == CapaExplorerRulegenFeatures.get_node_type_leaf():
+            self.editor.update_features([o.data(0, 0x100)])
+
+    def show_all_items(self):
+        """ """
+        for o in iterate_tree(self):
+            o.setHidden(False)
+            o.setExpanded(False)
+
+    def filter_items_by_text(self, text):
+        """ """
+        if text:
+            for o in iterate_tree(self):
+                data = o.data(0, 0x100)
+                if data:
+                    to_match = data.get_value_str()
+                    if not to_match or text.lower() not in to_match.lower():
+                        o.setHidden(True)
+                        continue
+                o.setHidden(False)
+                o.setExpanded(True)
+        else:
+            self.show_all_items()
+
+    def style_parent_node(self, o):
+        """ """
+        font = QtGui.QFont()
+        font.setBold(True)
+
+        o.setFont(CapaExplorerRulegenFeatures.get_column_feature_index(), font)
+
+    def style_leaf_node(self, o):
+        """ """
+        font = QtGui.QFont("Courier", weight=QtGui.QFont.Bold)
+        brush = QtGui.QBrush()
+
+        o.setFont(CapaExplorerRulegenFeatures.get_column_feature_index(), font)
+        o.setFont(CapaExplorerRulegenFeatures.get_column_address_index(), font)
+
+        brush.setColor(QtGui.QColor(*COLOR_GREEN_RGB))
+        o.setForeground(CapaExplorerRulegenFeatures.get_column_feature_index(), brush)
+
+        brush.setColor(QtGui.QColor(*COLOR_BLUE_RGB))
+        o.setForeground(CapaExplorerRulegenFeatures.get_column_address_index(), brush)
+
+    def set_parent_node(self, o):
+        """ """
+        o.setFlags(o.flags() & ~QtCore.Qt.ItemIsSelectable)
+        setattr(o, "capa_type", CapaExplorerRulegenFeatures.get_node_type_parent())
+        self.style_parent_node(o)
+
+    def set_leaf_node(self, o):
+        """ """
+        setattr(o, "capa_type", CapaExplorerRulegenFeatures.get_node_type_leaf())
+        self.style_leaf_node(o)
+
+    def new_parent_node(self, parent, data, feature=None):
+        """ """
+        o = QtWidgets.QTreeWidgetItem(parent)
+
+        self.set_parent_node(o)
+        for (i, v) in enumerate(data):
+            o.setText(i, v)
+        if feature:
+            o.setData(0, 0x100, feature)
+
+        return o
+
+    def new_leaf_node(self, parent, data, feature=None):
+        """ """
+        o = QtWidgets.QTreeWidgetItem(parent)
+
+        self.set_leaf_node(o)
+        for (i, v) in enumerate(data):
+            o.setText(i, v)
+        if feature:
+            o.setData(0, 0x100, feature)
+
+        return o
+
+    def load_features(self, file_features, func_features={}):
+        """ """
+        self.parse_features_for_tree(self.new_parent_node(self, ("File Scope",)), file_features)
+        if func_features:
+            self.parse_features_for_tree(self.new_parent_node(self, ("Function/Basic Block Scope",)), func_features)
+
+    def parse_features_for_tree(self, parent, features):
+        """ """
+        self.parent_items = {}
+
+        def format_address(e):
+            return "%X" % e if e else ""
+
+        def format_feature(feature):
+            """ """
+            name = feature.name.lower()
+            value = feature.get_value_str()
+            if isinstance(feature, (capa.features.String,)):
+                value = '"%s"' % capa.features.escape_string(value)
+            return "%s(%s)" % (name, value)
+
+        for (feature, eas) in sorted(features.items(), key=lambda k: sorted(k[1])):
+            if isinstance(feature, capa.features.basicblock.BasicBlock):
+                # filter basic blocks for now, we may want to add these back in some time
+                # in the future
+                continue
+
+            # level 0
+            if type(feature) not in self.parent_items:
+                self.parent_items[type(feature)] = self.new_parent_node(parent, (feature.name.lower(),))
+
+            # level 1
+            if feature not in self.parent_items:
+                if len(eas) > 1:
+                    self.parent_items[feature] = self.new_parent_node(
+                        self.parent_items[type(feature)], (format_feature(feature),), feature=feature
+                    )
+                else:
+                    self.parent_items[feature] = self.new_leaf_node(
+                        self.parent_items[type(feature)], (format_feature(feature),), feature=feature
+                    )
+
+            # level n > 1
+            if len(eas) > 1:
+                for ea in sorted(eas):
+                    self.new_leaf_node(
+                        self.parent_items[feature], (format_feature(feature), format_address(ea)), feature=feature
+                    )
+            else:
+                ea = eas.pop()
+                for (i, v) in enumerate((format_feature(feature), format_address(ea))):
+                    self.parent_items[feature].setText(i, v)
+                self.parent_items[feature].setData(0, 0x100, feature)
+

 class CapaExplorerQtreeView(QtWidgets.QTreeView):
    """tree view used to display hierarchical capa results
--- a/capa/main.py
+++ b/capa/main.py
@@ -32,7 +32,9 @@ import capa.features.extractors
 from capa.helpers import oint, get_file_taste

 RULES_PATH_DEFAULT_STRING = "(embedded rules)"
-SUPPORTED_FILE_MAGIC = set(["MZ"])
+SUPPORTED_FILE_MAGIC = set([b"MZ"])
+BACKEND_VIV = "vivisect"
+BACKEND_SMDA = "smda"


 logger = logging.getLogger("capa")
@@ -280,6 +282,8 @@ def get_workspace(path, format, should_save=True):
        vw = get_shellcode_vw(path, arch="i386", should_save=should_save)
    elif format == "sc64":
        vw = get_shellcode_vw(path, arch="amd64", should_save=should_save)
+    else:
+        raise ValueError("unexpected format: " + format)
    logger.debug("%s", get_meta_str(vw))
    return vw

@@ -303,29 +307,43 @@ class UnsupportedRuntimeError(RuntimeError):
    pass


-def get_extractor_py3(path, format, disable_progress=False):
-    from smda.SmdaConfig import SmdaConfig
-    from smda.Disassembler import Disassembler
+def get_extractor_py3(path, format, backend, disable_progress=False):
+    if backend == "smda":
+        from smda.SmdaConfig import SmdaConfig
+        from smda.Disassembler import Disassembler

-    import capa.features.extractors.smda
+        import capa.features.extractors.smda

-    smda_report = None
-    with halo.Halo(text="analyzing program", spinner="simpleDots", stream=sys.stderr, enabled=not disable_progress):
-        config = SmdaConfig()
-        config.STORE_BUFFER = True
-        smda_disasm = Disassembler(config)
-        smda_report = smda_disasm.disassembleFile(path)
+        smda_report = None
+        with halo.Halo(text="analyzing program", spinner="simpleDots", stream=sys.stderr, enabled=not disable_progress):
+            config = SmdaConfig()
+            config.STORE_BUFFER = True
+            smda_disasm = Disassembler(config)
+            smda_report = smda_disasm.disassembleFile(path)

-    return capa.features.extractors.smda.SmdaFeatureExtractor(smda_report, path)
+        return capa.features.extractors.smda.SmdaFeatureExtractor(smda_report, path)
+    else:
+        import capa.features.extractors.viv
+
+        with halo.Halo(text="analyzing program", spinner="simpleDots", stream=sys.stderr, enabled=not disable_progress):
+            vw = get_workspace(path, format, should_save=False)
+
+            try:
+                vw.saveWorkspace()
+            except IOError:
+                # see #168 for discussion around how to handle non-writable directories
+                logger.info("source directory is not writable, won't save intermediate workspace")
+
+        return capa.features.extractors.viv.VivisectFeatureExtractor(vw, path)


-def get_extractor(path, format, disable_progress=False):
+def get_extractor(path, format, backend, disable_progress=False):
    """
    raises:
      UnsupportedFormatError:
    """
    if sys.version_info >= (3, 0):
-        return get_extractor_py3(path, format, disable_progress=disable_progress)
+        return get_extractor_py3(path, format, backend, disable_progress=disable_progress)
    else:
        return get_extractor_py2(path, format, disable_progress=disable_progress)

@@ -361,8 +379,8 @@ def get_rules(rule_path, disable_progress=False):

            for file in files:
                if not file.endswith(".yml"):
-                    if not (file.endswith(".md") or file.endswith(".git") or file.endswith(".txt")):
-                        # expect to see readme.md, format.md, and maybe a .git directory
+                    if not (file.startswith(".git") or file.endswith((".git", ".md", ".txt"))):
+                        # expect to see .git* files, readme.md, format.md, and maybe a .git directory
                        # other things maybe are rules, but are mis-named.
                        logger.warning("skipping non-.yml file: %s", file)
                    continue
@@ -428,19 +446,162 @@ def collect_metadata(argv, sample_path, rules_path, format, extractor):
    }


+def install_common_args(parser, wanted=None):
+    """
+    register a common set of command line arguments for re-use by main & scripts.
+    these are things like logging/coloring/etc.
+    also enable callers to opt-in to common arguments, like specifying the input sample.
+
+    this routine lets many script use the same language for cli arguments.
+    see `handle_common_args` to do common configuration.
+
+    args:
+      parser (argparse.ArgumentParser): a parser to update in place, adding common arguments.
+      wanted (Set[str]): collection of arguments to opt-into, including:
+        - "sample": required positional argument to input file.
+        - "format": flag to override file format.
+        - "backend": flag to override analysis backend under py3.
+        - "rules": flag to override path to capa rules.
+        - "tag": flag to override/specify which rules to match.
+    """
+    if wanted is None:
+        wanted = set()
+
+    #
+    # common arguments that all scripts will have
+    #
+
+    parser.add_argument("--version", action="version", version="%(prog)s {:s}".format(capa.version.__version__))
+    parser.add_argument(
+        "-v", "--verbose", action="store_true", help="enable verbose result document (no effect with --json)"
+    )
+    parser.add_argument(
+        "-vv", "--vverbose", action="store_true", help="enable very verbose result document (no effect with --json)"
+    )
+    parser.add_argument("-d", "--debug", action="store_true", help="enable debugging output on STDERR")
+    parser.add_argument("-q", "--quiet", action="store_true", help="disable all output but errors")
+    parser.add_argument(
+        "--color",
+        type=str,
+        choices=("auto", "always", "never"),
+        default="auto",
+        help="enable ANSI color codes in results, default: only during interactive session",
+    )
+
+    #
+    # arguments that may be opted into:
+    #
+    #   - sample
+    #   - format
+    #   - rules
+    #   - tag
+    #
+
+    if "sample" in wanted:
+        if sys.version_info >= (3, 0):
+            parser.add_argument(
+                # Python 3 str handles non-ASCII arguments correctly
+                "sample",
+                type=str,
+                help="path to sample to analyze",
+            )
+        else:
+            parser.add_argument(
+                # in #328 we noticed that the sample path is not handled correctly if it contains non-ASCII characters
+                # https://stackoverflow.com/a/22947334/ offers a solution and decoding using getfilesystemencoding works
+                # in our testing, however other sources suggest `sys.stdin.encoding` (https://stackoverflow.com/q/4012571/)
+                "sample",
+                type=lambda s: s.decode(sys.getfilesystemencoding()),
+                help="path to sample to analyze",
+            )
+
+    if "format" in wanted:
+        formats = [
+            ("auto", "(default) detect file type automatically"),
+            ("pe", "Windows PE file"),
+            ("sc32", "32-bit shellcode"),
+            ("sc64", "64-bit shellcode"),
+            ("freeze", "features previously frozen by capa"),
+        ]
+        format_help = ", ".join(["%s: %s" % (f[0], f[1]) for f in formats])
+        parser.add_argument(
+            "-f",
+            "--format",
+            choices=[f[0] for f in formats],
+            default="auto",
+            help="select sample format, %s" % format_help,
+        )
+
+    if "backend" in wanted and sys.version_info >= (3, 0):
+        parser.add_argument(
+            "-b",
+            "--backend",
+            type=str,
+            help="select the backend to use",
+            choices=(BACKEND_VIV, BACKEND_SMDA),
+            default=BACKEND_VIV,
+        )
+
+    if "rules" in wanted:
+        parser.add_argument(
+            "-r",
+            "--rules",
+            type=str,
+            default=RULES_PATH_DEFAULT_STRING,
+            help="path to rule file or directory, use embedded rules by default",
+        )
+
+    if "tag" in wanted:
+        parser.add_argument("-t", "--tag", type=str, help="filter on rule meta field values")
+
+
+def handle_common_args(args):
+    """
+    handle the global config specified by `install_common_args`,
+    such as configuring logging/coloring/etc.
+
+    args:
+      args (argparse.Namespace): parsed arguments that included at least `install_common_args` args.
+    """
+    if args.quiet:
+        logging.basicConfig(level=logging.WARNING)
+        logging.getLogger().setLevel(logging.WARNING)
+    elif args.debug:
+        logging.basicConfig(level=logging.DEBUG)
+        logging.getLogger().setLevel(logging.DEBUG)
+    else:
+        logging.basicConfig(level=logging.INFO)
+        logging.getLogger().setLevel(logging.INFO)
+
+    # disable vivisect-related logging, it's verbose and not relevant for capa users
+    set_vivisect_log_level(logging.CRITICAL)
+
+    # py2 doesn't know about cp65001, which is a variant of utf-8 on windows
+    # tqdm bails when trying to render the progress bar in this setup.
+    # because cp65001 is utf-8, we just map that codepage to the utf-8 codec.
+    # see #380 and: https://stackoverflow.com/a/3259271/87207
+    import codecs
+
+    codecs.register(lambda name: codecs.lookup("utf-8") if name == "cp65001" else None)
+
+    if args.color == "always":
+        colorama.init(strip=False)
+    elif args.color == "auto":
+        # colorama will detect:
+        #  - when on Windows console, and fixup coloring, and
+        #  - when not an interactive session, and disable coloring
+        # renderers should use coloring and assume it will be stripped out if necessary.
+        colorama.init()
+    elif args.color == "never":
+        colorama.init(strip=True)
+    else:
+        raise RuntimeError("unexpected --color value: " + args.color)
+
+
 def main(argv=None):
    if argv is None:
        argv = sys.argv[1:]

-    formats = [
-        ("auto", "(default) detect file type automatically"),
-        ("pe", "Windows PE file"),
-        ("sc32", "32-bit shellcode"),
-        ("sc64", "64-bit shellcode"),
-        ("freeze", "features previously frozen by capa"),
-    ]
-    format_help = ", ".join(["%s: %s" % (f[0], f[1]) for f in formats])
-
    desc = "The FLARE team's open-source tool to identify capabilities in executable files."
    epilog = textwrap.dedent(
        """
@@ -473,65 +634,10 @@ def main(argv=None):
    parser = argparse.ArgumentParser(
        description=desc, epilog=epilog, formatter_class=argparse.RawDescriptionHelpFormatter
    )
-
-    if sys.version_info >= (3, 0):
-        parser.add_argument(
-            # Python 3 str handles non-ASCII arguments correctly
-            "sample",
-            type=str,
-            help="path to sample to analyze",
-        )
-    else:
-        parser.add_argument(
-            # in #328 we noticed that the sample path is not handled correctly if it contains non-ASCII characters
-            # https://stackoverflow.com/a/22947334/ offers a solution and decoding using getfilesystemencoding works
-            # in our testing, however other sources suggest `sys.stdin.encoding` (https://stackoverflow.com/q/4012571/)
-            "sample",
-            type=lambda s: s.decode(sys.getfilesystemencoding()),
-            help="path to sample to analyze",
-        )
-    parser.add_argument("--version", action="version", version="%(prog)s {:s}".format(capa.version.__version__))
-    parser.add_argument(
-        "-r",
-        "--rules",
-        type=str,
-        default=RULES_PATH_DEFAULT_STRING,
-        help="path to rule file or directory, use embedded rules by default",
-    )
-    parser.add_argument(
-        "-f", "--format", choices=[f[0] for f in formats], default="auto", help="select sample format, %s" % format_help
-    )
-    parser.add_argument("-t", "--tag", type=str, help="filter on rule meta field values")
+    install_common_args(parser, {"sample", "format", "backend", "rules", "tag"})
    parser.add_argument("-j", "--json", action="store_true", help="emit JSON instead of text")
-    parser.add_argument(
-        "-v", "--verbose", action="store_true", help="enable verbose result document (no effect with --json)"
-    )
-    parser.add_argument(
-        "-vv", "--vverbose", action="store_true", help="enable very verbose result document (no effect with --json)"
-    )
-    parser.add_argument("-d", "--debug", action="store_true", help="enable debugging output on STDERR")
-    parser.add_argument("-q", "--quiet", action="store_true", help="disable all output but errors")
-    parser.add_argument(
-        "--color",
-        type=str,
-        choices=("auto", "always", "never"),
-        default="auto",
-        help="enable ANSI color codes in results, default: only during interactive session",
-    )
    args = parser.parse_args(args=argv)
-
-    if args.quiet:
-        logging.basicConfig(level=logging.WARNING)
-        logging.getLogger().setLevel(logging.WARNING)
-    elif args.debug:
-        logging.basicConfig(level=logging.DEBUG)
-        logging.getLogger().setLevel(logging.DEBUG)
-    else:
-        logging.basicConfig(level=logging.INFO)
-        logging.getLogger().setLevel(logging.INFO)
-
-    # disable vivisect-related logging, it's verbose and not relevant for capa users
-    set_vivisect_log_level(logging.CRITICAL)
+    handle_common_args(args)

    try:
        taste = get_file_taste(args.sample)
@@ -541,14 +647,6 @@ def main(argv=None):
        logger.error("%s", e.args[0])
        return -1

-    # py2 doesn't know about cp65001, which is a variant of utf-8 on windows
-    # tqdm bails when trying to render the progress bar in this setup.
-    # because cp65001 is utf-8, we just map that codepage to the utf-8 codec.
-    # see #380 and: https://stackoverflow.com/a/3259271/87207
-    import codecs
-
-    codecs.register(lambda name: codecs.lookup("utf-8") if name == "cp65001" else None)
-
    if args.rules == RULES_PATH_DEFAULT_STRING:
        logger.debug("-" * 80)
        logger.debug(" Using default embedded rules.")
@@ -605,7 +703,8 @@ def main(argv=None):
    else:
        format = args.format
        try:
-            extractor = get_extractor(args.sample, args.format, disable_progress=args.quiet)
+            backend = args.backend if sys.version_info > (3, 0) else BACKEND_VIV
+            extractor = get_extractor(args.sample, args.format, backend, disable_progress=args.quiet)
        except UnsupportedFormatError:
            logger.error("-" * 80)
            logger.error(" Input file does not appear to be a PE file.")
@@ -638,19 +737,6 @@ def main(argv=None):
        if not (args.verbose or args.vverbose or args.json):
            return -1

-    if args.color == "always":
-        colorama.init(strip=False)
-    elif args.color == "auto":
-        # colorama will detect:
-        #  - when on Windows console, and fixup coloring, and
-        #  - when not an interactive session, and disable coloring
-        # renderers should use coloring and assume it will be stripped out if necessary.
-        colorama.init()
-    elif args.color == "never":
-        colorama.init(strip=True)
-    else:
-        raise RuntimeError("unexpected --color value: " + args.color)
-
    if args.json:
        print(capa.render.render_json(meta, rules, capabilities))
    elif args.vverbose:
--- a/capa/render/vverbose.py
+++ b/capa/render/vverbose.py
@@ -56,7 +56,11 @@ def render_statement(ostream, match, statement, indent=0):
        child = statement["child"]

        if child[child["type"]]:
-            value = rutils.bold2(child[child["type"]])
+            if child["type"] == "string":
+                value = '"%s"' % capa.features.escape_string(child[child["type"]])
+            else:
+                value = child[child["type"]]
+            value = rutils.bold2(value)
            if child.get("description"):
                ostream.write("count(%s(%s = %s)): " % (child["type"], value, child["description"]))
            else:
@@ -90,6 +94,9 @@ def render_feature(ostream, match, feature, indent=0):
        key = "string"  # render string for regex to mirror the rule source
        value = feature["match"]  # the match provides more information than the value for regex

+    if key == "string":
+        value = '"%s"' % capa.features.escape_string(value)
+
    ostream.write(key)
    ostream.write(": ")

--- a/capa/rules.py
+++ b/capa/rules.py
@@ -736,6 +736,8 @@ class Rule(object):
        # the below regex makes these adjustments and while ugly, we don't have to explore the ruamel.yaml insides
        doc = re.sub(r"!!int '0x-([0-9a-fA-F]+)'", r"-0x\1", doc)

+        # normalize CRLF to LF
+        doc = doc.replace("\r\n", "\n")
        return doc


--- a/capa/version.py
+++ b/capa/version.py
@@ -1 +1 @@
-__version__ = "1.5.1"
+__version__ = "1.6.2"
--- a/doc/img/changelog/tab.gif
+++ b/doc/img/changelog/tab.gif
--- a/doc/img/explorer_condensed.png
+++ b/doc/img/explorer_condensed.png
--- a/doc/img/explorer_expanded.png
+++ b/doc/img/explorer_expanded.png
--- a/doc/img/ida_plugin_example_1.png
+++ b/doc/img/ida_plugin_example_1.png
--- a/doc/img/ida_plugin_example_2.png
+++ b/doc/img/ida_plugin_example_2.png
--- a/doc/img/ida_plugin_intro.gif
+++ b/doc/img/ida_plugin_intro.gif
--- a/doc/img/rulegen_expanded.png
+++ b/doc/img/rulegen_expanded.png
--- a/doc/release.md
+++ b/doc/release.md
@@ -0,0 +1,44 @@
+# Release checklist
+
+- [ ] Ensure all [milestoned issues/PRs](https://github.com/fireeye/capa/milestones) are addressed, or reassign to a new milestone.
+- [ ] Add the `dont merge` label to all PRs that are close to be ready to merge (or merge them if they are ready) in [capa](https://github.com/fireeye/capa/pulls) and [capa-rules](https://github.com/fireeye/capa-rules/pulls).
+- [ ] Ensure the [CI workflow succeeds in master](https://github.com/fireeye/capa/actions/workflows/tests.yml?query=branch%3Amaster).
+- [ ] Ensure that `python scripts/lint.py rules/ --thorough` succeeds (only `missing examples`  offenses are allowed in the nursery).
+- [ ] Review changes
+  - capa https://github.com/fireeye/capa/compare/\<last-release\>...master
+  - capa-rules https://github.com/fireeye/capa-rules/compare/\<last-release>\...master
+- [ ] Update [CHANGELOG.md](https://github.com/fireeye/capa/blob/master/CHANGELOG.md)
+  - Do not forget to add a nice introduction thanking contributors
+  - Remember that we need a major release if we introduce breaking changes
+  - Sections
+    - New Features
+    - New Rules
+    - Bug Fixes
+    - Changes
+    - Development
+    - Raw diffs
+  - Update `Raw diffs` links
+  - Create placeholder for `master (unreleased)` section
+    ```
+    ## master (unreleased)
+
+    ### New Features
+
+    ### New Rules
+
+    ### Bug Fixes
+
+    ### Changes
+
+    ### Development
+
+    ### Raw diffs
+    - [capa <release>...master](https://github.com/fireeye/capa/compare/<release>...master)
+    - [capa-rules <release>...master](https://github.com/fireeye/capa-rules/compare/<release>...master)
+    ```
+- [ ] Update [capa/version.py](https://github.com/fireeye/capa/blob/master/capa/version.py)
+- [ ] Create a PR with the updated [CHANGELOG.md](https://github.com/fireeye/capa/blob/master/CHANGELOG.md) and [capa/version.py](https://github.com/fireeye/capa/blob/master/capa/version.py). Copy this checklist in the PR description.
+- [ ] After PR review, merge the PR and [create the release in GH](https://github.com/fireeye/capa/releases/new) using text from the [CHANGELOG.md](https://github.com/fireeye/capa/blob/master/CHANGELOG.md).
+- [ ] Verify GH actions [upload artifacts](https://github.com/fireeye/capa/releases), [publish to PyPI](https://pypi.org/project/flare-capa) and [create a tag in capa rules](https://github.com/fireeye/capa-rules/tags) upon completion.
+- [ ] [Spread the word](https://twitter.com)
+
--- a/2
+++ b/2
--- a/scripts/bulk-process.py
+++ b/scripts/bulk-process.py
@@ -1,247 +1,220 @@
-#!/usr/bin/env python
-"""
-bulk-process
-
-Invoke capa recursively against a directory of samples
-and emit a JSON document mapping the file paths to their results.
-
-By default, this will use subprocesses for parallelism.
-Use `-n/--parallelism` to change the subprocess count from
- the default of current CPU count.
-Use `--no-mp` to use threads instead of processes,
- which is probably not useful unless you set `--parallelism=1`.
-
-example:
-
-    $ python scripts/bulk-process /tmp/suspicious
-    {
-      "/tmp/suspicious/suspicious.dll_": {
-        "rules": {
-          "encode data using XOR": {
-            "matches": {
-              "268440358": {
-              [...]
-      "/tmp/suspicious/1.dll_": { ... }
-      "/tmp/suspicious/2.dll_": { ... }
-    }
-
-
-usage:
-
-    usage: bulk-process.py [-h] [-r RULES] [-d] [-q] [-n PARALLELISM] [--no-mp]
-                           input
-
-    detect capabilities in programs.
-
-    positional arguments:
-      input                 Path to directory of files to recursively analyze
-
-    optional arguments:
-      -h, --help            show this help message and exit
-      -r RULES, --rules RULES
-                            Path to rule file or directory, use embedded rules by
-                            default
-      -d, --debug           Enable debugging output on STDERR
-      -q, --quiet           Disable all output but errors
-      -n PARALLELISM, --parallelism PARALLELISM
-                            parallelism factor
-      --no-mp               disable subprocesses
-
-Copyright (C) 2020 FireEye, Inc. All Rights Reserved.
-Licensed under the Apache License, Version 2.0 (the "License");
- you may not use this file except in compliance with the License.
-You may obtain a copy of the License at: [package root]/LICENSE.txt
-Unless required by applicable law or agreed to in writing, software distributed under the License
- is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-See the License for the specific language governing permissions and limitations under the License.
-"""
-import sys
-import json
-import logging
-import os.path
-import argparse
-import multiprocessing
-import multiprocessing.pool
-
-import capa
-import capa.main
-import capa.render
-
-logger = logging.getLogger("capa")
-
-
-def get_capa_results(args):
-    """
-    run capa against the file at the given path, using the given rules.
-
-    args is a tuple, containing:
-      rules (capa.rules.RuleSet): the rules to match
-      format (str): the name of the sample file format
-      path (str): the file system path to the sample to process
-
-    args is a tuple because i'm not quite sure how to unpack multiple arguments using `map`.
-
-    returns an dict with two required keys:
-      path (str): the file system path of the sample to process
-      status (str): either "error" or "ok"
-
-    when status == "error", then a human readable message is found in property "error".
-    when status == "ok", then the capa results are found in the property "ok".
-
-    the capa results are a dictionary with the following keys:
-      meta (dict): the meta analysis results
-      capabilities (dict): the matched capabilities and their result objects
-    """
-    rules, format, path = args
-    logger.info("computing capa results for: %s", path)
-    try:
-        extractor = capa.main.get_extractor(path, format, disable_progress=True)
-    except capa.main.UnsupportedFormatError:
-        # i'm 100% sure if multiprocessing will reliably raise exceptions across process boundaries.
-        # so instead, return an object with explicit success/failure status.
-        #
-        # if success, then status=ok, and results found in property "ok"
-        # if error, then status=error, and human readable message in property "error"
-        return {
-            "path": path,
-            "status": "error",
-            "error": "input file does not appear to be a PE file: %s" % path,
-        }
-    except capa.main.UnsupportedRuntimeError:
-        return {
-            "path": path,
-            "status": "error",
-            "error": "unsupported runtime or Python interpreter",
-        }
-    except Exception as e:
-        return {
-            "path": path,
-            "status": "error",
-            "error": "unexpected error: %s" % (e),
-        }
-
-    meta = capa.main.collect_metadata("", path, "", format, extractor)
-    capabilities, counts = capa.main.find_capabilities(rules, extractor, disable_progress=True)
-    meta["analysis"].update(counts)
-
-    return {
-        "path": path,
-        "status": "ok",
-        "ok": {
-            "meta": meta,
-            "capabilities": capabilities,
-        },
-    }
-
-
-def main(argv=None):
-    if argv is None:
-        argv = sys.argv[1:]
-
-        parser = argparse.ArgumentParser(description="detect capabilities in programs.")
-        parser.add_argument("input", type=str, help="Path to directory of files to recursively analyze")
-        parser.add_argument(
-            "-r",
-            "--rules",
-            type=str,
-            default="(embedded rules)",
-            help="Path to rule file or directory, use embedded rules by default",
-        )
-        parser.add_argument("-d", "--debug", action="store_true", help="Enable debugging output on STDERR")
-        parser.add_argument("-q", "--quiet", action="store_true", help="Disable all output but errors")
-        parser.add_argument(
-            "-n", "--parallelism", type=int, default=multiprocessing.cpu_count(), help="parallelism factor"
-        )
-        parser.add_argument("--no-mp", action="store_true", help="disable subprocesses")
-        args = parser.parse_args(args=argv)
-
-        if args.quiet:
-            logging.basicConfig(level=logging.ERROR)
-            logging.getLogger().setLevel(logging.ERROR)
-        elif args.debug:
-            logging.basicConfig(level=logging.DEBUG)
-            logging.getLogger().setLevel(logging.DEBUG)
-        else:
-            logging.basicConfig(level=logging.INFO)
-            logging.getLogger().setLevel(logging.INFO)
-
-        # disable vivisect-related logging, it's verbose and not relevant for capa users
-        capa.main.set_vivisect_log_level(logging.CRITICAL)
-
-        # py2 doesn't know about cp65001, which is a variant of utf-8 on windows
-        # tqdm bails when trying to render the progress bar in this setup.
-        # because cp65001 is utf-8, we just map that codepage to the utf-8 codec.
-        # see #380 and: https://stackoverflow.com/a/3259271/87207
-        import codecs
-
-        codecs.register(lambda name: codecs.lookup("utf-8") if name == "cp65001" else None)
-
-        if args.rules == "(embedded rules)":
-            logger.info("using default embedded rules")
-            logger.debug("detected running from source")
-            args.rules = os.path.join(os.path.dirname(__file__), "..", "rules")
-            logger.debug("default rule path (source method): %s", args.rules)
-        else:
-            logger.info("using rules path: %s", args.rules)
-
-        try:
-            rules = capa.main.get_rules(args.rules)
-            rules = capa.rules.RuleSet(rules)
-            logger.info("successfully loaded %s rules", len(rules))
-        except (IOError, capa.rules.InvalidRule, capa.rules.InvalidRuleSet) as e:
-            logger.error("%s", str(e))
-            return -1
-
-        samples = []
-        for (base, directories, files) in os.walk(args.input):
-            for file in files:
-                samples.append(os.path.join(base, file))
-
-        def pmap(f, args, parallelism=multiprocessing.cpu_count()):
-            """apply the given function f to the given args using subprocesses"""
-            return multiprocessing.Pool(parallelism).imap(f, args)
-
-        def tmap(f, args, parallelism=multiprocessing.cpu_count()):
-            """apply the given function f to the given args using threads"""
-            return multiprocessing.pool.ThreadPool(parallelism).imap(f, args)
-
-        def map(f, args, parallelism=None):
-            """apply the given function f to the given args in the current thread"""
-            for arg in args:
-                yield f(arg)
-
-        if args.no_mp:
-            if args.parallelism == 1:
-                logger.debug("using current thread mapper")
-                mapper = map
-            else:
-                logger.debug("using threading mapper")
-                mapper = tmap
-        else:
-            logger.debug("using process mapper")
-            mapper = pmap
-
-        results = {}
-        for result in mapper(
-            get_capa_results, [(rules, "pe", sample) for sample in samples], parallelism=args.parallelism
-        ):
-            if result["status"] == "error":
-                logger.warning(result["error"])
-            elif result["status"] == "ok":
-                meta = result["ok"]["meta"]
-                capabilities = result["ok"]["capabilities"]
-                # our renderer expects to emit a json document for a single sample
-                # so we deserialize the json document, store it in a larger dict, and we'll subsequently re-encode.
-                results[result["path"]] = json.loads(capa.render.render_json(meta, rules, capabilities))
-            else:
-                raise ValueError("unexpected status: %s" % (result["status"]))
-
-        print(json.dumps(results))
-
-        logger.info("done.")
-
-        return 0
-
-
-if __name__ == "__main__":
-    sys.exit(main())
+#!/usr/bin/env python
+"""
+bulk-process
+
+Invoke capa recursively against a directory of samples
+and emit a JSON document mapping the file paths to their results.
+
+By default, this will use subprocesses for parallelism.
+Use `-n/--parallelism` to change the subprocess count from
+ the default of current CPU count.
+Use `--no-mp` to use threads instead of processes,
+ which is probably not useful unless you set `--parallelism=1`.
+
+example:
+
+    $ python scripts/bulk-process /tmp/suspicious
+    {
+      "/tmp/suspicious/suspicious.dll_": {
+        "rules": {
+          "encode data using XOR": {
+            "matches": {
+              "268440358": {
+              [...]
+      "/tmp/suspicious/1.dll_": { ... }
+      "/tmp/suspicious/2.dll_": { ... }
+    }
+
+
+usage:
+
+    usage: bulk-process.py [-h] [-r RULES] [-d] [-q] [-n PARALLELISM] [--no-mp]
+                           input
+
+    detect capabilities in programs.
+
+    positional arguments:
+      input                 Path to directory of files to recursively analyze
+
+    optional arguments:
+      -h, --help            show this help message and exit
+      -r RULES, --rules RULES
+                            Path to rule file or directory, use embedded rules by
+                            default
+      -d, --debug           Enable debugging output on STDERR
+      -q, --quiet           Disable all output but errors
+      -n PARALLELISM, --parallelism PARALLELISM
+                            parallelism factor
+      --no-mp               disable subprocesses
+
+Copyright (C) 2020 FireEye, Inc. All Rights Reserved.
+Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+You may obtain a copy of the License at: [package root]/LICENSE.txt
+Unless required by applicable law or agreed to in writing, software distributed under the License
+ is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+See the License for the specific language governing permissions and limitations under the License.
+"""
+import sys
+import json
+import logging
+import os.path
+import argparse
+import multiprocessing
+import multiprocessing.pool
+
+import capa
+import capa.main
+import capa.rules
+import capa.render
+
+logger = logging.getLogger("capa")
+
+
+def get_capa_results(args):
+    """
+    run capa against the file at the given path, using the given rules.
+
+    args is a tuple, containing:
+      rules (capa.rules.RuleSet): the rules to match
+      format (str): the name of the sample file format
+      path (str): the file system path to the sample to process
+
+    args is a tuple because i'm not quite sure how to unpack multiple arguments using `map`.
+
+    returns an dict with two required keys:
+      path (str): the file system path of the sample to process
+      status (str): either "error" or "ok"
+
+    when status == "error", then a human readable message is found in property "error".
+    when status == "ok", then the capa results are found in the property "ok".
+
+    the capa results are a dictionary with the following keys:
+      meta (dict): the meta analysis results
+      capabilities (dict): the matched capabilities and their result objects
+    """
+    rules, format, path = args
+    logger.info("computing capa results for: %s", path)
+    try:
+        extractor = capa.main.get_extractor(path, format, capa.main.BACKEND_VIV, disable_progress=True)
+    except capa.main.UnsupportedFormatError:
+        # i'm 100% sure if multiprocessing will reliably raise exceptions across process boundaries.
+        # so instead, return an object with explicit success/failure status.
+        #
+        # if success, then status=ok, and results found in property "ok"
+        # if error, then status=error, and human readable message in property "error"
+        return {
+            "path": path,
+            "status": "error",
+            "error": "input file does not appear to be a PE file: %s" % path,
+        }
+    except capa.main.UnsupportedRuntimeError:
+        return {
+            "path": path,
+            "status": "error",
+            "error": "unsupported runtime or Python interpreter",
+        }
+    except Exception as e:
+        return {
+            "path": path,
+            "status": "error",
+            "error": "unexpected error: %s" % (e),
+        }
+
+    meta = capa.main.collect_metadata("", path, "", format, extractor)
+    capabilities, counts = capa.main.find_capabilities(rules, extractor, disable_progress=True)
+    meta["analysis"].update(counts)
+
+    return {
+        "path": path,
+        "status": "ok",
+        "ok": {
+            "meta": meta,
+            "capabilities": capabilities,
+        },
+    }
+
+
+def main(argv=None):
+    if argv is None:
+        argv = sys.argv[1:]
+
+        parser = argparse.ArgumentParser(description="detect capabilities in programs.")
+        capa.main.install_common_args(parser, wanted={"rules"})
+        parser.add_argument("input", type=str, help="Path to directory of files to recursively analyze")
+        parser.add_argument(
+            "-n", "--parallelism", type=int, default=multiprocessing.cpu_count(), help="parallelism factor"
+        )
+        parser.add_argument("--no-mp", action="store_true", help="disable subprocesses")
+        args = parser.parse_args(args=argv)
+        capa.main.handle_common_args(args)
+
+        if args.rules == "(embedded rules)":
+            logger.info("using default embedded rules")
+            logger.debug("detected running from source")
+            args.rules = os.path.join(os.path.dirname(__file__), "..", "rules")
+            logger.debug("default rule path (source method): %s", args.rules)
+        else:
+            logger.info("using rules path: %s", args.rules)
+
+        try:
+            rules = capa.main.get_rules(args.rules)
+            rules = capa.rules.RuleSet(rules)
+            logger.info("successfully loaded %s rules", len(rules))
+        except (IOError, capa.rules.InvalidRule, capa.rules.InvalidRuleSet) as e:
+            logger.error("%s", str(e))
+            return -1
+
+        samples = []
+        for (base, directories, files) in os.walk(args.input):
+            for file in files:
+                samples.append(os.path.join(base, file))
+
+        def pmap(f, args, parallelism=multiprocessing.cpu_count()):
+            """apply the given function f to the given args using subprocesses"""
+            return multiprocessing.Pool(parallelism).imap(f, args)
+
+        def tmap(f, args, parallelism=multiprocessing.cpu_count()):
+            """apply the given function f to the given args using threads"""
+            return multiprocessing.pool.ThreadPool(parallelism).imap(f, args)
+
+        def map(f, args, parallelism=None):
+            """apply the given function f to the given args in the current thread"""
+            for arg in args:
+                yield f(arg)
+
+        if args.no_mp:
+            if args.parallelism == 1:
+                logger.debug("using current thread mapper")
+                mapper = map
+            else:
+                logger.debug("using threading mapper")
+                mapper = tmap
+        else:
+            logger.debug("using process mapper")
+            mapper = pmap
+
+        results = {}
+        for result in mapper(
+            get_capa_results, [(rules, "pe", sample) for sample in samples], parallelism=args.parallelism
+        ):
+            if result["status"] == "error":
+                logger.warning(result["error"])
+            elif result["status"] == "ok":
+                meta = result["ok"]["meta"]
+                capabilities = result["ok"]["capabilities"]
+                # our renderer expects to emit a json document for a single sample
+                # so we deserialize the json document, store it in a larger dict, and we'll subsequently re-encode.
+                results[result["path"]] = json.loads(capa.render.render_json(meta, rules, capabilities))
+            else:
+                raise ValueError("unexpected status: %s" % (result["status"]))
+
+        print(json.dumps(results))
+
+        logger.info("done.")
+
+        return 0
+
+
+if __name__ == "__main__":
+    sys.exit(main())
--- a/scripts/capa_as_library.py
+++ b/scripts/capa_as_library.py
@@ -6,6 +6,7 @@ import collections
 import capa.main
 import capa.rules
 import capa.engine
+import capa.render
 import capa.features
 import capa.render.utils as rutils
 from capa.engine import *
@@ -191,7 +192,7 @@ def render_dictionary(doc):
 def capa_details(file_path, output_format="dictionary"):

    # extract features and find capabilities
-    extractor = capa.main.get_extractor(file_path, "auto", disable_progress=True)
+    extractor = capa.main.get_extractor(file_path, "auto", capa.main.BACKEND_VIV, disable_progress=True)
    capabilities, counts = capa.main.find_capabilities(rules, extractor, disable_progress=True)

    # collect metadata (used only to make rendering more complete)
--- a/scripts/capafmt.py
+++ b/scripts/capafmt.py
@@ -65,6 +65,8 @@ def main(argv=None):
            return 0
        else:
            logger.info("rule requires reformatting (%s)", rule.name)
+            if "\r\n" in rule.definition:
+                logger.info("please make sure that the file uses LF (\\n) line endings only")
            return 1

    if args.in_place:
--- a/scripts/import-to-ida.py
+++ b/scripts/import-to-ida.py
@@ -31,10 +31,8 @@ See the License for the specific language governing permissions and limitations
 import json
 import logging

-import idc
 import idautils
 import ida_funcs
-import ida_idaapi
 import ida_kernwin

 logger = logging.getLogger("capa")
--- a/scripts/lint.py
+++ b/scripts/lint.py
@@ -25,6 +25,8 @@ import argparse
 import itertools
 import posixpath

+import ruamel.yaml
+
 import capa.main
 import capa.rules
 import capa.engine
@@ -35,7 +37,11 @@ logger = logging.getLogger("capa.lint")


 class Lint(object):
+    WARN = "WARN"
+    FAIL = "FAIL"
+
    name = "lint"
+    level = FAIL
    recommendation = ""

    def check_rule(self, ctx, rule):
@@ -197,7 +203,7 @@ class DoesntMatchExample(Lint):
                continue

            try:
-                extractor = capa.main.get_extractor(path, "auto", disable_progress=True)
+                extractor = capa.main.get_extractor(path, "auto", capa.main.BACKEND_VIV, disable_progress=True)
                capabilities, meta = capa.main.find_capabilities(ctx["rules"], extractor, disable_progress=True)
            except Exception as e:
                logger.error("failed to extract capabilities: %s %s %s", rule.name, path, e)
@@ -279,6 +285,34 @@ class FeatureNegativeNumber(Lint):
        return False


+class FeatureNtdllNtoskrnlApi(Lint):
+    name = "feature api may overlap with ntdll and ntoskrnl"
+    level = Lint.WARN
+    recommendation = (
+        "check if {:s} is exported by both ntdll and ntoskrnl; if true, consider removing {:s} "
+        "module requirement to improve detection"
+    )
+
+    def check_features(self, ctx, features):
+        for feature in features:
+            if isinstance(feature, capa.features.insn.API):
+                modname, _, impname = feature.value.rpartition(".")
+                if modname in ("ntdll", "ntoskrnl"):
+                    self.recommendation = self.recommendation.format(impname, modname)
+                    return True
+        return False
+
+
+class FormatLineFeedEOL(Lint):
+    name = "line(s) end with CRLF (\\r\\n)"
+    recommendation = "convert line endings to LF (\\n) for example using dos2unix"
+
+    def check_rule(self, ctx, rule):
+        if len(rule.definition.split("\r\n")) > 0:
+            return False
+        return True
+
+
 class FormatSingleEmptyLineEOF(Lint):
    name = "EOF format"
    recommendation = "end file with a single empty line"
@@ -298,13 +332,44 @@ class FormatIncorrect(Lint):
        expected = capa.rules.Rule.from_yaml(rule.definition, use_ruamel=True).to_yaml()

        if actual != expected:
-            diff = difflib.ndiff(actual.splitlines(1), expected.splitlines(1))
-            self.recommendation = self.recommendation_template.format("".join(diff))
+            diff = difflib.ndiff(actual.splitlines(1), expected.splitlines(True))
+            recommendation_template = self.recommendation_template
+            if "\r\n" in actual:
+                recommendation_template = (
+                    self.recommendation_template + "\nplease make sure that the file uses LF (\\n) line endings only"
+                )
+            self.recommendation = recommendation_template.format("".join(diff))
            return True

        return False


+class FormatStringQuotesIncorrect(Lint):
+    name = "rule string quotes incorrect"
+
+    def check_rule(self, ctx, rule):
+        events = capa.rules.Rule._get_ruamel_yaml_parser().parse(rule.definition)
+        for key in events:
+            if not (isinstance(key, ruamel.yaml.ScalarEvent) and key.value == "string"):
+                continue
+            value = next(events)  # assume value is next event
+            if not isinstance(value, ruamel.yaml.ScalarEvent):
+                # ignore non-scalar
+                continue
+            if value.value.startswith("/") and value.value.endswith(("/", "/i")):
+                # ignore regex for now
+                continue
+            if value.style is None:
+                # no quotes
+                self.recommendation = 'add double quotes to "%s"' % value.value
+                return True
+            if value.style == "'":
+                # single quote
+                self.recommendation = 'change single quotes to double quotes for "%s"' % value.value
+                return True
+        return False
+
+
 def run_lints(lints, ctx, rule):
    for lint in lints:
        if lint.check_rule(ctx, rule):
@@ -354,10 +419,7 @@ def lint_meta(ctx, rule):
    return run_lints(META_LINTS, ctx, rule)


-FEATURE_LINTS = (
-    FeatureStringTooShort(),
-    FeatureNegativeNumber(),
-)
+FEATURE_LINTS = (FeatureStringTooShort(), FeatureNegativeNumber(), FeatureNtdllNtoskrnlApi())


 def lint_features(ctx, rule):
@@ -366,7 +428,9 @@ def lint_features(ctx, rule):


 FORMAT_LINTS = (
+    FormatLineFeedEOL(),
    FormatSingleEmptyLineEOF(),
+    FormatStringQuotesIncorrect(),
    FormatIncorrect(),
 )

@@ -446,25 +510,28 @@ def lint_rule(ctx, rule):
            )
        )

-        level = "WARN" if is_nursery_rule(rule) else "FAIL"
-
        for violation in violations:
            print(
                "%s  %s: %s: %s"
                % (
                    "    " if is_nursery_rule(rule) else "",
-                    level,
+                    Lint.WARN if is_nursery_rule(rule) else violation.level,
                    violation.name,
                    violation.recommendation,
                )
            )

-    elif len(violations) == 0 and is_nursery_rule(rule):
+        print("")
+
+    lints_failed = any(map(lambda v: v.level == Lint.FAIL, violations))
+
+    if not lints_failed and is_nursery_rule(rule):
        print("")
        print("%s%s" % ("    (nursery) ", rule.name))
-        print("%s  %s: %s: %s" % ("    ", "WARN", "no violations", "Graduate the rule"))
+        print("%s  %s: %s: %s" % ("    ", Lint.WARN, "no lint failures", "Graduate the rule"))
+        print("")

-    return len(violations) > 0 and not is_nursery_rule(rule)
+    return lints_failed and not is_nursery_rule(rule)


 def lint(ctx, rules):
@@ -532,7 +599,8 @@ def main(argv=None):

    samples_path = os.path.join(os.path.dirname(__file__), "..", "tests", "data")

-    parser = argparse.ArgumentParser(description="A program.")
+    parser = argparse.ArgumentParser(description="Lint capa rules.")
+    capa.main.install_common_args(parser, wanted={"tag"})
    parser.add_argument("rules", type=str, help="Path to rules")
    parser.add_argument("--samples", type=str, default=samples_path, help="Path to samples")
    parser.add_argument(
@@ -540,24 +608,15 @@ def main(argv=None):
        action="store_true",
        help="Enable thorough linting - takes more time, but does a better job",
    )
-    parser.add_argument("-t", "--tag", type=str, help="filter on rule meta field values")
-    parser.add_argument("-v", "--verbose", action="store_true", help="Enable debug logging")
-    parser.add_argument("-q", "--quiet", action="store_true", help="Disable all output but errors")
    args = parser.parse_args(args=argv)
+    capa.main.handle_common_args(args)

-    if args.verbose:
-        level = logging.DEBUG
-    elif args.quiet:
-        level = logging.ERROR
+    if args.debug:
+        logging.getLogger("capa").setLevel(logging.DEBUG)
+        logging.getLogger("viv_utils").setLevel(logging.DEBUG)
    else:
-        level = logging.INFO
-
-    logging.basicConfig(level=level)
-    logging.getLogger("capa.lint").setLevel(level)
-
-    capa.main.set_vivisect_log_level(logging.CRITICAL)
-    logging.getLogger("capa").setLevel(logging.CRITICAL)
-    logging.getLogger("viv_utils").setLevel(logging.CRITICAL)
+        logging.getLogger("capa").setLevel(logging.ERROR)
+        logging.getLogger("viv_utils").setLevel(logging.ERROR)

    time0 = time.time()

@@ -593,7 +652,7 @@ def main(argv=None):
    logger.debug("lints ran for ~ %02d:%02dm", min, sec)

    if not did_violate:
-        logger.info("no suggestions, nice!")
+        logger.info("no lints failed, nice!")
        return 0
    else:
        return 1
--- a/scripts/migrate-rules.py
+++ b/scripts/migrate-rules.py
@@ -1,167 +0,0 @@
-#!/usr/bin/env python
-"""
-migrate rules and their namespaces.
-
-example:
-
-    $ python scripts/migrate-rules.py migration.csv ./rules ./new-rules
-
-Copyright (C) 2020 FireEye, Inc. All Rights Reserved.
-Licensed under the Apache License, Version 2.0 (the "License");
- you may not use this file except in compliance with the License.
-You may obtain a copy of the License at: [package root]/LICENSE.txt
-Unless required by applicable law or agreed to in writing, software distributed under the License
- is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-See the License for the specific language governing permissions and limitations under the License.
-"""
-import os
-import csv
-import sys
-import logging
-import os.path
-import argparse
-import collections
-
-import capa.rules
-
-logger = logging.getLogger("migrate-rules")
-
-
-def read_plan(plan_path):
-    with open(plan_path, "rb") as f:
-        return list(
-            csv.DictReader(
-                f,
-                restkey="other",
-                fieldnames=(
-                    "existing path",
-                    "existing name",
-                    "existing rule-category",
-                    "proposed name",
-                    "proposed namespace",
-                    "ATT&CK",
-                    "MBC",
-                    "comment1",
-                ),
-            )
-        )
-
-
-def read_rules(rule_directory):
-    rules = {}
-    for root, dirs, files in os.walk(rule_directory):
-        for file in files:
-            path = os.path.join(root, file)
-            if not path.endswith(".yml"):
-                logger.info("skipping file: %s", path)
-                continue
-
-            rule = capa.rules.Rule.from_yaml_file(path)
-            rules[rule.name] = rule
-
-            if "nursery" in path:
-                rule.meta["capa/nursery"] = True
-    return rules
-
-
-def main(argv=None):
-    if argv is None:
-        argv = sys.argv[1:]
-
-    parser = argparse.ArgumentParser(description="migrate rules.")
-    parser.add_argument("plan", type=str, help="Path to CSV describing migration")
-    parser.add_argument("source", type=str, help="Source directory of rules")
-    parser.add_argument("destination", type=str, help="Destination directory of rules")
-    args = parser.parse_args(args=argv)
-
-    logging.basicConfig(level=logging.INFO)
-    logging.getLogger().setLevel(logging.INFO)
-
-    plan = read_plan(args.plan)
-    logger.info("read %d plan entries", len(plan))
-
-    rules = read_rules(args.source)
-    logger.info("read %d rules", len(rules))
-
-    planned_rules = set([row["existing name"] for row in plan])
-    unplanned_rules = [rule for (name, rule) in rules.items() if name not in planned_rules]
-
-    if unplanned_rules:
-        logger.error("plan does not account for %d rules:" % (len(unplanned_rules)))
-        for rule in unplanned_rules:
-            logger.error("  " + rule.name)
-        return -1
-
-    # pairs of strings (needle, replacement)
-    match_translations = []
-
-    for row in plan:
-        if not row["existing name"]:
-            continue
-
-        rule = rules[row["existing name"]]
-
-        if rule.meta["name"] != row["proposed name"]:
-            logger.info("renaming rule '%s' -> '%s'", rule.meta["name"], row["proposed name"])
-
-            # assume the yaml is formatted like `- match: $rule-name`.
-            # but since its been linted, this should be ok.
-            match_translations.append(("- match: " + rule.meta["name"], "- match: " + row["proposed name"]))
-
-            rule.meta["name"] = row["proposed name"]
-            rule.name = row["proposed name"]
-
-        if "rule-category" in rule.meta:
-            logger.info("deleting rule category '%s'", rule.meta["rule-category"])
-            del rule.meta["rule-category"]
-
-        rule.meta["namespace"] = row["proposed namespace"]
-
-        if row["ATT&CK"] != "n/a" and row["ATT&CK"] != "":
-            tag = row["ATT&CK"]
-            name, _, id = tag.rpartition(" ")
-            tag = "%s [%s]" % (name, id)
-            rule.meta["att&ck"] = [tag]
-
-        if row["MBC"] != "n/a" and row["MBC"] != "":
-            tag = row["MBC"]
-            rule.meta["mbc"] = [tag]
-
-    for rule in rules.values():
-        filename = rule.name
-        filename = filename.lower()
-        filename = filename.replace(" ", "-")
-        filename = filename.replace("(", "")
-        filename = filename.replace(")", "")
-        filename = filename.replace("+", "")
-        filename = filename.replace("/", "")
-        filename = filename + ".yml"
-
-        try:
-            if rule.meta.get("capa/nursery"):
-                directory = os.path.join(args.destination, "nursery")
-            elif rule.meta.get("lib"):
-                directory = os.path.join(args.destination, "lib")
-            else:
-                directory = os.path.join(args.destination, rule.meta.get("namespace"))
-            os.makedirs(directory)
-        except OSError:
-            pass
-        else:
-            logger.info("created namespace: %s", directory)
-
-        path = os.path.join(directory, filename)
-        logger.info("writing rule %s", path)
-
-        doc = rule.to_yaml().decode("utf-8")
-        for (needle, replacement) in match_translations:
-            doc = doc.replace(needle, replacement)
-
-        with open(path, "wb") as f:
-            f.write(doc.encode("utf-8"))
-
-    return 0
-
-
-if __name__ == "__main__":
-    sys.exit(main())
--- a/scripts/show-capabilities-by-function.py
+++ b/scripts/show-capabilities-by-function.py
@@ -63,7 +63,6 @@ import capa.render
 import capa.features
 import capa.render.utils as rutils
 import capa.features.freeze
-import capa.features.extractors.viv
 from capa.helpers import get_file_taste

 logger = logging.getLogger("capa.show-capabilities-by-function")
@@ -111,143 +110,93 @@ def main(argv=None):
    if argv is None:
        argv = sys.argv[1:]

-        formats = [
-            ("auto", "(default) detect file type automatically"),
-            ("pe", "Windows PE file"),
-            ("sc32", "32-bit shellcode"),
-            ("sc64", "64-bit shellcode"),
-            ("freeze", "features previously frozen by capa"),
-        ]
-        format_help = ", ".join(["%s: %s" % (f[0], f[1]) for f in formats])
+    parser = argparse.ArgumentParser(description="detect capabilities in programs.")
+    capa.main.install_common_args(parser, wanted={"format", "sample", "rules", "tag"})
+    args = parser.parse_args(args=argv)
+    capa.main.handle_common_args(args)

-        parser = argparse.ArgumentParser(description="detect capabilities in programs.")
-        parser.add_argument("sample", type=str, help="Path to sample to analyze")
-        parser.add_argument(
-            "-r",
-            "--rules",
-            type=str,
-            default="(embedded rules)",
-            help="Path to rule file or directory, use embedded rules by default",
-        )
-        parser.add_argument("-t", "--tag", type=str, help="Filter on rule meta field values")
-        parser.add_argument("-d", "--debug", action="store_true", help="Enable debugging output on STDERR")
-        parser.add_argument("-q", "--quiet", action="store_true", help="Disable all output but errors")
-        parser.add_argument(
-            "-f",
-            "--format",
-            choices=[f[0] for f in formats],
-            default="auto",
-            help="Select sample format, %s" % format_help,
-        )
-        args = parser.parse_args(args=argv)
+    try:
+        taste = get_file_taste(args.sample)
+    except IOError as e:
+        logger.error("%s", str(e))
+        return -1

-        if args.quiet:
-            logging.basicConfig(level=logging.ERROR)
-            logging.getLogger().setLevel(logging.ERROR)
-        elif args.debug:
-            logging.basicConfig(level=logging.DEBUG)
-            logging.getLogger().setLevel(logging.DEBUG)
-        else:
-            logging.basicConfig(level=logging.INFO)
-            logging.getLogger().setLevel(logging.INFO)
+    if args.rules == "(embedded rules)":
+        logger.info("-" * 80)
+        logger.info(" Using default embedded rules.")
+        logger.info(" To provide your own rules, use the form `capa.exe -r ./path/to/rules/  /path/to/mal.exe`.")
+        logger.info(" You can see the current default rule set here:")
+        logger.info("     https://github.com/fireeye/capa-rules")
+        logger.info("-" * 80)

-        # disable vivisect-related logging, it's verbose and not relevant for capa users
-        capa.main.set_vivisect_log_level(logging.CRITICAL)
+        logger.debug("detected running from source")
+        args.rules = os.path.join(os.path.dirname(__file__), "..", "rules")
+        logger.debug("default rule path (source method): %s", args.rules)
+    else:
+        logger.info("using rules path: %s", args.rules)

+    try:
+        rules = capa.main.get_rules(args.rules)
+        rules = capa.rules.RuleSet(rules)
+        logger.info("successfully loaded %s rules", len(rules))
+        if args.tag:
+            rules = rules.filter_rules_by_meta(args.tag)
+            logger.info("selected %s rules", len(rules))
+    except (IOError, capa.rules.InvalidRule, capa.rules.InvalidRuleSet) as e:
+        logger.error("%s", str(e))
+        return -1
+
+    if (args.format == "freeze") or (args.format == "auto" and capa.features.freeze.is_freeze(taste)):
+        format = "freeze"
+        with open(args.sample, "rb") as f:
+            extractor = capa.features.freeze.load(f.read())
+    else:
+        format = args.format
        try:
-            taste = get_file_taste(args.sample)
-        except IOError as e:
-            logger.error("%s", str(e))
+            extractor = capa.main.get_extractor(args.sample, args.format)
+        except capa.main.UnsupportedFormatError:
+            logger.error("-" * 80)
+            logger.error(" Input file does not appear to be a PE file.")
+            logger.error(" ")
+            logger.error(
+                " capa currently only supports analyzing PE files (or shellcode, when using --format sc32|sc64)."
+            )
+            logger.error(" If you don't know the input file type, you can try using the `file` utility to guess it.")
+            logger.error("-" * 80)
+            return -1
+        except capa.main.UnsupportedRuntimeError:
+            logger.error("-" * 80)
+            logger.error(" Unsupported runtime or Python interpreter.")
+            logger.error(" ")
+            logger.error(" capa supports running under Python 2.7 using Vivisect for binary analysis.")
+            logger.error(" It can also run within IDA Pro, using either Python 2.7 or 3.5+.")
+            logger.error(" ")
+            logger.error(" If you're seeing this message on the command line, please ensure you're running Python 2.7.")
+            logger.error("-" * 80)
            return -1

-        # py2 doesn't know about cp65001, which is a variant of utf-8 on windows
-        # tqdm bails when trying to render the progress bar in this setup.
-        # because cp65001 is utf-8, we just map that codepage to the utf-8 codec.
-        # see #380 and: https://stackoverflow.com/a/3259271/87207
-        import codecs
+    meta = capa.main.collect_metadata(argv, args.sample, args.rules, format, extractor)
+    capabilities, counts = capa.main.find_capabilities(rules, extractor)
+    meta["analysis"].update(counts)

-        codecs.register(lambda name: codecs.lookup("utf-8") if name == "cp65001" else None)
-
-        if args.rules == "(embedded rules)":
-            logger.info("-" * 80)
-            logger.info(" Using default embedded rules.")
-            logger.info(" To provide your own rules, use the form `capa.exe -r ./path/to/rules/  /path/to/mal.exe`.")
-            logger.info(" You can see the current default rule set here:")
-            logger.info("     https://github.com/fireeye/capa-rules")
-            logger.info("-" * 80)
-
-            logger.debug("detected running from source")
-            args.rules = os.path.join(os.path.dirname(__file__), "..", "rules")
-            logger.debug("default rule path (source method): %s", args.rules)
-        else:
-            logger.info("using rules path: %s", args.rules)
-
-        try:
-            rules = capa.main.get_rules(args.rules)
-            rules = capa.rules.RuleSet(rules)
-            logger.info("successfully loaded %s rules", len(rules))
-            if args.tag:
-                rules = rules.filter_rules_by_meta(args.tag)
-                logger.info("selected %s rules", len(rules))
-        except (IOError, capa.rules.InvalidRule, capa.rules.InvalidRuleSet) as e:
-            logger.error("%s", str(e))
+    if capa.main.has_file_limitation(rules, capabilities):
+        # bail if capa encountered file limitation e.g. a packed binary
+        # do show the output in verbose mode, though.
+        if not (args.verbose or args.vverbose or args.json):
            return -1

-        if (args.format == "freeze") or (args.format == "auto" and capa.features.freeze.is_freeze(taste)):
-            format = "freeze"
-            with open(args.sample, "rb") as f:
-                extractor = capa.features.freeze.load(f.read())
-        else:
-            format = args.format
-            try:
-                extractor = capa.main.get_extractor(args.sample, args.format)
-            except capa.main.UnsupportedFormatError:
-                logger.error("-" * 80)
-                logger.error(" Input file does not appear to be a PE file.")
-                logger.error(" ")
-                logger.error(
-                    " capa currently only supports analyzing PE files (or shellcode, when using --format sc32|sc64)."
-                )
-                logger.error(
-                    " If you don't know the input file type, you can try using the `file` utility to guess it."
-                )
-                logger.error("-" * 80)
-                return -1
-            except capa.main.UnsupportedRuntimeError:
-                logger.error("-" * 80)
-                logger.error(" Unsupported runtime or Python interpreter.")
-                logger.error(" ")
-                logger.error(" capa supports running under Python 2.7 using Vivisect for binary analysis.")
-                logger.error(" It can also run within IDA Pro, using either Python 2.7 or 3.5+.")
-                logger.error(" ")
-                logger.error(
-                    " If you're seeing this message on the command line, please ensure you're running Python 2.7."
-                )
-                logger.error("-" * 80)
-                return -1
+    # colorama will detect:
+    #  - when on Windows console, and fixup coloring, and
+    #  - when not an interactive session, and disable coloring
+    # renderers should use coloring and assume it will be stripped out if necessary.
+    colorama.init()
+    doc = capa.render.convert_capabilities_to_result_document(meta, rules, capabilities)
+    print(render_matches_by_function(doc))
+    colorama.deinit()

-        meta = capa.main.collect_metadata(argv, args.sample, args.rules, format, extractor)
-        capabilities, counts = capa.main.find_capabilities(rules, extractor)
-        meta["analysis"].update(counts)
+    logger.info("done.")

-        if capa.main.has_file_limitation(rules, capabilities):
-            # bail if capa encountered file limitation e.g. a packed binary
-            # do show the output in verbose mode, though.
-            if not (args.verbose or args.vverbose or args.json):
-                return -1
-
-        # colorama will detect:
-        #  - when on Windows console, and fixup coloring, and
-        #  - when not an interactive session, and disable coloring
-        # renderers should use coloring and assume it will be stripped out if necessary.
-        colorama.init()
-        doc = capa.render.convert_capabilities_to_result_document(meta, rules, capabilities)
-        print(render_matches_by_function(doc))
-        colorama.deinit()
-
-        logger.info("done.")
-
-        return 0
+    return 0


 if __name__ == "__main__":
--- a/scripts/show-features.py
+++ b/scripts/show-features.py
@@ -71,41 +71,56 @@ import argparse
 import capa.main
 import capa.rules
 import capa.engine
+import capa.helpers
 import capa.features
 import capa.features.freeze
-import capa.features.extractors.viv
+
+logger = logging.getLogger("capa.show-features")


 def main(argv=None):
    if argv is None:
        argv = sys.argv[1:]

-    formats = [
-        ("auto", "(default) detect file type automatically"),
-        ("pe", "Windows PE file"),
-        ("sc32", "32-bit shellcode"),
-        ("sc64", "64-bit shellcode"),
-        ("freeze", "features previously frozen by capa"),
-    ]
-    format_help = ", ".join(["%s: %s" % (f[0], f[1]) for f in formats])
-
    parser = argparse.ArgumentParser(description="Show the features that capa extracts from the given sample")
-    parser.add_argument("sample", type=str, help="Path to sample to analyze")
-    parser.add_argument(
-        "-f", "--format", choices=[f[0] for f in formats], default="auto", help="Select sample format, %s" % format_help
-    )
+    capa.main.install_common_args(parser, wanted={"format", "sample"})
+
    parser.add_argument("-F", "--function", type=lambda x: int(x, 0x10), help="Show features for specific function")
    args = parser.parse_args(args=argv)
+    capa.main.handle_common_args(args)

-    logging.basicConfig(level=logging.INFO)
-    logging.getLogger().setLevel(logging.INFO)
+    try:
+        taste = capa.helpers.get_file_taste(args.sample)
+    except IOError as e:
+        logger.error("%s", str(e))
+        return -1

-    if args.format == "freeze":
+    if (args.format == "freeze") or (args.format == "auto" and capa.features.freeze.is_freeze(taste)):
        with open(args.sample, "rb") as f:
            extractor = capa.features.freeze.load(f.read())
    else:
-        vw = capa.main.get_workspace(args.sample, args.format)
-        extractor = capa.features.extractors.viv.VivisectFeatureExtractor(vw, args.sample)
+        try:
+            extractor = capa.main.get_extractor(args.sample, args.format, capa.main.BACKEND_VIV)
+        except capa.main.UnsupportedFormatError:
+            logger.error("-" * 80)
+            logger.error(" Input file does not appear to be a PE file.")
+            logger.error(" ")
+            logger.error(
+                " capa currently only supports analyzing PE files (or shellcode, when using --format sc32|sc64)."
+            )
+            logger.error(" If you don't know the input file type, you can try using the `file` utility to guess it.")
+            logger.error("-" * 80)
+            return -1
+        except capa.main.UnsupportedRuntimeError:
+            logger.error("-" * 80)
+            logger.error(" Unsupported runtime or Python interpreter.")
+            logger.error(" ")
+            logger.error(" capa supports running under Python 2.7 using Vivisect for binary analysis.")
+            logger.error(" It can also run within IDA Pro, using either Python 2.7 or 3.5+.")
+            logger.error(" ")
+            logger.error(" If you're seeing this message on the command line, please ensure you're running Python 2.7.")
+            logger.error("-" * 80)
+            return -1

    if not args.function:
        for feature, va in extractor.extract_file_features():
@@ -118,15 +133,13 @@ def main(argv=None):

    if args.function:
        if args.format == "freeze":
-            functions = filter(lambda f: f == args.function, functions)
+            functions = tuple(filter(lambda f: f == args.function, functions))
        else:
-            functions = filter(lambda f: f.va == args.function, functions)
+            functions = tuple(filter(lambda f: capa.helpers.oint(f) == args.function, functions))

-            if args.function not in [f.va for f in functions]:
-                print("0x%X not a function, creating it" % args.function)
-                vw.makeFunction(args.function)
-                functions = extractor.get_functions()
-                functions = filter(lambda f: f.va == args.function, functions)
+            if args.function not in [capa.helpers.oint(f) for f in functions]:
+                print("0x%X not a function" % args.function)
+                return -1

        if len(functions) == 0:
            print("0x%X not a function")
@@ -154,7 +167,7 @@ def ida_main():
    functions = extractor.get_functions()

    if function:
-        functions = filter(lambda f: f.start_ea == function, functions)
+        functions = tuple(filter(lambda f: f.start_ea == function, functions))

        if len(functions) == 0:
            print("0x%X not a function" % function)
--- a/scripts/vivisect-py2-vs-py3.sh
+++ b/scripts/vivisect-py2-vs-py3.sh
@@ -0,0 +1,69 @@
+#!/usr/bin/env bash
+
+int() {
+  int=$(bc <<< "scale=0; ($1 + 0.5)/1")
+}
+
+export TIMEFORMAT='%3R'
+threshold_time=90
+threshold_py3_time=60 # Do not warn if it doesn't take at least 1 minute to run
+rm tests/data/*.viv 2>/dev/null
+mkdir results
+for file in tests/data/*
+do
+  file=$(printf %q "$file") # Handle names with white spaces
+  file_name=$(basename $file)
+  echo $file_name
+
+  rm "$file.viv" 2>/dev/null
+  py3_time=$(sh -c "time python3 scripts/show-features.py $file >> results/p3-$file_name.out 2>/dev/null" 2>&1)
+  rm "$file.viv" 2>/dev/null
+  py2_time=$(sh -c "time python2 scripts/show-features.py $file >> results/p2-$file_name.out 2>/dev/null" 2>&1)
+
+  int $py3_time
+  if (($int > $threshold_py3_time))
+  then
+    percentage=$(bc <<< "scale=3; $py2_time/$py3_time*100 + 0.5")
+    int $percentage
+    if (($int < $threshold_py3_time))
+    then
+      echo -n "  SLOWER ($percentage): "
+    fi
+  fi
+  echo "  PY2($py2_time) PY3($py3_time)"
+done
+
+threshold_features=98
+counter=0
+average=0
+results_for() {
+  py3=$(cat "results/p3-$file_name.out" | grep "$1" | wc -l)
+  py2=$(cat "results/p2-$file_name.out" | grep "$1" | wc -l)
+  if (($py2 > 0))
+  then
+    percentage=$(bc <<< "scale=2; 100*$py3/$py2")
+    average=$(bc <<< "scale=2; $percentage + $average")
+    count=$(($count + 1))
+    int $percentage
+    if (($int < $threshold_features))
+    then
+      echo -e "$1: py2($py2) py3($py3) $percentage% - $file_name"
+    fi
+  fi
+}
+
+rm tests/data/*.viv 2>/dev/null
+echo -e '\nRESULTS:'
+for file in tests/data/*
+do
+  file_name=$(basename $file)
+  if test -f "results/p2-$file_name.out"; then
+    results_for 'insn'
+    results_for 'file'
+    results_for 'func'
+    results_for 'bb'
+  fi
+done
+
+average=$(bc <<< "scale=2; $average/$count")
+echo "TOTAL: $average"
--- a/setup.py
+++ b/setup.py
@@ -12,30 +12,32 @@ import sys
 import setuptools

 requirements = [
-    "six",
-    "tqdm",
-    "pyyaml",
-    "tabulate",
-    "colorama",
-    "termcolor",
-    "ruamel.yaml",
-    "wcwidth",
+    "six==1.15.0",
+    "tqdm==4.60.0",
+    "pyyaml==5.4.1",
+    "tabulate==0.8.9",
+    "colorama==0.4.4",
+    "termcolor==1.1.0",
+    "wcwidth==0.2.5",
    "ida-settings==2.1.0",
+    "viv-utils==0.6.0",
 ]

 if sys.version_info >= (3, 0):
    # py3
-    requirements.append("halo")
-    requirements.append("networkx")
+    requirements.append("halo==0.0.31")
+    requirements.append("networkx==2.5.1")
+    requirements.append("ruamel.yaml==0.17.0")
+    requirements.append("vivisect==1.0.1")
    requirements.append("smda==1.5.13")
 else:
    # py2
    requirements.append("enum34==1.1.6")  # v1.1.6 is needed by halo 0.0.30 / spinners 0.0.24
    requirements.append("halo==0.0.30")  # halo==0.0.30 is the last version to support py2.7
-    requirements.append("vivisect==0.1.0")
-    requirements.append("viv-utils==0.3.19")
+    requirements.append("vivisect==0.2.1")
    requirements.append("networkx==2.2")  # v2.2 is last version supported by Python 2.7
-    requirements.append("backports.functools-lru-cache")
+    requirements.append("ruamel.yaml==0.16.13")  # last version tested with Python 2.7
+    requirements.append("backports.functools-lru-cache==1.6.1")

 # this sets __version__
 # via: http://stackoverflow.com/a/7071358/87207
@@ -75,13 +77,13 @@ setuptools.setup(
    install_requires=requirements,
    extras_require={
        "dev": [
-            "pytest",
-            "pytest-sugar",
-            "pytest-instafail",
-            "pytest-cov",
-            "pycodestyle",
-            "black ; python_version>'3.0'",
-            "isort",
+            "pytest==4.6.11",  # TODO: Change to 6.2.3 when removing py2
+            "pytest-sugar==0.9.4",
+            "pytest-instafail==0.4.2",
+            "pytest-cov==2.11.1",
+            "pycodestyle==2.7.0",
+            "black==20.8b1 ; python_version>'3.0'",
+            "isort==4.3.21",  # TODO: Change to 5.8.0 when removing py2
        ]
    },
    zip_safe=False,
--- a/tests/data
+++ b/tests/data
--- a/tests/fixtures.py
+++ b/tests/fixtures.py
@@ -520,11 +520,7 @@ def do_test_feature_count(get_extractor, sample, scope, feature, expected):


 def get_extractor(path):
-    if sys.version_info >= (3, 0):
-        extractor = get_smda_extractor(path)
-    else:
-        extractor = get_viv_extractor(path)
-
+    extractor = get_viv_extractor(path)
    # overload the extractor so that the fixture exposes `extractor.path`
    setattr(extractor, "path", path)
    return extractor
--- a/tests/test_ida_features.py
+++ b/tests/test_ida_features.py
@@ -1,104 +1,104 @@
-# run this script from within IDA with ./tests/data/mimikatz.exe open
-import sys
-import logging
-import os.path
-import binascii
-import traceback
-
-import pytest
-
-try:
-    sys.path.append(os.path.dirname(__file__))
-    from fixtures import *
-finally:
-    sys.path.pop()
-
-
-logger = logging.getLogger("test_ida_features")
-
-
-def check_input_file(wanted):
-    import idautils
-
-    # some versions (7.4) of IDA return a truncated version of the MD5.
-    # https://github.com/idapython/bin/issues/11
-    try:
-        found = idautils.GetInputFileMD5()[:31].decode("ascii").lower()
-    except UnicodeDecodeError:
-        # in IDA 7.5 or so, GetInputFileMD5 started returning raw binary
-        # rather than the hex digest
-        found = binascii.hexlify(idautils.GetInputFileMD5()[:15]).decode("ascii").lower()
-
-    if not wanted.startswith(found):
-        raise RuntimeError("please run the tests against sample with MD5: `%s`" % (wanted))
-
-
-def get_ida_extractor(_path):
-    check_input_file("5f66b82558ca92e54e77f216ef4c066c")
-
-    # have to import import this inline so pytest doesn't bail outside of IDA
-    import capa.features.extractors.ida
-
-    return capa.features.extractors.ida.IdaFeatureExtractor()
-
-
-@pytest.mark.skip(reason="IDA Pro tests must be run within IDA")
-def test_ida_features():
-    for (sample, scope, feature, expected) in FEATURE_PRESENCE_TESTS + FEATURE_PRESENCE_TESTS_IDA:
-        id = make_test_id((sample, scope, feature, expected))
-
-        try:
-            check_input_file(get_sample_md5_by_name(sample))
-        except RuntimeError:
-            print("SKIP %s" % (id))
-            continue
-
-        scope = resolve_scope(scope)
-        sample = resolve_sample(sample)
-
-        try:
-            do_test_feature_presence(get_ida_extractor, sample, scope, feature, expected)
-        except Exception as e:
-            print("FAIL %s" % (id))
-            traceback.print_exc()
-        else:
-            print("OK   %s" % (id))
-
-
-@pytest.mark.skip(reason="IDA Pro tests must be run within IDA")
-def test_ida_feature_counts():
-    for (sample, scope, feature, expected) in FEATURE_COUNT_TESTS:
-        id = make_test_id((sample, scope, feature, expected))
-
-        try:
-            check_input_file(get_sample_md5_by_name(sample))
-        except RuntimeError:
-            print("SKIP %s" % (id))
-            continue
-
-        scope = resolve_scope(scope)
-        sample = resolve_sample(sample)
-
-        try:
-            do_test_feature_count(get_ida_extractor, sample, scope, feature, expected)
-        except Exception as e:
-            print("FAIL %s" % (id))
-            traceback.print_exc()
-        else:
-            print("OK   %s" % (id))
-
-
-if __name__ == "__main__":
-    print("-" * 80)
-
-    # invoke all functions in this module that start with `test_`
-    for name in dir(sys.modules[__name__]):
-        if not name.startswith("test_"):
-            continue
-
-        test = getattr(sys.modules[__name__], name)
-        logger.debug("invoking test: %s", name)
-        sys.stderr.flush()
-        test()
-
-    print("DONE")
+# run this script from within IDA with ./tests/data/mimikatz.exe open
+import sys
+import logging
+import os.path
+import binascii
+import traceback
+
+import pytest
+
+try:
+    sys.path.append(os.path.dirname(__file__))
+    from fixtures import *
+finally:
+    sys.path.pop()
+
+
+logger = logging.getLogger("test_ida_features")
+
+
+def check_input_file(wanted):
+    import idautils
+
+    # some versions (7.4) of IDA return a truncated version of the MD5.
+    # https://github.com/idapython/bin/issues/11
+    try:
+        found = idautils.GetInputFileMD5()[:31].decode("ascii").lower()
+    except UnicodeDecodeError:
+        # in IDA 7.5 or so, GetInputFileMD5 started returning raw binary
+        # rather than the hex digest
+        found = binascii.hexlify(idautils.GetInputFileMD5()[:15]).decode("ascii").lower()
+
+    if not wanted.startswith(found):
+        raise RuntimeError("please run the tests against sample with MD5: `%s`" % (wanted))
+
+
+def get_ida_extractor(_path):
+    check_input_file("5f66b82558ca92e54e77f216ef4c066c")
+
+    # have to import import this inline so pytest doesn't bail outside of IDA
+    import capa.features.extractors.ida
+
+    return capa.features.extractors.ida.IdaFeatureExtractor()
+
+
+@pytest.mark.skip(reason="IDA Pro tests must be run within IDA")
+def test_ida_features():
+    for (sample, scope, feature, expected) in FEATURE_PRESENCE_TESTS + FEATURE_PRESENCE_TESTS_IDA:
+        id = make_test_id((sample, scope, feature, expected))
+
+        try:
+            check_input_file(get_sample_md5_by_name(sample))
+        except RuntimeError:
+            print("SKIP %s" % (id))
+            continue
+
+        scope = resolve_scope(scope)
+        sample = resolve_sample(sample)
+
+        try:
+            do_test_feature_presence(get_ida_extractor, sample, scope, feature, expected)
+        except Exception as e:
+            print("FAIL %s" % (id))
+            traceback.print_exc()
+        else:
+            print("OK   %s" % (id))
+
+
+@pytest.mark.skip(reason="IDA Pro tests must be run within IDA")
+def test_ida_feature_counts():
+    for (sample, scope, feature, expected) in FEATURE_COUNT_TESTS:
+        id = make_test_id((sample, scope, feature, expected))
+
+        try:
+            check_input_file(get_sample_md5_by_name(sample))
+        except RuntimeError:
+            print("SKIP %s" % (id))
+            continue
+
+        scope = resolve_scope(scope)
+        sample = resolve_sample(sample)
+
+        try:
+            do_test_feature_count(get_ida_extractor, sample, scope, feature, expected)
+        except Exception as e:
+            print("FAIL %s" % (id))
+            traceback.print_exc()
+        else:
+            print("OK   %s" % (id))
+
+
+if __name__ == "__main__":
+    print("-" * 80)
+
+    # invoke all functions in this module that start with `test_`
+    for name in dir(sys.modules[__name__]):
+        if not name.startswith("test_"):
+            continue
+
+        test = getattr(sys.modules[__name__], name)
+        logger.debug("invoking test: %s", name)
+        sys.stderr.flush()
+        test()
+
+    print("DONE")
--- a/tests/test_main.py
+++ b/tests/test_main.py
@@ -7,6 +7,7 @@
 #  is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 # See the License for the specific language governing permissions and limitations under the License.
 import sys
+import json
 import textwrap

 import pytest
@@ -365,3 +366,20 @@ def test_not_render_rules_also_matched(z9324d_extractor, capsys):
    assert "act as TCP client" in std.out
    assert "connect TCP socket" in std.out
    assert "create TCP socket" in std.out
+
+
+# It tests main works with different backends
+def test_backend_option(capsys):
+    if sys.version_info > (3, 0):
+        path = get_data_path_by_name("pma16-01")
+        assert capa.main.main([path, "-j", "-b", capa.main.BACKEND_VIV]) == 0
+        std = capsys.readouterr()
+        std_json = json.loads(std.out)
+        assert std_json["meta"]["analysis"]["extractor"] == "VivisectFeatureExtractor"
+        assert len(std_json["rules"]) > 0
+
+        assert capa.main.main([path, "-j", "-b", capa.main.BACKEND_SMDA]) == 0
+        std = capsys.readouterr()
+        std_json = json.loads(std.out)
+        assert std_json["meta"]["analysis"]["extractor"] == "SmdaFeatureExtractor"
+        assert len(std_json["rules"]) > 0
--- a/tests/test_rules.py
+++ b/tests/test_rules.py
@@ -681,6 +681,25 @@ def test_explicit_string_values_int():
    assert (String("0x123") in children) == True


+def test_string_values_special_characters():
+    rule = textwrap.dedent(
+        """
+        rule:
+            meta:
+                name: test rule
+            features:
+                - or:
+                    - string: "hello\\r\\nworld"
+                    - string: "bye\\nbye"
+                      description: "test description"
+        """
+    )
+    r = capa.rules.Rule.from_yaml(rule)
+    children = list(r.statement.get_children())
+    assert (String("hello\r\nworld") in children) == True
+    assert (String("bye\nbye") in children) == True
+
+
 def test_regex_values_always_string():
    rules = [
        capa.rules.Rule.from_yaml(
--- a/tests/test_smda_features.py
+++ b/tests/test_smda_features.py
@@ -15,9 +15,10 @@ from fixtures import *
    FEATURE_PRESENCE_TESTS,
    indirect=["sample", "scope"],
 )
+@pytest.mark.xfail(sys.version_info < (3, 0), reason="SMDA only works on py3")
+@pytest.mark.xfail(sys.platform == "win32", reason="SMDA bug: https://github.com/danielplohmann/smda/issues/20")
 def test_smda_features(sample, scope, feature, expected):
-    with xfail(sys.version_info < (3, 0), reason="SMDA only works on py3"):
-        do_test_feature_presence(get_smda_extractor, sample, scope, feature, expected)
+    do_test_feature_presence(get_smda_extractor, sample, scope, feature, expected)


@parametrize(
--- a/tests/test_viv_features.py
+++ b/tests/test_viv_features.py
@@ -16,8 +16,7 @@ from fixtures import *
    indirect=["sample", "scope"],
 )
 def test_viv_features(sample, scope, feature, expected):
-    with xfail(sys.version_info >= (3, 0), reason="vivsect only works on py2"):
-        do_test_feature_presence(get_viv_extractor, sample, scope, feature, expected)
+    do_test_feature_presence(get_viv_extractor, sample, scope, feature, expected)


@parametrize(
@@ -26,5 +25,4 @@ def test_viv_features(sample, scope, feature, expected):
    indirect=["sample", "scope"],
 )
 def test_viv_feature_counts(sample, scope, feature, expected):
-    with xfail(sys.version_info >= (3, 0), reason="vivsect only works on py2"):
-        do_test_feature_count(get_viv_extractor, sample, scope, feature, expected)
+    do_test_feature_count(get_viv_extractor, sample, scope, feature, expected)