# pdf-text-img

> npm i pdf-text-img

Latest version **1.0.12** (published 2019-12-08) · ISC license · 0 weekly downloads

## Install

```sh
npm install pdf-text-img
pnpm add pdf-text-img
yarn add pdf-text-img
bun add pdf-text-img
```

## Health

**Score 15/100 (F)** — status: abandoned.

Positive: no vulnerabilities.

Warnings: low downloads; no types; no esm support.

Negative: abandoned; low maintenance score.

## Facts

| | |
|---|---|
| Version | 1.0.12 |
| Published | 2019-12-08 |
| First published | 2019-12-04 |
| Weekly downloads | 0 |
| License | ISC |
| TypeScript types | none |
| Module format | CommonJS |
| Dependencies | 0 |
| Unpacked size | 6.7 MB |
| Known vulnerabilities | 0 |
| Install scripts | no |
| Maintainers | yfile |

## Links

- npm: https://www.npmjs.com/package/pdf-text-img
- npm.io page: https://npm.io/package/pdf-text-img

## Recent versions

- 1.0.12 (latest) — 2019-12-08
- 1.0.10 — 2019-12-08
- 1.0.9 — 2019-12-06
- 1.0.8 — 2019-12-06
- 1.0.7 — 2019-12-06
- 1.0.6 — 2019-12-06
- 1.0.5 — 2019-12-06
- 1.0.4 — 2019-12-06
- 1.0.3 — 2019-12-04
- 1.0.2 — 2019-12-04
- 1.0.1 — 2019-12-04
- 1.0.0 — 2019-12-04

## README

##  pdf-text-img

```text
Extract text, images, and export pages as images from PDF
Written in pure JS
```

## Usage

```js
const fs = require('fs');
const pdf = require('pdf-text-img');

//todo
let yourPDF = 'pdf.pdf';
let dataBuffer = fs.readFileSync(yourPDF);

pdf.LoadingPDF(dataBuffer).then(function (pageCount) {
    for (let i = 1; i <= pageCount; i++) {
        /**
         * Extract text
         */
        pdf.ExtractText(i).then(function (data) {
            console.log(data);
        });

        /**
         * Extract image
         */
        pdf.ExtractImg(i).then(function (data) {
            for (let j = 0; j < data.length; j++) {
                // width: data[j]['width']
                // height: data[j]['height']

                let dataBuffer = new Buffer.from(data[j]['imgBase64'], 'base64');

                fs.writeFile('a_' + i + '_' + j + '.jpg', dataBuffer, function (err) {
                    if (err) throw err;
                });
            }
        });

        /**
         * Export as picture
         * Tips: Because of the use of native canvas, it needs to be used in H5 environment
         */
        pdf.ExportImg(i).then(function (data) {
            // width: data['width']
            // height: data['height']

            let dataBuffer = new Buffer.from(data['imgBase64'], 'base64');
            fs.writeFile('b_' + i + '.jpg', dataBuffer, function (err) {
                if (err) throw err;
            });
        });
    }
});
```
#### It is worth noting that if a PDF has 1000 pages and each page has 100 pictures, it is obviously a waste of resources according to the above logic. Therefore, we can refer to the following logic of "batch extraction of pictures":
```js
const fs = require('fs');
const pdf = require('pdf-text-img');

//todo
let yourPDF = 'pdf.pdf';
let dataBuffer = fs.readFileSync(yourPDF);

async function Batch(startPage) {
    /**
     * Download pictures
     */
    startPage = startPage == undefined ? 1 : startPage;
    if (startPage > pdf.pdfInfo.numPages) {
        console.log('Completed!');
        pdf.End();
        return;
    }

    pdf.ExtractImg(startPage).then(function (data) {
        let index = 0;
        let downloadCount = data.length; // Total number of pictures to download

        // No picture on current page
        if (downloadCount == 0) {
            startPage++;

            console.log('No picture on current page, extract next page...');

            Batch(startPage);
            return;
        }

        for (let i = 0; i < data.length; i++) {
            let dataBuffer = new Buffer.from(data[i]['imgBase64'], 'base64');

            downloadName++;
            fs.writeFile('pdf_' + startPage + '_' + i + '.jpg', dataBuffer, function (err) {
                if (err) throw err;

                index++;

                if (index == downloadCount) {
                    startPage++;

                    console.log('Continue extraction...');

                    Batch(startPage);
                }
            });
        }
    });
}

pdf.LoadingPDF(dataBuffer).then(function (pageCount) {
    /**
     * batch extraction of pictures
     */
    Batch();
});
```
```text
Welcome to exchange, email: yfilemail@163.com
```

---
_Source: https://npm.io/package/pdf-text-img · Machine-readable twin of the npm.io package page. Health data is recomputed on every publish._
