PDF to Text Java Examples
These examples demonstrate common PDF to Text API workflows in Java.
Basic examples
PDF file to text file
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and save the result to a file. client.convertFileToFile("/path/to/invoice.pdf", "invoice.txt"); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
PDF file to in-memory text
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and store the result in the `txt` variable. byte[] txt = client.convertFile("/path/to/invoice.pdf"); // at this point the "txt" variable contains TXT raw data and // can be sent in an HTTP response, saved to a file, etc. } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } } }
PDF file to text stream
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Create an output stream for the conversion result FileOutputStream outputStream = new FileOutputStream("invoice.txt"); // run the conversion and write the result to the output stream. client.convertFileToStream("/path/to/invoice.pdf", outputStream); // Close the output stream. outputStream.close(); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
PDF url to text file
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and save the result to a file. client.convertUrlToFile("https://pdfcrowd.com/static/pdf/apisamples/invoice.pdf", "invoice.txt"); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
PDF url to in-memory text
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and store the result in the `txt` variable. byte[] txt = client.convertUrl("https://pdfcrowd.com/static/pdf/apisamples/invoice.pdf"); // at this point the "txt" variable contains TXT raw data and // can be sent in an HTTP response, saved to a file, etc. } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } } }
PDF url to text stream
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Create an output stream for the conversion result FileOutputStream outputStream = new FileOutputStream("invoice.txt"); // run the conversion and write the result to the output stream. client.convertUrlToStream("https://pdfcrowd.com/static/pdf/apisamples/invoice.pdf", outputStream); // Close the output stream. outputStream.close(); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
In-memory PDF to text file
import com.pdfcrowd.*; import java.io.*; import java.nio.file.Files; import java.nio.file.Paths; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and save the result to a file. client.convertRawDataToFile(Files.readAllBytes(Paths.get("/path/to/hello_world.pdf")), "invoice.txt"); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
In-memory PDF to in-memory text
import com.pdfcrowd.*; import java.io.*; import java.nio.file.Files; import java.nio.file.Paths; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and store the result in the `txt` variable. byte[] txt = client.convertRawData(Files.readAllBytes(Paths.get("/path/to/hello_world.pdf"))); // at this point the "txt" variable contains TXT raw data and // can be sent in an HTTP response, saved to a file, etc. } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
In-memory PDF to text stream
import com.pdfcrowd.*; import java.io.*; import java.nio.file.Files; import java.nio.file.Paths; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Create an output stream for the conversion result FileOutputStream outputStream = new FileOutputStream("invoice.txt"); // run the conversion and write the result to the output stream. client.convertRawDataToStream(Files.readAllBytes(Paths.get("/path/to/hello_world.pdf")), outputStream); // Close the output stream. outputStream.close(); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
Get info about the current conversion
import com.pdfcrowd.*; import java.io.*; public class ApiTest { public static void main(String[] args) throws IOException, Pdfcrowd.Error { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Configure the conversion. client.setDebugLog(true); client.setPageBreakMode("default"); // Run the conversion and save the result to a file. client.convertFileToFile("/path/to/invoice.pdf", "invoice.txt"); // print URL pointing to the debug log for this request. System.out.println("Debug log url: " + client.getDebugLogUrl()); // print number of conversion credits remaining in your account. System.out.println("Remaining credit count: " + client.getRemainingCreditCount()); // print number of credits consumed for this conversion. System.out.println("Consumed credit count: " + client.getConsumedCreditCount()); // print unique identifier assigned to this conversion job. System.out.println("Job id: " + client.getJobId()); // print total number of pages in the output document. System.out.println("Page count: " + client.getPageCount()); // print size of the output data in bytes. System.out.println("Output size: " + client.getOutputSize()); } catch(Pdfcrowd.Error why) { System.err.println("PDFCrowd Error: " + why); throw why; } catch(IOException why) { System.err.println("IO Error: " + why); throw why; } } }
Spring examples
PDF file to text in Spring
import java.net.URLEncoder; import java.io.UnsupportedEncodingException; import org.springframework.stereotype.Controller; import org.springframework.web.bind.annotation.PostMapping; import org.springframework.http.ResponseEntity; import org.springframework.http.HttpHeaders; import org.springframework.http.HttpStatus; import com.pdfcrowd.*; @Controller public class DemoController { // The recommended method is POST. @PostMapping("/") public ResponseEntity<byte[]> convert() throws UnsupportedEncodingException { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and store the result in the `txt` variable. byte[] txt = client.convertFile("/path/to/invoice.pdf"); // Set HTTP response headers. HttpHeaders headers = new HttpHeaders(); headers.add("Content-Type", "text/plain"); headers.add("Cache-Control", "max-age=0"); headers.add("Accept-Ranges", "none"); headers.add("Content-Disposition", "attachment; filename*=UTF-8''" + URLEncoder.encode("invoice.txt", "UTF-8").replace("+", "%20")); // Send the result in the HTTP response. return new ResponseEntity<>(txt, headers, HttpStatus.OK); } catch(Pdfcrowd.Error why) { // Send the error in the HTTP response. HttpHeaders headers = new HttpHeaders(); headers.add("Content-Type", "text/plain"); return new ResponseEntity<>(why.toString().getBytes(), headers, HttpStatus.BAD_REQUEST); } } }
PDF url to text in Spring
import java.net.URLEncoder; import java.io.UnsupportedEncodingException; import org.springframework.stereotype.Controller; import org.springframework.web.bind.annotation.PostMapping; import org.springframework.http.ResponseEntity; import org.springframework.http.HttpHeaders; import org.springframework.http.HttpStatus; import com.pdfcrowd.*; @Controller public class DemoController { // The recommended method is POST. @PostMapping("/") public ResponseEntity<byte[]> convert() throws UnsupportedEncodingException { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and store the result in the `txt` variable. byte[] txt = client.convertUrl("https://pdfcrowd.com/static/pdf/apisamples/invoice.pdf"); // Set HTTP response headers. HttpHeaders headers = new HttpHeaders(); headers.add("Content-Type", "text/plain"); headers.add("Cache-Control", "max-age=0"); headers.add("Accept-Ranges", "none"); headers.add("Content-Disposition", "attachment; filename*=UTF-8''" + URLEncoder.encode("invoice.txt", "UTF-8").replace("+", "%20")); // Send the result in the HTTP response. return new ResponseEntity<>(txt, headers, HttpStatus.OK); } catch(Pdfcrowd.Error why) { // Send the error in the HTTP response. HttpHeaders headers = new HttpHeaders(); headers.add("Content-Type", "text/plain"); return new ResponseEntity<>(why.toString().getBytes(), headers, HttpStatus.BAD_REQUEST); } } }
In-memory PDF to text in Spring
import java.net.URLEncoder; import java.io.UnsupportedEncodingException; import org.springframework.stereotype.Controller; import org.springframework.web.bind.annotation.PostMapping; import org.springframework.http.ResponseEntity; import org.springframework.http.HttpHeaders; import org.springframework.http.HttpStatus; import com.pdfcrowd.*; import java.nio.file.Files; import java.nio.file.Paths; import java.io.IOException; @Controller public class DemoController { // The recommended method is POST. @PostMapping("/") public ResponseEntity<byte[]> convert() throws IOException, UnsupportedEncodingException { try { // Create an API client instance. Pdfcrowd.PdfToTextClient client = new Pdfcrowd.PdfToTextClient("demo", "demo"); // Run the conversion and store the result in the `txt` variable. byte[] txt = client.convertRawData(Files.readAllBytes(Paths.get("/path/to/hello_world.pdf"))); // Set HTTP response headers. HttpHeaders headers = new HttpHeaders(); headers.add("Content-Type", "text/plain"); headers.add("Cache-Control", "max-age=0"); headers.add("Accept-Ranges", "none"); headers.add("Content-Disposition", "attachment; filename*=UTF-8''" + URLEncoder.encode("invoice.txt", "UTF-8").replace("+", "%20")); // Send the result in the HTTP response. return new ResponseEntity<>(txt, headers, HttpStatus.OK); } catch(Pdfcrowd.Error why) { // Send the error in the HTTP response. HttpHeaders headers = new HttpHeaders(); headers.add("Content-Type", "text/plain"); return new ResponseEntity<>(why.toString().getBytes(), headers, HttpStatus.BAD_REQUEST); } } }