<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:media="http://search.yahoo.com/mrss/" xmlns:atom="http://www.w3.org/2005/Atom"><channel><title>Sam Wilson's Website :: OCR</title><link>https://twyne.samwilson.id.au/T1994</link><atom:link href="https://twyne.samwilson.id.au/T1994/rss.xml" rel="self" type="application/rss+xml"/><lastBuildDate>Fri, 07 Aug 2026 16:20:29 +0000</lastBuildDate><item><title>P25252</title><description>&lt;p&gt;A good trick from  dpeach about using Tesseract to get text out of a scanned PDF:&lt;/p&gt;
&lt;p&gt;&lt;code&gt;convert -density 300 file.pdf -depth 8 file.tiff&lt;/code&gt;&lt;/p&gt;
&lt;p&gt;&lt;code&gt;tesseract file.tiff OutputFileName&lt;/code&gt;&lt;/p&gt;
&lt;p&gt;&lt;a href="http://www.mythoughtspot.com/2014/10/23/use-tesseract-ocr-with-pdf-file/"&gt;http://www.mythoughtspot.com/2014/10/23/use-tesseract-ocr-with-pdf-file/&lt;/a&gt;&lt;/p&gt;</description><link>https://twyne.samwilson.id.au/P25252</link><guid isPermaLink="true">https://twyne.samwilson.id.au/P25252</guid><pubDate>Mon, 10 Jan 2022 01:14:09 +0000</pubDate></item></channel></rss>
