开发者

how to extract pdf index/table-of-contents with poppler?

开发者 https://www.devze.com 2023-03-29 05:52 出处:网络
I see that pdf-viewers like okular and evince开发者_如何学Python are able to display the index of a pdf document (book) very well, with link to every paragraph.

I see that pdf-viewers like okular and evince开发者_如何学Python are able to display the index of a pdf document (book) very well, with link to every paragraph. How can they do so? They use poppler library, how could I do extract that index with poppler, or in general?


it just stops at first level (recursion needed to go more deeply)

toc=document->toc();

QDomElement docElem = toc->documentElement();

 QDomNode n = docElem.firstChild();
 while(!n.isNull()) {
     QDomElement e = n.toElement(); // try to convert the node to an element.
     if(!e.isNull()) {
         qDebug("elem %s\n",qPrintable(e.tagName())); // the node really is an element.

     }
     n = n.nextSibling();
 }


Here is a demo how to do this with poppler in Python:

import poppler

def walk_index(iterp, doc):
    while iterp.next():
      link=iterp.get_action()
      s = doc.find_dest(link.dest.named_dest)
      print link.title,' ', doc.get_page(s.page_num).get_label()
      child = iterp.get_child()
      if child:
        walk_index(child, doc)

def main():
    uri = ("file:///"+path_to_pdf)
    doc = poppler.document_new_from_file(uri, None)

    iterp = poppler.IndexIter(doc)
    link = iterp.get_action()
    s = doc.find_dest(link.dest.named_dest)
    print link.title,' ', doc.get_page(s.page_num).get_label()
    walk_index(iterp, doc)
    return 0


if __name__ == '__main__':
    main()

python poppler library is obsolete, here is how to do it with Gobject:

#!/usr/bin/env python
# -*- coding: utf-8 -*-
# walk to table of contents and print titles and pages

import sys
from gi.repository import Poppler

def walk_index(iterp, doc):
    while iterp.next():
        link=iterp.get_action()
        dest=doc.find_dest(link.goto_dest.dest.named_dest)
        s = doc.get_page(dest.page_num-1)
        print link.goto_dest.title, dest.page_num, s.get_label()
        child = iterp.get_child()
        if child:
            walk_index(child, doc)

def main():
    uri = ("file:///"+sys.argv[1])
    doc = Poppler.Document.new_from_file(uri, None)
    iterp = Poppler.IndexIter.new(doc)
    link = iterp.get_action()
    dest=doc.find_dest(link.goto_dest.dest.named_dest)
    s = doc.get_page(dest.page_num-1)
    print link.goto_dest.title, dest.page_num, s.get_label()
    walk_index(iterp, doc)
    return 0


if __name__ == '__main__':
    main()
0

精彩评论

暂无评论...
验证码 换一张
取 消