/[swish]/trunk/spider/progspider
This is repository of my old source code which isn't updated any more. Go to git.rot13.org for current projects!
ViewVC logotype

Annotation of /trunk/spider/progspider

Parent Directory Parent Directory | Revision Log Revision Log


Revision 84 - (hide annotations)
Sun Aug 29 21:19:13 2004 UTC (19 years, 7 months ago) by dpavlin
File size: 2954 byte(s)
if pdf file doesn't have a title, display filesname and page number

1 dpavlin 81 #!/usr/bin/perl -w
2 dpavlin 46 use strict;
3     use File::Find;
4 dpavlin 56 use Getopt::Long;
5 dpavlin 63 use File::Which;
6 dpavlin 46
7 dpavlin 56 my $collection; # name which will be inserted
8     my $path_add; # add additional info in path
9     my $verbose;
10 dpavlin 46
11 dpavlin 57 #$verbose = 1;
12    
13 dpavlin 56 my $result = GetOptions(
14     "collection=s" => \$collection,
15     "path=s" => \$path_add,
16     "verbose!" => \$verbose,
17     "debug!" => \$verbose,
18     );
19    
20 dpavlin 46 my $dir = shift @ARGV || die "usage: $0 [dir]";
21    
22     my $basedir = $0;
23     $basedir =~ s,/[^/]+$,/,;
24     require "$basedir/filter.pm";
25    
26 dpavlin 63 my $pdftotext = which('pdftotext');
27 dpavlin 56
28 dpavlin 66 select(STDERR); $|=1;
29     select(STDOUT); $|=1;
30    
31 dpavlin 63 print STDERR "using $pdftotext to convert pdf into html\n" if ($pdftotext && $verbose);
32    
33 dpavlin 46 find({ wanted => \&file,
34     follow => 1,
35     no_chdir => 1
36     }, $dir);
37    
38 dpavlin 66 sub dump_contents($$$) {
39     my ($contents,$mtime,$path) = @_;
40    
41     return if (! $contents); # don't die on empty files
42    
43     use bytes;
44     my $size = length $contents;
45    
46     print STDERR " [$size]" if ($verbose);
47    
48     # Output the document (to swish)
49     print <<EOF;
50     Path-Name: $path
51     Content-Length: $size
52     Last-Mtime: $mtime
53 dpavlin 81 Document-Type: html*
54 dpavlin 66
55     EOF
56     print $contents;
57    
58     }
59    
60 dpavlin 46 sub file {
61    
62 dpavlin 63 my $path = $_;
63     my $contents;
64 dpavlin 46
65 dpavlin 63 if ($pdftotext && -f $path && $path =~ m/\.pdf$/i) {
66 dpavlin 56
67 dpavlin 63 print STDERR "$path {converting}" if ($verbose);
68 dpavlin 46
69 dpavlin 66 open(F,"$pdftotext -htmlmeta \"$path\" - |") || die "can't open $pdftotext with '$path'";
70 dpavlin 63 my $html;
71     while(<F>) {
72     # XXX why pdftotext barks if I try to use this is beyond me.
73     #$contents .= $_;
74    
75     $html .= $_;
76     }
77     close(F);
78    
79 dpavlin 81 return if (! $html);
80    
81 dpavlin 84 my $file_only = $path;
82     $file_only =~ s/^.*\/([^\/]+)$/$1/g;
83    
84 dpavlin 66 my ($pre_html,$pages,$post_html) = ('<html><head><title>$path :: page ##page_nr##</title></head><body><pre>',$html,'</pre></body></html>');
85 dpavlin 63
86 dpavlin 72 ($pre_html,$pages,$post_html) = ($1,$2,$3) if ($html =~ m/^(<html>.+?<pre>)(.+)(<\/pre>.+?)$/si);
87 dpavlin 66
88 dpavlin 72 if ($collection) {
89     $pre_html =~ s/<title>(.+?)<\/title>/<title>$collection :: page ##page_nr##<\/title>/si;
90     } else {
91 dpavlin 84 $pre_html =~ s/<title>(.+?)<\/title>/<title>$1 :: page ##page_nr##<\/title>/si ||
92     $pre_html =~ s/<title><\/title>/<title>$file_only :: page ##page_nr##<\/title>/si;
93 dpavlin 72 }
94 dpavlin 66
95     my $page_nr = 1;
96 dpavlin 72 foreach my $page (split(/\f/s,$pages)) {
97     print STDERR " $page_nr" if ($verbose);
98 dpavlin 66 my $pre_tmp = $pre_html;
99     $pre_tmp =~ s/##page_nr##/$page_nr<\/title>/s;
100 dpavlin 68 dump_contents($pre_tmp . $page . $post_html,time(), $path) if ($page !~ m/^\s*$/s);
101 dpavlin 66 $page_nr++;
102     }
103    
104 dpavlin 63 } else {
105    
106 dpavlin 81 return if (! -f $path || ! m/\.(html*|php|pl|txt|info|log|text)$/i);
107 dpavlin 63
108     # skip index files
109     return if (m/index_[a-z]\.html*/i || m/index_symbol\.html*/i);
110    
111     open(F,"$path") || die "can't open file: $path";
112     print STDERR "$path" if ($verbose);
113     while(<F>) {
114     $contents .= "$_";
115     }
116     $contents .= "\n\n";
117    
118     $contents = filter($contents,$collection);
119 dpavlin 66
120     # add optional components to path
121     $path .= " $path_add" if ($path_add);
122    
123     dump_contents($contents,time(), $path);
124 dpavlin 46 }
125    
126 dpavlin 66 print STDERR "\n" if ($verbose);
127 dpavlin 50 # die "zero size content in '$path'" if (! $contents);
128    
129 dpavlin 66 }
130 dpavlin 46

Properties

Name Value
cvs2svn:cvs-rev 1.6
svn:executable *

  ViewVC Help
Powered by ViewVC 1.1.26