ViewVC Help
View File | Revision Log | Show Annotations | Download File
/cvs/AnyEvent-HTTP/HTTP.pm
(Generate patch)

Comparing AnyEvent-HTTP/HTTP.pm (file contents):
Revision 1.78 by root, Sat Jan 1 19:32:41 2011 UTC vs.
Revision 1.90 by root, Mon Jan 3 00:41:25 2011 UTC

36 36
37=cut 37=cut
38 38
39package AnyEvent::HTTP; 39package AnyEvent::HTTP;
40 40
41use strict; 41use common::sense;
42no warnings;
43 42
44use Errno (); 43use Errno ();
45 44
46use AnyEvent 5.0 (); 45use AnyEvent 5.0 ();
47use AnyEvent::Util (); 46use AnyEvent::Util ();
58our $MAX_PERSISTENT = 8; 57our $MAX_PERSISTENT = 8;
59our $PERSISTENT_TIMEOUT = 2; 58our $PERSISTENT_TIMEOUT = 2;
60our $TIMEOUT = 300; 59our $TIMEOUT = 300;
61 60
62# changing these is evil 61# changing these is evil
63our $MAX_PERSISTENT_PER_HOST = 0; 62our $MAX_PERSISTENT_PER_HOST = 2;
64our $MAX_PER_HOST = 4; 63our $MAX_PER_HOST = 4;
65 64
66our $PROXY; 65our $PROXY;
67our $ACTIVE = 0; 66our $ACTIVE = 0;
68 67
170C<Host:>, C<Content-Length:>, C<Connection:> and C<Cookie:> headers and 169C<Host:>, C<Content-Length:>, C<Connection:> and C<Cookie:> headers and
171will provide defaults at least for C<TE:>, C<Referer:> and C<User-Agent:> 170will provide defaults at least for C<TE:>, C<Referer:> and C<User-Agent:>
172(this can be suppressed by using C<undef> for these headers in which case 171(this can be suppressed by using C<undef> for these headers in which case
173they won't be sent at all). 172they won't be sent at all).
174 173
174You really should provide your own C<User-Agent:> header value that is
175appropriate for your program - I wouldn't be surprised if the default
176AnyEvent string gets blocked by webservers sooner or later.
177
175=item timeout => $seconds 178=item timeout => $seconds
176 179
177The time-out to use for various stages - each connect attempt will reset 180The time-out to use for various stages - each connect attempt will reset
178the timeout, as will read or write activity, i.e. this is not an overall 181the timeout, as will read or write activity, i.e. this is not an overall
179timeout. 182timeout.
196=item cookie_jar => $hash_ref 199=item cookie_jar => $hash_ref
197 200
198Passing this parameter enables (simplified) cookie-processing, loosely 201Passing this parameter enables (simplified) cookie-processing, loosely
199based on the original netscape specification. 202based on the original netscape specification.
200 203
201The C<$hash_ref> must be an (initially empty) hash reference which will 204The C<$hash_ref> must be an (initially empty) hash reference which
202get updated automatically. It is possible to save the cookie jar to 205will get updated automatically. It is possible to save the cookie jar
203persistent storage with something like JSON or Storable, but this is not 206to persistent storage with something like JSON or Storable - see the
204recommended, as session-only cookies might survive longer than expected. 207C<AnyEvent::HTTP::cookie_jar_expire> function if you wish to remove
208expired or session-only cookies, and also for documentation on the format
209of the cookie jar.
205 210
206Note that this cookie implementation is not meant to be complete. If 211Note that this cookie implementation is not meant to be complete. If
207you want complete cookie management you have to do that on your 212you want complete cookie management you have to do that on your
208own. C<cookie_jar> is meant as a quick fix to get some cookie-using sites 213own. C<cookie_jar> is meant as a quick fix to get most cookie-using sites
209working. Cookies are a privacy disaster, do not use them unless required 214working. Cookies are a privacy disaster, do not use them unless required
210to. 215to.
211 216
212When cookie processing is enabled, the C<Cookie:> and C<Set-Cookie:> 217When cookie processing is enabled, the C<Cookie:> and C<Set-Cookie:>
213headers will be set and handled by this module, otherwise they will be 218headers will be set and handled by this module, otherwise they will be
328Example: do a HTTP HEAD request on https://www.google.com/, use a 333Example: do a HTTP HEAD request on https://www.google.com/, use a
329timeout of 30 seconds. 334timeout of 30 seconds.
330 335
331 http_request 336 http_request
332 GET => "https://www.google.com", 337 GET => "https://www.google.com",
338 headers => { "user-agent" => "MySearchClient 1.0" },
333 timeout => 30, 339 timeout => 30,
334 sub { 340 sub {
335 my ($body, $hdr) = @_; 341 my ($body, $hdr) = @_;
336 use Data::Dumper; 342 use Data::Dumper;
337 print Dumper $hdr; 343 print Dumper $hdr;
378 push @{ $CO_SLOT{$_[0]}[1] }, $_[1]; 384 push @{ $CO_SLOT{$_[0]}[1] }, $_[1];
379 385
380 _slot_schedule $_[0]; 386 _slot_schedule $_[0];
381} 387}
382 388
389#############################################################################
390
391# expire cookies
392sub cookie_jar_expire($;$) {
393 my ($jar, $session_end) = @_;
394
395 %$jar = () if $jar->{version} != 1;
396
397 my $anow = AE::now;
398
399 while (my ($chost, $paths) = each %$jar) {
400 next unless ref $paths;
401
402 while (my ($cpath, $cookies) = each %$paths) {
403 while (my ($cookie, $kv) = each %$cookies) {
404 if (exists $kv->{_expires}) {
405 delete $cookies->{$cookie}
406 if $anow > $kv->{_expires};
407 } elsif ($session_end) {
408 delete $cookies->{$cookie};
409 }
410 }
411
412 delete $paths->{$cpath}
413 unless %$cookies;
414 }
415
416 delete $jar->{$chost}
417 unless %$paths;
418 }
419}
420
383# extract cookies from jar 421# extract cookies from jar
384sub cookie_jar_extract($$$$) { 422sub cookie_jar_extract($$$$) {
385 my ($jar, $uscheme, $uhost, $upath) = @_; 423 my ($jar, $uscheme, $uhost, $upath) = @_;
386 424
387 %$jar = () if $jar->{version} != 1; 425 %$jar = () if $jar->{version} != 1;
403 next unless $cpath eq substr $upath, 0, length $cpath; 441 next unless $cpath eq substr $upath, 0, length $cpath;
404 442
405 while (my ($cookie, $kv) = each %$cookies) { 443 while (my ($cookie, $kv) = each %$cookies) {
406 next if $uscheme ne "https" && exists $kv->{secure}; 444 next if $uscheme ne "https" && exists $kv->{secure};
407 445
408 if (exists $kv->{expires}) { 446 if (exists $kv->{_expires} and AE::now > $kv->{_expires}) {
409 if (AE::now > parse_date ($kv->{expires})) {
410 delete $cookies->{$cookie}; 447 delete $cookies->{$cookie};
411 next; 448 next;
412 }
413 } 449 }
414 450
415 my $value = $kv->{value}; 451 my $value = $kv->{value};
416 452
417 if ($value =~ /[=;,[:space:]]/) { 453 if ($value =~ /[=;,[:space:]]/) {
426 462
427 \@cookies 463 \@cookies
428} 464}
429 465
430# parse set_cookie header into jar 466# parse set_cookie header into jar
431sub cookie_jar_set_cookie($$$) { 467sub cookie_jar_set_cookie($$$$) {
432 my ($jar, $set_cookie, $uhost) = @_; 468 my ($jar, $set_cookie, $uhost, $date) = @_;
469
470 my $anow = int AE::now;
471 my $snow; # server-now
433 472
434 for ($set_cookie) { 473 for ($set_cookie) {
435 # parse NAME=VALUE 474 # parse NAME=VALUE
436 my @kv; 475 my @kv;
437 476
477 # expires is not http-compliant in the original cookie-spec,
478 # we support the official date format and some extensions
438 while ( 479 while (
439 m{ 480 m{
440 \G\s* 481 \G\s*
441 (?: 482 (?:
442 expires \s*=\s* ([A-Z][a-z][a-z],\ [^,;]+) 483 expires \s*=\s* ([A-Z][a-z][a-z]+,\ [^,;]+)
443 | ([^=;,[:space:]]+) \s*=\s* (?: "((?:[^\\"]+|\\.)*)" | ([^=;,[:space:]]*) ) 484 | ([^=;,[:space:]]+) (?: \s*=\s* (?: "((?:[^\\"]+|\\.)*)" | ([^=;,[:space:]]*) ) )?
444 ) 485 )
445 }gcxsi 486 }gcxsi
446 ) { 487 ) {
447 my $name = $2; 488 my $name = $2;
448 my $value = $4; 489 my $value = $4;
449 490
450 unless (defined $name) { 491 if (defined $1) {
451 # expires 492 # expires
452 $name = "expires"; 493 $name = "expires";
453 $value = $1; 494 $value = $1;
454 } elsif (!defined $value) { 495 } elsif (defined $3) {
455 # quoted 496 # quoted
456 $value = $3; 497 $value = $3;
457 $value =~ s/\\(.)/$1/gs; 498 $value =~ s/\\(.)/$1/gs;
458 } 499 }
459 500
465 last unless @kv; 506 last unless @kv;
466 507
467 my $name = shift @kv; 508 my $name = shift @kv;
468 my %kv = (value => shift @kv, @kv); 509 my %kv = (value => shift @kv, @kv);
469 510
470 $kv{expires} ||= format_date (AE::now + $kv{"max-age"})
471 if exists $kv{"max-age"}; 511 if (exists $kv{"max-age"}) {
512 $kv{_expires} = $anow + delete $kv{"max-age"};
513 } elsif (exists $kv{expires}) {
514 $snow ||= parse_date ($date) || $anow;
515 $kv{_expires} = $anow + (parse_date (delete $kv{expires}) - $snow);
516 } else {
517 delete $kv{_expires};
518 }
472 519
473 my $cdom; 520 my $cdom;
474 my $cpath = (delete $kv{path}) || "/"; 521 my $cpath = (delete $kv{path}) || "/";
475 522
476 if (exists $kv{domain}) { 523 if (exists $kv{domain}) {
487 $cdom = $uhost; 534 $cdom = $uhost;
488 } 535 }
489 536
490 # store it 537 # store it
491 $jar->{version} = 1; 538 $jar->{version} = 1;
492 $jar->{$cdom}{$cpath}{$name} = \%kv; 539 $jar->{lc $cdom}{$cpath}{$name} = \%kv;
493 540
494 redo if /\G\s*,/gc; 541 redo if /\G\s*,/gc;
495 } 542 }
496} 543}
497 544
564 : return $cb->(undef, { @pseudo, Status => 599, Reason => "Only http and https URL schemes supported" }); 611 : return $cb->(undef, { @pseudo, Status => 599, Reason => "Only http and https URL schemes supported" });
565 612
566 $uauthority =~ /^(?: .*\@ )? ([^\@:]+) (?: : (\d+) )?$/x 613 $uauthority =~ /^(?: .*\@ )? ([^\@:]+) (?: : (\d+) )?$/x
567 or return $cb->(undef, { @pseudo, Status => 599, Reason => "Unparsable URL" }); 614 or return $cb->(undef, { @pseudo, Status => 599, Reason => "Unparsable URL" });
568 615
569 my $uhost = $1; 616 my $uhost = lc $1;
570 $uport = $2 if defined $2; 617 $uport = $2 if defined $2;
571 618
572 $hdr{host} = defined $2 ? "$uhost:$2" : "$uhost" 619 $hdr{host} = defined $2 ? "$uhost:$2" : "$uhost"
573 unless exists $hdr{host}; 620 unless exists $hdr{host};
574 621
593 $rscheme = "http" unless defined $rscheme; 640 $rscheme = "http" unless defined $rscheme;
594 641
595 # don't support https requests over https-proxy transport, 642 # don't support https requests over https-proxy transport,
596 # can't be done with tls as spec'ed, unless you double-encrypt. 643 # can't be done with tls as spec'ed, unless you double-encrypt.
597 $rscheme = "http" if $uscheme eq "https" && $rscheme eq "https"; 644 $rscheme = "http" if $uscheme eq "https" && $rscheme eq "https";
645
646 $rhost = lc $rhost;
647 $rscheme = lc $rscheme;
598 } else { 648 } else {
599 ($rhost, $rport, $rscheme, $rpath) = ($uhost, $uport, $uscheme, $upath); 649 ($rhost, $rport, $rscheme, $rpath) = ($uhost, $uport, $uscheme, $upath);
600 } 650 }
601 651
602 # leave out fragment and query string, just a heuristic 652 # leave out fragment and query string, just a heuristic
604 $hdr{"user-agent"} = $USERAGENT unless exists $hdr{"user-agent"}; 654 $hdr{"user-agent"} = $USERAGENT unless exists $hdr{"user-agent"};
605 655
606 $hdr{"content-length"} = length $arg{body} 656 $hdr{"content-length"} = length $arg{body}
607 if length $arg{body} || $method ne "GET"; 657 if length $arg{body} || $method ne "GET";
608 658
609 $hdr{connection} = "close TE"; #1.1 659 my $idempotent = $method =~ /^(?:GET|HEAD|PUT|DELETE|OPTIONS|TRACE)$/;
660
661 # default value for keepalive is true iff the request is for an idempotent method
662 my $keepalive = exists $arg{keepalive}
663 ? $arg{keepalive}*1
664 : $idempotent ? $PERSISTENT_TIMEOUT : 0;
665
666 $hdr{connection} = ($keepalive ? "" : "close ") . "Te"; #1.1
610 $hdr{te} = "trailers" unless exists $hdr{te}; #1.1 667 $hdr{te} = "trailers" unless exists $hdr{te}; #1.1
611 668
612 my %state = (connect_guard => 1); 669 my %state = (connect_guard => 1);
613 670
614 _get_slot $uhost, sub {
615 $state{slot_guard} = shift;
616
617 return unless $state{connect_guard};
618
619 my $ae_error = 595; # connecting 671 my $ae_error = 595; # connecting
620 672
621 my $connect_cb = sub { 673 # handle actual, non-tunneled, request
622 $state{fh} = shift 674 my $handle_actual_request = sub {
675 $ae_error = 596; # request phase
676
677 $state{handle}->starttls ("connect") if $uscheme eq "https" && !exists $state{handle}{tls};
678
679 # send request
680 $state{handle}->push_write (
681 "$method $rpath HTTP/1.1\015\012"
682 . (join "", map "\u$_: $hdr{$_}\015\012", grep defined $hdr{$_}, keys %hdr)
683 . "\015\012"
684 . (delete $arg{body})
685 );
686
687 # return if error occured during push_write()
688 return unless %state;
689
690 # reduce memory usage, save a kitten, also re-use it for the response headers.
691 %hdr = ();
692
693 # status line and headers
694 $state{read_response} = sub {
695 for ("$_[1]") {
696 y/\015//d; # weed out any \015, as they show up in the weirdest of places.
697
698 /^HTTP\/0*([0-9\.]+) \s+ ([0-9]{3}) (?: \s+ ([^\012]*) )? \012/gxci
699 or return (%state = (), $cb->(undef, { @pseudo, Status => 599, Reason => "Invalid server response" }));
700
701 # 100 Continue handling
702 # should not happen as we don't send expect: 100-continue,
703 # but we handle it just in case.
704 # since we send the request body regardless, if we get an error
705 # we are out of-sync, which we currently do NOT handle correctly.
706 return $state{handle}->push_read (line => $qr_nlnl, $state{read_response})
707 if $2 eq 100;
708
709 push @pseudo,
710 HTTPVersion => $1,
711 Status => $2,
712 Reason => $3,
623 or do { 713 ;
624 my $err = "$!"; 714
715 my $hdr = parse_hdr
716 or return (%state = (), $cb->(undef, { @pseudo, Status => 599, Reason => "Garbled response headers" }));
717
718 %hdr = (%$hdr, @pseudo);
719 }
720
721 # redirect handling
722 # microsoft and other shitheads don't give a shit for following standards,
723 # try to support some common forms of broken Location headers.
724 if ($hdr{location} !~ /^(?: $ | [^:\/?\#]+ : )/x) {
725 $hdr{location} =~ s/^\.\/+//;
726
727 my $url = "$rscheme://$uhost:$uport";
728
729 unless ($hdr{location} =~ s/^\///) {
730 $url .= $upath;
731 $url =~ s/\/[^\/]*$//;
732 }
733
734 $hdr{location} = "$url/$hdr{location}";
735 }
736
737 my $redirect;
738
739 if ($recurse) {
740 my $status = $hdr{Status};
741
742 # industry standard is to redirect POST as GET for
743 # 301, 302 and 303, in contrast to HTTP/1.0 and 1.1.
744 # also, the UA should ask the user for 301 and 307 and POST,
745 # industry standard seems to be to simply follow.
746 # we go with the industry standard.
747 if ($status == 301 or $status == 302 or $status == 303) {
748 # HTTP/1.1 is unclear on how to mutate the method
749 $method = "GET" unless $method eq "HEAD";
750 $redirect = 1;
751 } elsif ($status == 307) {
752 $redirect = 1;
753 }
754 }
755
756 my $finish = sub { # ($data, $err_status, $err_reason[, $keepalive])
757 my $may_keep_alive = $_[3];
758
759 $state{handle}->destroy if $state{handle};
625 %state = (); 760 %state = ();
626 return $cb->(undef, { @pseudo, Status => $ae_error, Reason => $err }); 761
762 if (defined $_[1]) {
763 $hdr{OrigStatus} = $hdr{Status}; $hdr{Status} = $_[1];
764 $hdr{OrigReason} = $hdr{Reason}; $hdr{Reason} = $_[2];
765 }
766
767 # set-cookie processing
768 if ($arg{cookie_jar}) {
769 cookie_jar_set_cookie $arg{cookie_jar}, $hdr{"set-cookie"}, $uhost, $hdr{date};
770 }
771
772 if ($redirect && exists $hdr{location}) {
773 # we ignore any errors, as it is very common to receive
774 # Content-Length != 0 but no actual body
775 # we also access %hdr, as $_[1] might be an erro
776 http_request (
777 $method => $hdr{location},
778 %arg,
779 recurse => $recurse - 1,
780 Redirect => [$_[0], \%hdr],
781 $cb);
782 } else {
783 $cb->($_[0], \%hdr);
784 }
785 };
786
787 $ae_error = 597; # body phase
788
789 my $len = $hdr{"content-length"};
790
791 # body handling, many different code paths
792 # - no body expected
793 # - want_body_handle
794 # - te chunked
795 # - 2x length known (with or without on_body)
796 # - 2x length not known (with or without on_body)
797 if (!$redirect && $arg{on_header} && !$arg{on_header}(\%hdr)) {
798 $finish->(undef, 598 => "Request cancelled by on_header");
799 } elsif (
800 $hdr{Status} =~ /^(?:1..|204|205|304)$/
801 or $method eq "HEAD"
802 or (defined $len && $len == 0) # == 0, not !, because "0 " is true
803 ) {
804 # no body
805 $finish->("", undef, undef, 1);
806
807 } elsif (!$redirect && $arg{want_body_handle}) {
808 $_[0]->on_eof (undef);
809 $_[0]->on_error (undef);
810 $_[0]->on_read (undef);
811
812 $finish->(delete $state{handle});
813
814 } elsif ($hdr{"transfer-encoding"} =~ /\bchunked\b/i) {
815 my $cl = 0;
816 my $body = undef;
817 my $on_body = $arg{on_body} || sub { $body .= shift; 1 };
818
819 $state{read_chunk} = sub {
820 $_[1] =~ /^([0-9a-fA-F]+)/
821 or $finish->(undef, $ae_error => "Garbled chunked transfer encoding");
822
823 my $len = hex $1;
824
825 if ($len) {
826 $cl += $len;
827
828 $_[0]->push_read (chunk => $len, sub {
829 $on_body->($_[1], \%hdr)
830 or return $finish->(undef, 598 => "Request cancelled by on_body");
831
832 $_[0]->push_read (line => sub {
833 length $_[1]
834 and return $finish->(undef, $ae_error => "Garbled chunked transfer encoding");
835 $_[0]->push_read (line => $state{read_chunk});
836 });
837 });
838 } else {
839 $hdr{"content-length"} ||= $cl;
840
841 $_[0]->push_read (line => $qr_nlnl, sub {
842 if (length $_[1]) {
843 for ("$_[1]") {
844 y/\015//d; # weed out any \015, as they show up in the weirdest of places.
845
846 my $hdr = parse_hdr
847 or return $finish->(undef, $ae_error => "Garbled response trailers");
848
849 %hdr = (%hdr, %$hdr);
850 }
851 }
852
853 $finish->($body, undef, undef, 1);
854 });
855 }
627 }; 856 };
628 857
858 $_[0]->push_read (line => $state{read_chunk});
859
860 } elsif ($arg{on_body}) {
861 if (defined $len) {
862 $_[0]->on_read (sub {
863 $len -= length $_[0]{rbuf};
864
865 $arg{on_body}(delete $_[0]{rbuf}, \%hdr)
866 or return $finish->(undef, 598 => "Request cancelled by on_body");
867
868 $len > 0
869 or $finish->("", undef, undef, 1);
870 });
871 } else {
872 $_[0]->on_eof (sub {
873 $finish->("");
874 });
875 $_[0]->on_read (sub {
876 $arg{on_body}(delete $_[0]{rbuf}, \%hdr)
877 or $finish->(undef, 598 => "Request cancelled by on_body");
878 });
879 }
880 } else {
881 $_[0]->on_eof (undef);
882
883 if (defined $len) {
884 $_[0]->on_read (sub {
885 $finish->((substr delete $_[0]{rbuf}, 0, $len, ""), undef, undef, 1)
886 if $len <= length $_[0]{rbuf};
887 });
888 } else {
889 $_[0]->on_error (sub {
890 ($! == Errno::EPIPE || !$!)
891 ? $finish->(delete $_[0]{rbuf})
892 : $finish->(undef, $ae_error => $_[2]);
893 });
894 $_[0]->on_read (sub { });
895 }
896 }
897 };
898
899 $state{handle}->push_read (line => $qr_nlnl, $state{read_response});
900 };
901
902 my $connect_cb = sub {
903 $state{fh} = shift
904 or do {
905 my $err = "$!";
906 %state = ();
907 return $cb->(undef, { @pseudo, Status => $ae_error, Reason => $err });
908 };
909
629 return unless delete $state{connect_guard}; 910 return unless delete $state{connect_guard};
630 911
631 # get handle 912 # get handle
632 $state{handle} = new AnyEvent::Handle 913 $state{handle} = new AnyEvent::Handle
633 fh => $state{fh}, 914 fh => $state{fh},
634 peername => $rhost, 915 peername => $rhost,
635 tls_ctx => $arg{tls_ctx}, 916 tls_ctx => $arg{tls_ctx},
636 # these need to be reconfigured on keepalive handles 917 # these need to be reconfigured on keepalive handles
637 timeout => $timeout, 918 timeout => $timeout,
638 on_error => sub { 919 on_error => sub {
639 %state = (); 920 %state = ();
640 $cb->(undef, { @pseudo, Status => $ae_error, Reason => $_[2] }); 921 $cb->(undef, { @pseudo, Status => $ae_error, Reason => $_[2] });
641 }, 922 },
642 on_eof => sub { 923 on_eof => sub {
643 %state = (); 924 %state = ();
644 $cb->(undef, { @pseudo, Status => $ae_error, Reason => "Unexpected end-of-file" }); 925 $cb->(undef, { @pseudo, Status => $ae_error, Reason => "Unexpected end-of-file" });
645 }, 926 },
646 ; 927 ;
647 928
648 # limit the number of persistent connections 929 # limit the number of persistent connections
649 # keepalive not yet supported 930 # keepalive not yet supported
650# if ($KA_COUNT{$_[1]} < $MAX_PERSISTENT_PER_HOST) { 931# if ($KA_COUNT{$_[1]} < $MAX_PERSISTENT_PER_HOST) {
651# ++$KA_COUNT{$_[1]}; 932# ++$KA_COUNT{$_[1]};
652# $state{handle}{ka_count_guard} = AnyEvent::Util::guard { 933# $state{handle}{ka_count_guard} = AnyEvent::Util::guard {
653# --$KA_COUNT{$_[1]} 934# --$KA_COUNT{$_[1]}
654# }; 935# };
655# $hdr{connection} = "keep-alive"; 936# $hdr{connection} = "keep-alive";
656# } 937# }
657 938
658 $state{handle}->starttls ("connect") if $rscheme eq "https"; 939 $state{handle}->starttls ("connect") if $rscheme eq "https";
659 940
660 # handle actual, non-tunneled, request
661 my $handle_actual_request = sub {
662 $ae_error = 596; # request phase
663
664 $state{handle}->starttls ("connect") if $uscheme eq "https" && !exists $state{handle}{tls};
665
666 # send request
667 $state{handle}->push_write (
668 "$method $rpath HTTP/1.1\015\012"
669 . (join "", map "\u$_: $hdr{$_}\015\012", grep defined $hdr{$_}, keys %hdr)
670 . "\015\012"
671 . (delete $arg{body})
672 );
673
674 # return if error occured during push_write()
675 return unless %state;
676
677 %hdr = (); # reduce memory usage, save a kitten, also make it possible to re-use
678
679 # status line and headers
680 $state{read_response} = sub {
681 for ("$_[1]") {
682 y/\015//d; # weed out any \015, as they show up in the weirdest of places.
683
684 /^HTTP\/0*([0-9\.]+) \s+ ([0-9]{3}) (?: \s+ ([^\012]*) )? \012/gxci
685 or return (%state = (), $cb->(undef, { @pseudo, Status => 599, Reason => "Invalid server response" }));
686
687 # 100 Continue handling
688 # should not happen as we don't send expect: 100-continue,
689 # but we handle it just in case.
690 # since we send the request body regardless, if we get an error
691 # we are out of-sync, which we currently do NOT handle correctly.
692 return $state{handle}->push_read (line => $qr_nlnl, $state{read_response})
693 if $2 eq 100;
694
695 push @pseudo,
696 HTTPVersion => $1,
697 Status => $2,
698 Reason => $3,
699 ;
700
701 my $hdr = parse_hdr
702 or return (%state = (), $cb->(undef, { @pseudo, Status => 599, Reason => "Garbled response headers" }));
703
704 %hdr = (%$hdr, @pseudo);
705 }
706
707 # redirect handling
708 # microsoft and other shitheads don't give a shit for following standards,
709 # try to support some common forms of broken Location headers.
710 if ($hdr{location} !~ /^(?: $ | [^:\/?\#]+ : )/x) {
711 $hdr{location} =~ s/^\.\/+//;
712
713 my $url = "$rscheme://$uhost:$uport";
714
715 unless ($hdr{location} =~ s/^\///) {
716 $url .= $upath;
717 $url =~ s/\/[^\/]*$//;
718 }
719
720 $hdr{location} = "$url/$hdr{location}";
721 }
722
723 my $redirect;
724
725 if ($recurse) {
726 my $status = $hdr{Status};
727
728 # industry standard is to redirect POST as GET for
729 # 301, 302 and 303, in contrast to HTTP/1.0 and 1.1.
730 # also, the UA should ask the user for 301 and 307 and POST,
731 # industry standard seems to be to simply follow.
732 # we go with the industry standard.
733 if ($status == 301 or $status == 302 or $status == 303) {
734 # HTTP/1.1 is unclear on how to mutate the method
735 $method = "GET" unless $method eq "HEAD";
736 $redirect = 1;
737 } elsif ($status == 307) {
738 $redirect = 1;
739 }
740 }
741
742 my $finish = sub { # ($data, $err_status, $err_reason[, $keepalive])
743 my $may_keep_alive = $_[3];
744
745 $state{handle}->destroy if $state{handle};
746 %state = ();
747
748 if (defined $_[1]) {
749 $hdr{OrigStatus} = $hdr{Status}; $hdr{Status} = $_[1];
750 $hdr{OrigReason} = $hdr{Reason}; $hdr{Reason} = $_[2];
751 }
752
753 # set-cookie processing
754 if ($arg{cookie_jar}) {
755 cookie_jar_set_cookie $arg{cookie_jar}, $hdr{"set-cookie"}, $uhost;
756 }
757
758 if ($redirect && exists $hdr{location}) {
759 # we ignore any errors, as it is very common to receive
760 # Content-Length != 0 but no actual body
761 # we also access %hdr, as $_[1] might be an erro
762 http_request (
763 $method => $hdr{location},
764 %arg,
765 recurse => $recurse - 1,
766 Redirect => [$_[0], \%hdr],
767 $cb);
768 } else {
769 $cb->($_[0], \%hdr);
770 }
771 };
772
773 $ae_error = 597; # body phase
774
775 my $len = $hdr{"content-length"};
776
777 if (!$redirect && $arg{on_header} && !$arg{on_header}(\%hdr)) {
778 $finish->(undef, 598 => "Request cancelled by on_header");
779 } elsif (
780 $hdr{Status} =~ /^(?:1..|204|205|304)$/
781 or $method eq "HEAD"
782 or (defined $len && !$len)
783 ) {
784 # no body
785 $finish->("", undef, undef, 1);
786 } else {
787 # body handling, many different code paths
788 # - no body expected
789 # - want_body_handle
790 # - te chunked
791 # - 2x length known (with or without on_body)
792 # - 2x length not known (with or without on_body)
793 if (!$redirect && $arg{want_body_handle}) {
794 $_[0]->on_eof (undef);
795 $_[0]->on_error (undef);
796 $_[0]->on_read (undef);
797
798 $finish->(delete $state{handle});
799
800 } elsif ($hdr{"transfer-encoding"} =~ /\bchunked\b/i) {
801 my $cl = 0;
802 my $body = undef;
803 my $on_body = $arg{on_body} || sub { $body .= shift; 1 };
804
805 my $read_chunk; $read_chunk = sub {
806 $_[1] =~ /^([0-9a-fA-F]+)/
807 or $finish->(undef, $ae_error => "Garbled chunked transfer encoding");
808
809 my $len = hex $1;
810
811 if ($len) {
812 $cl += $len;
813
814 $_[0]->push_read (chunk => $len, sub {
815 $on_body->($_[1], \%hdr)
816 or return $finish->(undef, 598 => "Request cancelled by on_body");
817
818 $_[0]->push_read (line => sub {
819 length $_[1]
820 and return $finish->(undef, $ae_error => "Garbled chunked transfer encoding");
821 $_[0]->push_read (line => $read_chunk);
822 });
823 });
824 } else {
825 $hdr{"content-length"} ||= $cl;
826
827 $_[0]->push_read (line => $qr_nlnl, sub {
828 if (length $_[1]) {
829 for ("$_[1]") {
830 y/\015//d; # weed out any \015, as they show up in the weirdest of places.
831
832 my $hdr = parse_hdr
833 or return $finish->(undef, $ae_error => "Garbled response trailers");
834
835 %hdr = (%hdr, %$hdr);
836 }
837 }
838
839 $finish->($body, undef, undef, 1);
840 });
841 }
842 };
843
844 $_[0]->push_read (line => $read_chunk);
845
846 } elsif ($arg{on_body}) {
847 if ($len) {
848 $_[0]->on_read (sub {
849 $len -= length $_[0]{rbuf};
850
851 $arg{on_body}(delete $_[0]{rbuf}, \%hdr)
852 or return $finish->(undef, 598 => "Request cancelled by on_body");
853
854 $len > 0
855 or $finish->("", undef, undef, 1);
856 });
857 } else {
858 $_[0]->on_eof (sub {
859 $finish->("");
860 });
861 $_[0]->on_read (sub {
862 $arg{on_body}(delete $_[0]{rbuf}, \%hdr)
863 or $finish->(undef, 598 => "Request cancelled by on_body");
864 });
865 }
866 } else {
867 $_[0]->on_eof (undef);
868
869 if ($len) {
870 $_[0]->on_read (sub {
871 $finish->((substr delete $_[0]{rbuf}, 0, $len, ""), undef, undef, 1)
872 if $len <= length $_[0]{rbuf};
873 });
874 } else {
875 $_[0]->on_error (sub {
876 ($! == Errno::EPIPE || !$!)
877 ? $finish->(delete $_[0]{rbuf})
878 : $finish->(undef, $ae_error => $_[2]);
879 });
880 $_[0]->on_read (sub { });
881 }
882 }
883 }
884 };
885
886 $state{handle}->push_read (line => $qr_nlnl, $state{read_response});
887 };
888
889 # now handle proxy-CONNECT method 941 # now handle proxy-CONNECT method
890 if ($proxy && $uscheme eq "https") { 942 if ($proxy && $uscheme eq "https") {
891 # oh dear, we have to wrap it into a connect request 943 # oh dear, we have to wrap it into a connect request
892 944
893 # maybe re-use $uauthority with patched port? 945 # maybe re-use $uauthority with patched port?
894 $state{handle}->push_write ("CONNECT $uhost:$uport HTTP/1.0\015\012Host: $uhost\015\012\015\012"); 946 $state{handle}->push_write ("CONNECT $uhost:$uport HTTP/1.0\015\012\015\012");
895 $state{handle}->push_read (line => $qr_nlnl, sub { 947 $state{handle}->push_read (line => $qr_nlnl, sub {
896 $_[1] =~ /^HTTP\/([0-9\.]+) \s+ ([0-9]{3}) (?: \s+ ([^\015\012]*) )?/ix 948 $_[1] =~ /^HTTP\/([0-9\.]+) \s+ ([0-9]{3}) (?: \s+ ([^\015\012]*) )?/ix
897 or return (%state = (), $cb->(undef, { @pseudo, Status => 599, Reason => "Invalid proxy connect response ($_[1])" })); 949 or return (%state = (), $cb->(undef, { @pseudo, Status => 599, Reason => "Invalid proxy connect response ($_[1])" }));
898 950
899 if ($2 == 200) { 951 if ($2 == 200) {
900 $rpath = $upath; 952 $rpath = $upath;
901 &$handle_actual_request; 953 $handle_actual_request->();
902 } else { 954 } else {
903 %state = (); 955 %state = ();
904 $cb->(undef, { @pseudo, Status => $2, Reason => $3 }); 956 $cb->(undef, { @pseudo, Status => $2, Reason => $3 });
905 }
906 }); 957 }
907 } else {
908 &$handle_actual_request;
909 } 958 });
959 } else {
960 $handle_actual_request->();
910 }; 961 }
962 };
963
964 _get_slot $uhost, sub {
965 $state{slot_guard} = shift;
966
967 return unless $state{connect_guard};
911 968
912 my $tcp_connect = $arg{tcp_connect} 969 my $tcp_connect = $arg{tcp_connect}
913 || do { require AnyEvent::Socket; \&AnyEvent::Socket::tcp_connect }; 970 || do { require AnyEvent::Socket; \&AnyEvent::Socket::tcp_connect };
914 971
915 $state{connect_guard} = $tcp_connect->($rhost, $rport, $connect_cb, $arg{on_prepare} || sub { $timeout }); 972 $state{connect_guard} = $tcp_connect->($rhost, $rport, $connect_cb, $arg{on_prepare} || sub { $timeout });
916
917 }; 973 };
918 974
919 defined wantarray && AnyEvent::Util::guard { %state = () } 975 defined wantarray && AnyEvent::Util::guard { %state = () }
920} 976}
921 977
956string of the form C<http://host:port> (optionally C<https:...>), croaks 1012string of the form C<http://host:port> (optionally C<https:...>), croaks
957otherwise. 1013otherwise.
958 1014
959To clear an already-set proxy, use C<undef>. 1015To clear an already-set proxy, use C<undef>.
960 1016
1017=item AnyEvent::HTTP::cookie_jar_expire $jar[, $session_end]
1018
1019Remove all cookies from the cookie jar that have been expired. If
1020C<$session_end> is given and true, then additionally remove all session
1021cookies.
1022
1023You should call this function (with a true C<$session_end>) before you
1024save cookies to disk, and you should call this function after loading them
1025again. If you have a long-running program you can additonally call this
1026function from time to time.
1027
1028A cookie jar is initially an empty hash-reference that is managed by this
1029module. It's format is subject to change, but currently it is like this:
1030
1031The key C<version> has to contain C<1>, otherwise the hash gets
1032emptied. All other keys are hostnames or IP addresses pointing to
1033hash-references. The key for these inner hash references is the
1034server path for which this cookie is meant, and the values are again
1035hash-references. The keys of those hash-references is the cookie name, and
1036the value, you guessed it, is another hash-reference, this time with the
1037key-value pairs from the cookie, except for C<expires> and C<max-age>,
1038which have been replaced by a C<_expires> key that contains the cookie
1039expiry timestamp.
1040
1041Here is an example of a cookie jar with a single cookie, so you have a
1042chance of understanding the above paragraph:
1043
1044 {
1045 version => 1,
1046 "10.0.0.1" => {
1047 "/" => {
1048 "mythweb_id" => {
1049 _expires => 1293917923,
1050 value => "ooRung9dThee3ooyXooM1Ohm",
1051 },
1052 },
1053 },
1054 }
1055
961=item $date = AnyEvent::HTTP::format_date $timestamp 1056=item $date = AnyEvent::HTTP::format_date $timestamp
962 1057
963Takes a POSIX timestamp (seconds since the epoch) and formats it as a HTTP 1058Takes a POSIX timestamp (seconds since the epoch) and formats it as a HTTP
964Date (RFC 2616). 1059Date (RFC 2616).
965 1060
966=item $timestamp = AnyEvent::HTTP::parse_date $date 1061=item $timestamp = AnyEvent::HTTP::parse_date $date
967 1062
968Takes a HTTP Date (RFC 2616) or a Cookie date (netscape cookie spec) and 1063Takes a HTTP Date (RFC 2616) or a Cookie date (netscape cookie spec) or a
969returns the corresponding POSIX timestamp, or C<undef> if the date cannot 1064bunch of minor variations of those, and returns the corresponding POSIX
970be parsed. 1065timestamp, or C<undef> if the date cannot be parsed.
971 1066
972=item $AnyEvent::HTTP::MAX_RECURSE 1067=item $AnyEvent::HTTP::MAX_RECURSE
973 1068
974The default value for the C<recurse> request parameter (default: C<10>). 1069The default value for the C<recurse> request parameter (default: C<10>).
975 1070
1014sub parse_date($) { 1109sub parse_date($) {
1015 my ($date) = @_; 1110 my ($date) = @_;
1016 1111
1017 my ($d, $m, $y, $H, $M, $S); 1112 my ($d, $m, $y, $H, $M, $S);
1018 1113
1019 if ($date =~ /^[A-Z][a-z][a-z], ([0-9][0-9])[\- ]([A-Z][a-z][a-z])[\- ]([0-9][0-9][0-9][0-9]) ([0-9][0-9]):([0-9][0-9]):([0-9][0-9]) GMT$/) { 1114 if ($date =~ /^[A-Z][a-z][a-z]+, ([0-9][0-9]?)[\- ]([A-Z][a-z][a-z])[\- ]([0-9][0-9][0-9][0-9]) ([0-9][0-9]?):([0-9][0-9]?):([0-9][0-9]?) GMT$/) {
1020 # RFC 822/1123, required by RFC 2616 (with " ") 1115 # RFC 822/1123, required by RFC 2616 (with " ")
1021 # cookie dates (with "-") 1116 # cookie dates (with "-")
1022 1117
1023 ($d, $m, $y, $H, $M, $S) = ($1, $2, $3, $4, $5, $6); 1118 ($d, $m, $y, $H, $M, $S) = ($1, $2, $3, $4, $5, $6);
1024 1119
1025 } elsif ($date =~ /^[A-Z][a-z]+, ([0-9][0-9])-([A-Z][a-z][a-z])-([0-9][0-9]) ([0-9][0-9]):([0-9][0-9]):([0-9][0-9]) GMT$/) { 1120 } elsif ($date =~ /^[A-Z][a-z][a-z]+, ([0-9][0-9]?)-([A-Z][a-z][a-z])-([0-9][0-9]) ([0-9][0-9]?):([0-9][0-9]?):([0-9][0-9]?) GMT$/) {
1026 # RFC 850 1121 # RFC 850
1027 ($d, $m, $y, $H, $M, $S) = ($1, $2, $3 < 69 ? $3 + 2000 : $3 + 1900, $4, $5, $6); 1122 ($d, $m, $y, $H, $M, $S) = ($1, $2, $3 < 69 ? $3 + 2000 : $3 + 1900, $4, $5, $6);
1028 1123
1029 } elsif ($date =~ /^[A-Z][a-z][a-z] ([A-Z][a-z][a-z]) ([0-9 ][0-9]) ([0-9][0-9]):([0-9][0-9]):([0-9][0-9]) ([0-9][0-9][0-9][0-9])$/) { 1124 } elsif ($date =~ /^[A-Z][a-z][a-z]+ ([A-Z][a-z][a-z]) ([0-9 ]?[0-9]) ([0-9][0-9]?):([0-9][0-9]?):([0-9][0-9]?) ([0-9][0-9][0-9][0-9])$/) {
1030 # ISO C's asctime 1125 # ISO C's asctime
1031 ($d, $m, $y, $H, $M, $S) = ($2, $1, $6, $3, $4, $5); 1126 ($d, $m, $y, $H, $M, $S) = ($2, $1, $6, $3, $4, $5);
1032 } 1127 }
1033 # other formats fail in the loop below 1128 # other formats fail in the loop below
1034 1129

Diff Legend

Removed lines
+ Added lines
< Changed lines
> Changed lines